[yt-dlp.git] / youtube_dl / extractor / disney.py

# coding: utf-8
from __future__ import unicode_literals

import re

from .common import InfoExtractor
from ..utils import (
    int_or_none,
    unified_strdate,
    compat_str,
    determine_ext,
    ExtractorError,
)


class DisneyIE(InfoExtractor):
    _VALID_URL = r'''(?x)
        https?://(?P<domain>(?:[^/]+\.)?(?:disney\.[a-z]{2,3}(?:\.[a-z]{2})?|disney(?:(?:me|latino)\.com|turkiye\.com\.tr)|(?:starwars|marvelkids)\.com))/(?:(?:embed/|(?:[^/]+/)+[\w-]+-)(?P<id>[a-z0-9]{24})|(?:[^/]+/)?(?P<display_id>[^/?#]+))'''
    _TESTS = [{
        # Disney.EmbedVideo
        'url': 'http://video.disney.com/watch/moana-trailer-545ed1857afee5a0ec239977',
        'info_dict': {
            'id': '545ed1857afee5a0ec239977',
            'ext': 'mp4',
            'title': 'Moana - Trailer',
            'description': 'A fun adventure for the entire Family!  Bring home Moana on Digital HD Feb 21 & Blu-ray March 7',
            'upload_date': '20170112',
        },
        'params': {
            # m3u8 download
            'skip_download': True,
        }
    }, {
        # Grill.burger
        'url': 'http://www.starwars.com/video/rogue-one-a-star-wars-story-intro-featurette',
        'info_dict': {
            'id': '5454e9f4e9804a552e3524c8',
            'ext': 'mp4',
            'title': '"Intro" Featurette: Rogue One: A Star Wars Story',
            'upload_date': '20170104',
            'description': 'Go behind-the-scenes of Rogue One: A Star Wars Story in this featurette with Director Gareth Edwards and the cast of the film.',
        },
        'params': {
            # m3u8 download
            'skip_download': True,
        }
    }, {
        'url': 'http://videos.disneylatino.com/ver/spider-man-de-regreso-a-casa-primer-adelanto-543a33a1850bdcfcca13bae2',
        'only_matching': True,
    }, {
        'url': 'http://video.en.disneyme.com/watch/future-worm/robo-carp-2001-544b66002aa7353cdd3f5114',
        'only_matching': True,
    }, {
        'url': 'http://video.disneyturkiye.com.tr/izle/7c-7-cuceler/kimin-sesi-zaten-5456f3d015f6b36c8afdd0e2',
        'only_matching': True,
    }, {
        'url': 'http://disneyjunior.disney.com/embed/546a4798ddba3d1612e4005d',
        'only_matching': True,
    }, {
        'url': 'http://www.starwars.com/embed/54690d1e6c42e5f09a0fb097',
        'only_matching': True,
    }, {
        'url': 'http://spiderman.marvelkids.com/embed/522900d2ced3c565e4cc0677',
        'only_matching': True,
    }, {
        'url': 'http://spiderman.marvelkids.com/videos/contest-of-champions-part-four-clip-1',
        'only_matching': True,
    }, {
        'url': 'http://disneyjunior.en.disneyme.com/dj/watch-my-friends-tigger-and-pooh-promo',
        'only_matching': True,
    }, {
        'url': 'http://disneyjunior.disney.com/galactech-the-galactech-grab-galactech-an-admiral-rescue',
        'only_matching': True,
    }]

    def _real_extract(self, url):
        domain, video_id, display_id = re.match(self._VALID_URL, url).groups()
        if not video_id:
            webpage = self._download_webpage(url, display_id)
            grill = re.sub(r'"\s*\+\s*"', '', self._search_regex(
                r'Grill\.burger\s*=\s*({.+})\s*:',
                webpage, 'grill data'))
            page_data = next(s for s in self._parse_json(grill, display_id)['stack'] if s.get('type') == 'video')
            video_data = page_data['data'][0]
        else:
            webpage = self._download_webpage(
                'http://%s/embed/%s' % (domain, video_id), video_id)
            page_data = self._parse_json(self._search_regex(
                r'Disney\.EmbedVideo\s*=\s*({.+});',
                webpage, 'embed data'), video_id)
            video_data = page_data['video']

        for external in video_data.get('externals', []):
            if external.get('source') == 'vevo':
                return self.url_result('vevo:' + external['data_id'], 'Vevo')

        video_id = video_data['id']
        title = video_data['title']

        formats = []
        for flavor in video_data.get('flavors', []):
            flavor_format = flavor.get('format')
            flavor_url = flavor.get('url')
            if not flavor_url or not re.match(r'https?://', flavor_url) or flavor_format == 'mp4_access':
                continue
            tbr = int_or_none(flavor.get('bitrate'))
            if tbr == 99999:
                formats.extend(self._extract_m3u8_formats(
                    flavor_url, video_id, 'mp4',
                    m3u8_id=flavor_format, fatal=False))
                continue
            format_id = []
            if flavor_format:
                format_id.append(flavor_format)
            if tbr:
                format_id.append(compat_str(tbr))
            ext = determine_ext(flavor_url)
            if flavor_format == 'applehttp' or ext == 'm3u8':
                ext = 'mp4'
            width = int_or_none(flavor.get('width'))
            height = int_or_none(flavor.get('height'))
            formats.append({
                'format_id': '-'.join(format_id),
                'url': flavor_url,
                'width': width,
                'height': height,
                'tbr': tbr,
                'ext': ext,
                'vcodec': 'none' if (width == 0 and height == 0) else None,
            })
        if not formats and video_data.get('expired'):
            raise ExtractorError(
                '%s said: %s' % (self.IE_NAME, page_data['translations']['video_expired']),
                expected=True)
        self._sort_formats(formats)

        subtitles = {}
        for caption in video_data.get('captions', []):
            caption_url = caption.get('url')
            caption_format = caption.get('format')
            if not caption_url or caption_format.startswith('unknown'):
                continue
            subtitles.setdefault(caption.get('language', 'en'), []).append({
                'url': caption_url,
                'ext': {
                    'webvtt': 'vtt',
                }.get(caption_format, caption_format),
            })

        return {
            'id': video_id,
            'title': title,
            'description': video_data.get('description') or video_data.get('short_desc'),
            'thumbnail': video_data.get('thumb') or video_data.get('thumb_secure'),
            'duration': int_or_none(video_data.get('duration_sec')),
            'upload_date': unified_strdate(video_data.get('publish_date')),
            'formats': formats,
            'subtitles': subtitles,
        }
Commit	Line	Data
b3277115 RA	1	# coding: utf-8
	2	from __future__ import unicode_literals
	3
	4	import re
	5
	6	from .common import InfoExtractor
	7	from ..utils import (
	8	int_or_none,
	9	unified_strdate,
	10	compat_str,
	11	determine_ext,
9dad9418	12	ExtractorError,
b3277115 RA	13	)
	14
	15
	16	class DisneyIE(InfoExtractor):
	17	_VALID_URL = r'''(?x)
9dad9418	18	https?://(?P<domain>(?:[^/]+\.)?(?:disney\.[a-z]{2,3}(?:\.[a-z]{2})?\|disney(?:(?:me\|latino)\.com\|turkiye\.com\.tr)\|(?:starwars\|marvelkids)\.com))/(?:(?:embed/\|(?:[^/]+/)+[\w-]+-)(?P<id>[a-z0-9]{24})\|(?:[^/]+/)?(?P<display_id>[^/?#]+))'''
b3277115	19	_TESTS = [{
9dad9418	20	# Disney.EmbedVideo
b3277115 RA	21	'url': 'http://video.disney.com/watch/moana-trailer-545ed1857afee5a0ec239977',
	22	'info_dict': {
	23	'id': '545ed1857afee5a0ec239977',
	24	'ext': 'mp4',
	25	'title': 'Moana - Trailer',
	26	'description': 'A fun adventure for the entire Family! Bring home Moana on Digital HD Feb 21 & Blu-ray March 7',
	27	'upload_date': '20170112',
	28	},
	29	'params': {
	30	# m3u8 download
	31	'skip_download': True,
	32	}
9dad9418 RA	33	}, {
	34	# Grill.burger
	35	'url': 'http://www.starwars.com/video/rogue-one-a-star-wars-story-intro-featurette',
	36	'info_dict': {
	37	'id': '5454e9f4e9804a552e3524c8',
	38	'ext': 'mp4',
	39	'title': '"Intro" Featurette: Rogue One: A Star Wars Story',
	40	'upload_date': '20170104',
	41	'description': 'Go behind-the-scenes of Rogue One: A Star Wars Story in this featurette with Director Gareth Edwards and the cast of the film.',
	42	},
	43	'params': {
	44	# m3u8 download
	45	'skip_download': True,
	46	}
b3277115 RA	47	}, {
	48	'url': 'http://videos.disneylatino.com/ver/spider-man-de-regreso-a-casa-primer-adelanto-543a33a1850bdcfcca13bae2',
	49	'only_matching': True,
	50	}, {
	51	'url': 'http://video.en.disneyme.com/watch/future-worm/robo-carp-2001-544b66002aa7353cdd3f5114',
	52	'only_matching': True,
	53	}, {
	54	'url': 'http://video.disneyturkiye.com.tr/izle/7c-7-cuceler/kimin-sesi-zaten-5456f3d015f6b36c8afdd0e2',
	55	'only_matching': True,
	56	}, {
	57	'url': 'http://disneyjunior.disney.com/embed/546a4798ddba3d1612e4005d',
	58	'only_matching': True,
	59	}, {
	60	'url': 'http://www.starwars.com/embed/54690d1e6c42e5f09a0fb097',
	61	'only_matching': True,
9dad9418 RA	62	}, {
	63	'url': 'http://spiderman.marvelkids.com/embed/522900d2ced3c565e4cc0677',
	64	'only_matching': True,
	65	}, {
	66	'url': 'http://spiderman.marvelkids.com/videos/contest-of-champions-part-four-clip-1',
	67	'only_matching': True,
	68	}, {
	69	'url': 'http://disneyjunior.en.disneyme.com/dj/watch-my-friends-tigger-and-pooh-promo',
	70	'only_matching': True,
	71	}, {
	72	'url': 'http://disneyjunior.disney.com/galactech-the-galactech-grab-galactech-an-admiral-rescue',
	73	'only_matching': True,
b3277115 RA	74	}]
	75
	76	def _real_extract(self, url):
9dad9418 RA	77	domain, video_id, display_id = re.match(self._VALID_URL, url).groups()
	78	if not video_id:
	79	webpage = self._download_webpage(url, display_id)
	80	grill = re.sub(r'"\s\+\s"', '', self._search_regex(
	81	r'Grill\.burger\s=\s({.+})\s*:',
	82	webpage, 'grill data'))
	83	page_data = next(s for s in self._parse_json(grill, display_id)['stack'] if s.get('type') == 'video')
	84	video_data = page_data['data'][0]
	85	else:
	86	webpage = self._download_webpage(
	87	'http://%s/embed/%s' % (domain, video_id), video_id)
	88	page_data = self._parse_json(self._search_regex(
	89	r'Disney\.EmbedVideo\s=\s({.+});',
	90	webpage, 'embed data'), video_id)
	91	video_data = page_data['video']
b3277115 RA	92
	93	for external in video_data.get('externals', []):
	94	if external.get('source') == 'vevo':
	95	return self.url_result('vevo:' + external['data_id'], 'Vevo')
	96
9dad9418	97	video_id = video_data['id']
b3277115 RA	98	title = video_data['title']
	99
	100	formats = []
	101	for flavor in video_data.get('flavors', []):
	102	flavor_format = flavor.get('format')
	103	flavor_url = flavor.get('url')
9dad9418	104	if not flavor_url or not re.match(r'https?://', flavor_url) or flavor_format == 'mp4_access':
b3277115 RA	105	continue
	106	tbr = int_or_none(flavor.get('bitrate'))
	107	if tbr == 99999:
	108	formats.extend(self._extract_m3u8_formats(
9dad9418 RA	109	flavor_url, video_id, 'mp4',
9dad9418 RA	110	m3u8_id=flavor_format, fatal=False))
b3277115 RA	111	continue
	112	format_id = []
	113	if flavor_format:
	114	format_id.append(flavor_format)
	115	if tbr:
	116	format_id.append(compat_str(tbr))
	117	ext = determine_ext(flavor_url)
	118	if flavor_format == 'applehttp' or ext == 'm3u8':
	119	ext = 'mp4'
	120	width = int_or_none(flavor.get('width'))
	121	height = int_or_none(flavor.get('height'))
	122	formats.append({
	123	'format_id': '-'.join(format_id),
	124	'url': flavor_url,
	125	'width': width,
	126	'height': height,
	127	'tbr': tbr,
	128	'ext': ext,
	129	'vcodec': 'none' if (width == 0 and height == 0) else None,
	130	})
9dad9418 RA	131	if not formats and video_data.get('expired'):
	132	raise ExtractorError(
	133	'%s said: %s' % (self.IE_NAME, page_data['translations']['video_expired']),
	134	expected=True)
b3277115 RA	135	self._sort_formats(formats)
	136
	137	subtitles = {}
	138	for caption in video_data.get('captions', []):
	139	caption_url = caption.get('url')
	140	caption_format = caption.get('format')
	141	if not caption_url or caption_format.startswith('unknown'):
	142	continue
	143	subtitles.setdefault(caption.get('language', 'en'), []).append({
	144	'url': caption_url,
	145	'ext': {
	146	'webvtt': 'vtt',
	147	}.get(caption_format, caption_format),
	148	})
	149
	150	return {
	151	'id': video_id,
	152	'title': title,
	153	'description': video_data.get('description') or video_data.get('short_desc'),
	154	'thumbnail': video_data.get('thumb') or video_data.get('thumb_secure'),
	155	'duration': int_or_none(video_data.get('duration_sec')),
	156	'upload_date': unified_strdate(video_data.get('publish_date')),
	157	'formats': formats,
	158	'subtitles': subtitles,
	159	}