[yt-dlp.git] / youtube_dl / extractor / togglesg.py

# coding: utf-8
from __future__ import unicode_literals

import json
import re

from .common import InfoExtractor
from ..utils import (
    determine_ext,
    ExtractorError,
    int_or_none,
    parse_iso8601,
    sanitized_Request,
)


class ToggleSgIE(InfoExtractor):
    IE_NAME = 'togglesg'
    _VALID_URL = r'https?://video\.toggle\.sg/(?:en|zh)/(?:series|clips|movies)/.+?/(?P<id>[0-9]+)'
    _TESTS = [{
        'url': 'http://video.toggle.sg/en/series/lion-moms-tif/trailers/lion-moms-premier/343115',
        'info_dict': {
            'id': '343115',
            'ext': 'mp4',
            'title': 'Lion Moms Premiere',
            'description': 'md5:aea1149404bff4d7f7b6da11fafd8e6b',
            'upload_date': '20150910',
            'timestamp': 1441858274,
        },
        'params': {
            'skip_download': 'm3u8 download',
        }
    }, {
        'note': 'DRM-protected video',
        'url': 'http://video.toggle.sg/en/movies/dug-s-special-mission/341413',
        'info_dict': {
            'id': '341413',
            'ext': 'wvm',
            'title': 'Dug\'s Special Mission',
            'description': 'md5:e86c6f4458214905c1772398fabc93e0',
            'upload_date': '20150827',
            'timestamp': 1440644006,
        },
        'params': {
            'skip_download': 'DRM-protected wvm download',
        }
    }, {
        'note': 'm3u8 links are geo-restricted, but Android/mp4 is okay',
        'url': 'http://video.toggle.sg/en/series/28th-sea-games-5-show/ep11/332861',
        'info_dict': {
            'id': '332861',
            'ext': 'mp4',
            'title': '28th SEA Games (5 Show) -  Episode  11',
            'description': 'md5:3cd4f5f56c7c3b1340c50a863f896faa',
            'upload_date': '20150605',
            'timestamp': 1433480166,
        },
        'params': {
            'skip_download': 'DRM-protected wvm download',
        },
        'skip': 'm3u8 links are geo-restricted'
    }, {
        'url': 'http://video.toggle.sg/en/clips/seraph-sun-aloysius-will-suddenly-sing-some-old-songs-in-high-pitch-on-set/343331',
        'only_matching': True,
    }, {
        'url': 'http://video.toggle.sg/zh/series/zero-calling-s2-hd/ep13/336367',
        'only_matching': True,
    }, {
        'url': 'http://video.toggle.sg/en/series/vetri-s2/webisodes/jeeva-is-an-orphan-vetri-s2-webisode-7/342302',
        'only_matching': True,
    }, {
        'url': 'http://video.toggle.sg/en/movies/seven-days/321936',
        'only_matching': True,
    }]

    _FORMAT_PREFERENCES = {
        'wvm-STBMain': -10,
        'wvm-iPadMain': -20,
        'wvm-iPhoneMain': -30,
        'wvm-Android': -40,
    }
    _API_USER = 'tvpapi_147'
    _API_PASS = '11111'

    def _real_extract(self, url):
        video_id = self._match_id(url)

        webpage = self._download_webpage(
            url, video_id, note='Downloading video page')

        api_user = self._search_regex(
            r'apiUser\s*:\s*(["\'])(?P<user>.+?)\1', webpage, 'apiUser',
            default=self._API_USER, group='user')
        api_pass = self._search_regex(
            r'apiPass\s*:\s*(["\'])(?P<pass>.+?)\1', webpage, 'apiPass',
            default=self._API_PASS, group='pass')

        params = {
            'initObj': {
                'Locale': {
                    'LocaleLanguage': '',
                    'LocaleCountry': '',
                    'LocaleDevice': '',
                    'LocaleUserState': 0
                },
                'Platform': 0,
                'SiteGuid': 0,
                'DomainID': '0',
                'UDID': '',
                'ApiUser': api_user,
                'ApiPass': api_pass
            },
            'MediaID': video_id,
            'mediaType': 0,
        }

        req = sanitized_Request(
            'http://tvpapi.as.tvinci.com/v2_9/gateways/jsonpostgw.aspx?m=GetMediaInfo',
            json.dumps(params).encode('utf-8'))
        info = self._download_json(req, video_id, 'Downloading video info json')

        title = info['MediaName']

        formats = []
        for video_file in info.get('Files', []):
            ext = determine_ext(video_file['URL'])
            vid_format = video_file['Format'].replace(' ', '')
            # if geo-restricted, m3u8 is inaccessible, but mp4 is okay
            if ext == 'm3u8':
                m3u8_formats = self._extract_m3u8_formats(
                    video_file['URL'], video_id, ext='mp4', m3u8_id=vid_format,
                    note='Downloading %s m3u8 information' % vid_format,
                    errnote='Failed to download %s m3u8 information' % vid_format,
                    fatal=False)
                if m3u8_formats:
                    formats.extend(m3u8_formats)
            elif ext in ('mp4', 'wvm'):
                # wvm are drm-protected files
                formats.append({
                    'ext': ext,
                    'url': video_file['URL'],
                    'format_id': vid_format,
                    'preference': self._FORMAT_PREFERENCES.get(ext + '-' + vid_format) or -1,
                    'format_note': 'DRM-protected video' if ext == 'wvm' else None
                })
        if not formats:
            # Most likely because geo-blocked
            raise ExtractorError('No downloadable videos found', expected=True)
        self._sort_formats(formats)

        duration = int_or_none(info.get('Duration'))
        description = info.get('Description')
        created_at = parse_iso8601(info.get('CreationDate') or None)

        thumbnails = []
        for picture in info.get('Pictures', []):
            if not isinstance(picture, dict):
                continue
            pic_url = picture.get('URL')
            if not pic_url:
                continue
            thumbnail = {
                'url': pic_url,
            }
            pic_size = picture.get('PicSize', '')
            m = re.search(r'(?P<width>\d+)[xX](?P<height>\d+)', pic_size)
            if m:
                thumbnail.update({
                    'width': int(m.group('width')),
                    'height': int(m.group('height')),
                })
            thumbnails.append(thumbnail)

        return {
            'id': video_id,
            'title': title,
            'description': description,
            'duration': duration,
            'timestamp': created_at,
            'thumbnails': thumbnails,
            'formats': formats,
        }
Commit	Line	Data
ee0f0393	1	# coding: utf-8
	2	from __future__ import unicode_literals
	3
	4	import json
c40dbb19	5	import re
ee0f0393	6
	7	from .common import InfoExtractor
	8	from ..utils import (
c82a8dd1	9	determine_ext,
ee0f0393	10	ExtractorError,
ee0f0393	11	int_or_none,
ee0f0393	12	parse_iso8601,
f8253af5	13	sanitized_Request,
ee0f0393	14	)
ee0f0393	15
	16
	17	class ToggleSgIE(InfoExtractor):
	18	IE_NAME = 'togglesg'
ed370ff0	19	_VALID_URL = r'https?://video\.toggle\.sg/(?:en\|zh)/(?:series\|clips\|movies)/.+?/(?P<id>[0-9]+)'
ee0f0393	20	_TESTS = [{
	21	'url': 'http://video.toggle.sg/en/series/lion-moms-tif/trailers/lion-moms-premier/343115',
	22	'info_dict': {
	23	'id': '343115',
	24	'ext': 'mp4',
	25	'title': 'Lion Moms Premiere',
	26	'description': 'md5:aea1149404bff4d7f7b6da11fafd8e6b',
	27	'upload_date': '20150910',
	28	'timestamp': 1441858274,
	29	},
	30	'params': {
	31	'skip_download': 'm3u8 download',
	32	}
	33	}, {
	34	'note': 'DRM-protected video',
	35	'url': 'http://video.toggle.sg/en/movies/dug-s-special-mission/341413',
	36	'info_dict': {
	37	'id': '341413',
	38	'ext': 'wvm',
	39	'title': 'Dug\'s Special Mission',
	40	'description': 'md5:e86c6f4458214905c1772398fabc93e0',
	41	'upload_date': '20150827',
	42	'timestamp': 1440644006,
	43	},
	44	'params': {
	45	'skip_download': 'DRM-protected wvm download',
	46	}
	47	}, {
	48	'note': 'm3u8 links are geo-restricted, but Android/mp4 is okay',
	49	'url': 'http://video.toggle.sg/en/series/28th-sea-games-5-show/ep11/332861',
	50	'info_dict': {
	51	'id': '332861',
	52	'ext': 'mp4',
	53	'title': '28th SEA Games (5 Show) - Episode 11',
	54	'description': 'md5:3cd4f5f56c7c3b1340c50a863f896faa',
	55	'upload_date': '20150605',
	56	'timestamp': 1433480166,
	57	},
	58	'params': {
	59	'skip_download': 'DRM-protected wvm download',
	60	},
	61	'skip': 'm3u8 links are geo-restricted'
	62	}, {
	63	'url': 'http://video.toggle.sg/en/clips/seraph-sun-aloysius-will-suddenly-sing-some-old-songs-in-high-pitch-on-set/343331',
	64	'only_matching': True,
	65	}, {
	66	'url': 'http://video.toggle.sg/zh/series/zero-calling-s2-hd/ep13/336367',
	67	'only_matching': True,
	68	}, {
	69	'url': 'http://video.toggle.sg/en/series/vetri-s2/webisodes/jeeva-is-an-orphan-vetri-s2-webisode-7/342302',
	70	'only_matching': True,
	71	}, {
	72	'url': 'http://video.toggle.sg/en/movies/seven-days/321936',
	73	'only_matching': True,
	74	}]
	75
	76	_FORMAT_PREFERENCES = {
	77	'wvm-STBMain': -10,
	78	'wvm-iPadMain': -20,
	79	'wvm-iPhoneMain': -30,
	80	'wvm-Android': -40,
	81	}
	82	_API_USER = 'tvpapi_147'
	83	_API_PASS = '11111'
84
85	def _real_extract(self, url):
86	video_id = self._match_id(url)
87
ffaf6e66 S	88	webpage = self._download_webpage(
ffaf6e66 S	89	url, video_id, note='Downloading video page')
ee0f0393	90
ee0f0393	91	api_user = self._search_regex(
ffaf6e66 S	92	r'apiUser\s:\s(["\'])(?P<user>.+?)\1', webpage, 'apiUser',
ffaf6e66 S	93	default=self._API_USER, group='user')
ee0f0393	94	api_pass = self._search_regex(
ffaf6e66 S	95	r'apiPass\s:\s(["\'])(?P<pass>.+?)\1', webpage, 'apiPass',
ffaf6e66 S	96	default=self._API_PASS, group='pass')
ee0f0393	97
	98	params = {
	99	'initObj': {
	100	'Locale': {
74c73017 S	101	'LocaleLanguage': '',
	102	'LocaleCountry': '',
	103	'LocaleDevice': '',
	104	'LocaleUserState': 0
ee0f0393	105	},
74c73017 S	106	'Platform': 0,
	107	'SiteGuid': 0,
	108	'DomainID': '0',
	109	'UDID': '',
	110	'ApiUser': api_user,
	111	'ApiPass': api_pass
ee0f0393	112	},
	113	'MediaID': video_id,
	114	'mediaType': 0,
	115	}
	116
f8253af5	117	req = sanitized_Request(
ee0f0393	118	'http://tvpapi.as.tvinci.com/v2_9/gateways/jsonpostgw.aspx?m=GetMediaInfo',
	119	json.dumps(params).encode('utf-8'))
	120	info = self._download_json(req, video_id, 'Downloading video info json')
	121
	122	title = info['MediaName']
ee0f0393	123
c40dbb19	124	formats = []
ee0f0393	125	for video_file in info.get('Files', []):
	126	ext = determine_ext(video_file['URL'])
	127	vid_format = video_file['Format'].replace(' ', '')
	128	# if geo-restricted, m3u8 is inaccessible, but mp4 is okay
	129	if ext == 'm3u8':
	130	m3u8_formats = self._extract_m3u8_formats(
	131	video_file['URL'], video_id, ext='mp4', m3u8_id=vid_format,
	132	note='Downloading %s m3u8 information' % vid_format,
	133	errnote='Failed to download %s m3u8 information' % vid_format,
ffaf6e66	134	fatal=False)
ee0f0393	135	if m3u8_formats:
ee0f0393	136	formats.extend(m3u8_formats)
ffaf6e66	137	elif ext in ('mp4', 'wvm'):
ee0f0393	138	# wvm are drm-protected files
	139	formats.append({
	140	'ext': ext,
	141	'url': video_file['URL'],
	142	'format_id': vid_format,
	143	'preference': self._FORMAT_PREFERENCES.get(ext + '-' + vid_format) or -1,
	144	'format_note': 'DRM-protected video' if ext == 'wvm' else None
	145	})
ee0f0393	146	if not formats:
	147	# Most likely because geo-blocked
	148	raise ExtractorError('No downloadable videos found', expected=True)
ee0f0393	149	self._sort_formats(formats)
ee0f0393	150
c40dbb19 S	151	duration = int_or_none(info.get('Duration'))
	152	description = info.get('Description')
	153	created_at = parse_iso8601(info.get('CreationDate') or None)
	154
	155	thumbnails = []
	156	for picture in info.get('Pictures', []):
	157	if not isinstance(picture, dict):
	158	continue
	159	pic_url = picture.get('URL')
	160	if not pic_url:
	161	continue
	162	thumbnail = {
	163	'url': pic_url,
	164	}
	165	pic_size = picture.get('PicSize', '')
	166	m = re.search(r'(?P<width>\d+)[xX](?P<height>\d+)', pic_size)
	167	if m:
	168	thumbnail.update({
	169	'width': int(m.group('width')),
	170	'height': int(m.group('height')),
	171	})
	172	thumbnails.append(thumbnail)
	173
ee0f0393	174	return {
	175	'id': video_id,
	176	'title': title,
	177	'description': description,
	178	'duration': duration,
	179	'timestamp': created_at,
c40dbb19	180	'thumbnails': thumbnails,
ee0f0393	181	'formats': formats,
ee0f0393	182	}