[yt-dlp.git] / youtube_dl / extractor / cctv.py

# coding: utf-8
from __future__ import unicode_literals

import re

from .common import InfoExtractor
from ..compat import compat_str
from ..utils import (
    float_or_none,
    try_get,
    unified_timestamp,
)


class CCTVIE(InfoExtractor):
    IE_DESC = '央视网'
    _VALID_URL = r'https?://(?:[^/]+)\.(?:cntv|cctv)\.(?:com|cn)/(?:[^/]+/)*?(?P<id>[^/?#&]+?)(?:/index)?(?:\.s?html|[?#&]|$)'
    _TESTS = [{
        'url': 'http://sports.cntv.cn/2016/02/12/ARTIaBRxv4rTT1yWf1frW2wi160212.shtml',
        'md5': 'd61ec00a493e09da810bf406a078f691',
        'info_dict': {
            'id': '5ecdbeab623f4973b40ff25f18b174e8',
            'ext': 'mp4',
            'title': '[NBA]二少联手砍下46分 雷霆主场击败鹈鹕（快讯）',
            'description': 'md5:7e14a5328dc5eb3d1cd6afbbe0574e95',
            'duration': 98,
            'uploader': 'songjunjie',
            'timestamp': 1455279956,
            'upload_date': '20160212',
        },
    }, {
        'url': 'http://tv.cctv.com/2016/02/05/VIDEUS7apq3lKrHG9Dncm03B160205.shtml',
        'info_dict': {
            'id': 'efc5d49e5b3b4ab2b34f3a502b73d3ae',
            'ext': 'mp4',
            'title': '[赛车]“车王”舒马赫恢复情况成谜（快讯）',
            'description': '2月4日，蒙特泽莫罗透露了关于“车王”舒马赫恢复情况，但情况是否属实遭到了质疑。',
            'duration': 37,
            'uploader': 'shujun',
            'timestamp': 1454677291,
            'upload_date': '20160205',
        },
        'params': {
            'skip_download': True,
        },
    }, {
        'url': 'http://english.cntv.cn/special/four_comprehensives/index.shtml',
        'info_dict': {
            'id': '4bb9bb4db7a6471ba85fdeda5af0381e',
            'ext': 'mp4',
            'title': 'NHnews008 ANNUAL POLITICAL SEASON',
            'description': 'Four Comprehensives',
            'duration': 60,
            'uploader': 'zhangyunlei',
            'timestamp': 1425385521,
            'upload_date': '20150303',
        },
        'params': {
            'skip_download': True,
        },
    }, {
        'url': 'http://cctv.cntv.cn/lm/tvseries_russian/yilugesanghua/index.shtml',
        'info_dict': {
            'id': 'b15f009ff45c43968b9af583fc2e04b2',
            'ext': 'mp4',
            'title': 'Путь，усыпанный космеями Серия 1',
            'description': 'Путь, усыпанный космеями',
            'duration': 2645,
            'uploader': 'renxue',
            'timestamp': 1477479241,
            'upload_date': '20161026',
        },
        'params': {
            'skip_download': True,
        },
    }, {
        'url': 'http://ent.cntv.cn/2016/01/18/ARTIjprSSJH8DryTVr5Bx8Wb160118.shtml',
        'only_matching': True,
    }, {
        'url': 'http://tv.cntv.cn/video/C39296/e0210d949f113ddfb38d31f00a4e5c44',
        'only_matching': True,
    }, {
        'url': 'http://english.cntv.cn/2016/09/03/VIDEhnkB5y9AgHyIEVphCEz1160903.shtml',
        'only_matching': True,
    }, {
        'url': 'http://tv.cctv.com/2016/09/07/VIDE5C1FnlX5bUywlrjhxXOV160907.shtml',
        'only_matching': True,
    }, {
        'url': 'http://tv.cntv.cn/video/C39296/95cfac44cabd3ddc4a9438780a4e5c44',
        'only_matching': True
    }]

    def _real_extract(self, url):
        video_id = self._match_id(url)
        webpage = self._download_webpage(url, video_id)

        video_id = self._search_regex(
            [r'var\s+guid\s*=\s*["\']([\da-fA-F]+)',
             r'videoCenterId["\']\s*,\s*["\']([\da-fA-F]+)',
             r'"changePlayer\s*\(\s*["\']([\da-fA-F]+)',
             r'"load[Vv]ideo\s*\(\s*["\']([\da-fA-F]+)'],
            webpage, 'video id')

        data = self._download_json(
            'http://vdn.apps.cntv.cn/api/getHttpVideoInfo.do', video_id,
            query={
                'pid': video_id,
                'url': url,
                'idl': 32,
                'idlr': 32,
                'modifyed': 'false',
            })

        title = data['title']

        formats = []

        video = data.get('video')
        if isinstance(video, dict):
            for quality, chapters_key in enumerate(('lowChapters', 'chapters')):
                video_url = try_get(
                    video, lambda x: x[chapters_key][0]['url'], compat_str)
                if video_url:
                    formats.append({
                        'url': video_url,
                        'format_id': 'http',
                        'quality': quality,
                        'preference': -1,
                    })

        hls_url = try_get(data, lambda x: x['hls_url'], compat_str)
        if hls_url:
            hls_url = re.sub(r'maxbr=\d+&?', '', hls_url)
            formats.extend(self._extract_m3u8_formats(
                hls_url, video_id, 'mp4', entry_protocol='m3u8_native',
                m3u8_id='hls', fatal=False))

        self._sort_formats(formats)

        uploader = data.get('editer_name')
        description = self._html_search_meta('description', webpage)
        timestamp = unified_timestamp(data.get('f_pgmtime'))
        duration = float_or_none(try_get(video, lambda x: x['totalLength']))

        return {
            'id': video_id,
            'title': title,
            'description': description,
            'uploader': uploader,
            'timestamp': timestamp,
            'duration': duration,
            'formats': formats,
        }
Commit	Line	Data
846d8b76 RA	1	# coding: utf-8
	2	from __future__ import unicode_literals
	3
	4	import re
	5
	6	from .common import InfoExtractor
ce7ccb1c S	7	from ..compat import compat_str
	8	from ..utils import (
	9	float_or_none,
	10	try_get,
	11	unified_timestamp,
	12	)
846d8b76 RA	13
	14
	15	class CCTVIE(InfoExtractor):
ce7ccb1c	16	IE_DESC = '央视网'
3783a5cc	17	_VALID_URL = r'https?://(?:[^/]+)\.(?:cntv\|cctv)\.(?:com\|cn)/(?:[^/]+/)*?(?P<id>[^/?#&]+?)(?:/index)?(?:\.s?html\|[?#&]\|$)'
846d8b76	18	_TESTS = [{
ce7ccb1c S	19	'url': 'http://sports.cntv.cn/2016/02/12/ARTIaBRxv4rTT1yWf1frW2wi160212.shtml',
ce7ccb1c S	20	'md5': 'd61ec00a493e09da810bf406a078f691',
846d8b76	21	'info_dict': {
ce7ccb1c	22	'id': '5ecdbeab623f4973b40ff25f18b174e8',
846d8b76	23	'ext': 'mp4',
ce7ccb1c S	24	'title': '[NBA]二少联手砍下46分雷霆主场击败鹈鹕（快讯）',
	25	'description': 'md5:7e14a5328dc5eb3d1cd6afbbe0574e95',
	26	'duration': 98,
	27	'uploader': 'songjunjie',
	28	'timestamp': 1455279956,
	29	'upload_date': '20160212',
	30	},
	31	}, {
	32	'url': 'http://tv.cctv.com/2016/02/05/VIDEUS7apq3lKrHG9Dncm03B160205.shtml',
	33	'info_dict': {
	34	'id': 'efc5d49e5b3b4ab2b34f3a502b73d3ae',
	35	'ext': 'mp4',
	36	'title': '[赛车]“车王”舒马赫恢复情况成谜（快讯）',
	37	'description': '2月4日，蒙特泽莫罗透露了关于“车王”舒马赫恢复情况，但情况是否属实遭到了质疑。',
	38	'duration': 37,
	39	'uploader': 'shujun',
	40	'timestamp': 1454677291,
	41	'upload_date': '20160205',
	42	},
	43	'params': {
	44	'skip_download': True,
	45	},
	46	}, {
	47	'url': 'http://english.cntv.cn/special/four_comprehensives/index.shtml',
	48	'info_dict': {
	49	'id': '4bb9bb4db7a6471ba85fdeda5af0381e',
	50	'ext': 'mp4',
	51	'title': 'NHnews008 ANNUAL POLITICAL SEASON',
	52	'description': 'Four Comprehensives',
	53	'duration': 60,
	54	'uploader': 'zhangyunlei',
	55	'timestamp': 1425385521,
	56	'upload_date': '20150303',
	57	},
	58	'params': {
	59	'skip_download': True,
	60	},
	61	}, {
	62	'url': 'http://cctv.cntv.cn/lm/tvseries_russian/yilugesanghua/index.shtml',
	63	'info_dict': {
	64	'id': 'b15f009ff45c43968b9af583fc2e04b2',
	65	'ext': 'mp4',
	66	'title': 'Путь，усыпанный космеями Серия 1',
	67	'description': 'Путь, усыпанный космеями',
	68	'duration': 2645,
	69	'uploader': 'renxue',
	70	'timestamp': 1477479241,
	71	'upload_date': '20161026',
	72	},
	73	'params': {
	74	'skip_download': True,
	75	},
	76	}, {
	77	'url': 'http://ent.cntv.cn/2016/01/18/ARTIjprSSJH8DryTVr5Bx8Wb160118.shtml',
	78	'only_matching': True,
	79	}, {
	80	'url': 'http://tv.cntv.cn/video/C39296/e0210d949f113ddfb38d31f00a4e5c44',
	81	'only_matching': True,
	82	}, {
	83	'url': 'http://english.cntv.cn/2016/09/03/VIDEhnkB5y9AgHyIEVphCEz1160903.shtml',
	84	'only_matching': True,
846d8b76 RA	85	}, {
	86	'url': 'http://tv.cctv.com/2016/09/07/VIDE5C1FnlX5bUywlrjhxXOV160907.shtml',
	87	'only_matching': True,
	88	}, {
	89	'url': 'http://tv.cntv.cn/video/C39296/95cfac44cabd3ddc4a9438780a4e5c44',
	90	'only_matching': True
	91	}]
	92
	93	def _real_extract(self, url):
ce7ccb1c S	94	video_id = self._match_id(url)
	95	webpage = self._download_webpage(url, video_id)
	96
	97	video_id = self._search_regex(
	98	[r'var\s+guid\s=\s["\']([\da-fA-F]+)',
	99	r'videoCenterId["\']\s,\s["\']([\da-fA-F]+)',
	100	r'"changePlayer\s\(\s["\']([\da-fA-F]+)',
	101	r'"load[Vv]ideo\s\(\s["\']([\da-fA-F]+)'],
327caf66	102	webpage, 'video id')
ce7ccb1c S	103
	104	data = self._download_json(
	105	'http://vdn.apps.cntv.cn/api/getHttpVideoInfo.do', video_id,
	106	query={
	107	'pid': video_id,
	108	'url': url,
	109	'idl': 32,
	110	'idlr': 32,
	111	'modifyed': 'false',
	112	})
	113
	114	title = data['title']
	115
	116	formats = []
	117
	118	video = data.get('video')
	119	if isinstance(video, dict):
	120	for quality, chapters_key in enumerate(('lowChapters', 'chapters')):
	121	video_url = try_get(
	122	video, lambda x: x[chapters_key][0]['url'], compat_str)
	123	if video_url:
	124	formats.append({
	125	'url': video_url,
	126	'format_id': 'http',
	127	'quality': quality,
	128	'preference': -1,
	129	})
	130
	131	hls_url = try_get(data, lambda x: x['hls_url'], compat_str)
	132	if hls_url:
	133	hls_url = re.sub(r'maxbr=\d+&?', '', hls_url)
	134	formats.extend(self._extract_m3u8_formats(
	135	hls_url, video_id, 'mp4', entry_protocol='m3u8_native',
	136	m3u8_id='hls', fatal=False))
	137
	138	self._sort_formats(formats)
	139
	140	uploader = data.get('editer_name')
	141	description = self._html_search_meta('description', webpage)
	142	timestamp = unified_timestamp(data.get('f_pgmtime'))
	143	duration = float_or_none(try_get(video, lambda x: x['totalLength']))
846d8b76 RA	144
	145	return {
	146	'id': video_id,
ce7ccb1c S	147	'title': title,
	148	'description': description,
	149	'uploader': uploader,
	150	'timestamp': timestamp,
	151	'duration': duration,
	152	'formats': formats,
846d8b76	153	}