[yt-dlp.git] / yt_dlp / extractor / viidea.py

import re
import urllib.parse

from .common import InfoExtractor
from ..networking.exceptions import HTTPError
from ..utils import (
    ExtractorError,
    js_to_json,
    parse_duration,
    parse_iso8601,
)


class ViideaIE(InfoExtractor):
    _VALID_URL = r'''(?x)https?://(?:www\.)?(?:
            videolectures\.net|
            flexilearn\.viidea\.net|
            presentations\.ocwconsortium\.org|
            video\.travel-zoom\.si|
            video\.pomp-forum\.si|
            tv\.nil\.si|
            video\.hekovnik.com|
            video\.szko\.si|
            kpk\.viidea\.com|
            inside\.viidea\.net|
            video\.kiberpipa\.org|
            bvvideo\.si|
            kongres\.viidea\.net|
            edemokracija\.viidea\.com
        )(?:/lecture)?/(?P<id>[^/]+)(?:/video/(?P<part>\d+))?/*(?:[#?].*)?$'''

    _TESTS = [{
        'url': 'http://videolectures.net/promogram_igor_mekjavic_eng/',
        'info_dict': {
            'id': '20171',
            'display_id': 'promogram_igor_mekjavic_eng',
            'ext': 'mp4',
            'title': 'Automatics, robotics and biocybernetics',
            'description': 'md5:815fc1deb6b3a2bff99de2d5325be482',
            'thumbnail': r're:http://.*\.jpg',
            'timestamp': 1372349289,
            'upload_date': '20130627',
            'duration': 565,
        },
        'params': {
            # m3u8 download
            'skip_download': True,
        },
    }, {
        # video with invalid direct format links (HTTP 403)
        'url': 'http://videolectures.net/russir2010_filippova_nlp/',
        'info_dict': {
            'id': '14891',
            'display_id': 'russir2010_filippova_nlp',
            'ext': 'flv',
            'title': 'NLP at Google',
            'description': 'md5:fc7a6d9bf0302d7cc0e53f7ca23747b3',
            'thumbnail': r're:http://.*\.jpg',
            'timestamp': 1284375600,
            'upload_date': '20100913',
            'duration': 5352,
        },
        'params': {
            # rtmp download
            'skip_download': True,
        },
    }, {
        # event playlist
        'url': 'http://videolectures.net/deeplearning2015_montreal/',
        'info_dict': {
            'id': '23181',
            'title': 'Deep Learning Summer School, Montreal 2015',
            'description': 'md5:0533a85e4bd918df52a01f0e1ebe87b7',
            'thumbnail': r're:http://.*\.jpg',
            'timestamp': 1438560000,
        },
        'playlist_count': 30,
    }, {
        # multi part lecture
        'url': 'http://videolectures.net/mlss09uk_bishop_ibi/',
        'info_dict': {
            'id': '9737',
            'display_id': 'mlss09uk_bishop_ibi',
            'title': 'Introduction To Bayesian Inference',
            'thumbnail': r're:http://.*\.jpg',
            'timestamp': 1251622800,
        },
        'playlist': [{
            'info_dict': {
                'id': '9737_part1',
                'display_id': 'mlss09uk_bishop_ibi_part1',
                'ext': 'wmv',
                'title': 'Introduction To Bayesian Inference (Part 1)',
                'thumbnail': r're:http://.*\.jpg',
                'duration': 4622,
                'timestamp': 1251622800,
                'upload_date': '20090830',
            },
        }, {
            'info_dict': {
                'id': '9737_part2',
                'display_id': 'mlss09uk_bishop_ibi_part2',
                'ext': 'wmv',
                'title': 'Introduction To Bayesian Inference (Part 2)',
                'thumbnail': r're:http://.*\.jpg',
                'duration': 5641,
                'timestamp': 1251622800,
                'upload_date': '20090830',
            },
        }],
        'playlist_count': 2,
    }]

    def _real_extract(self, url):
        lecture_slug, explicit_part_id = self._match_valid_url(url).groups()

        webpage = self._download_webpage(url, lecture_slug)

        cfg = self._parse_json(self._search_regex(
            [r'cfg\s*:\s*({.+?})\s*,\s*[\da-zA-Z_]+\s*:\s*\(?\s*function',
             r'cfg\s*:\s*({[^}]+})'],
            webpage, 'cfg'), lecture_slug, js_to_json)

        lecture_id = str(cfg['obj_id'])

        base_url = self._proto_relative_url(cfg['livepipe'], 'http:')

        try:
            lecture_data = self._download_json(
                f'{base_url}/site/api/lecture/{lecture_id}?format=json',
                lecture_id)['lecture'][0]
        except ExtractorError as e:
            if isinstance(e.cause, HTTPError) and e.cause.status == 403:
                msg = self._parse_json(
                    e.cause.response.read().decode('utf-8'), lecture_id)
                raise ExtractorError(msg['detail'], expected=True)
            raise

        lecture_info = {
            'id': lecture_id,
            'display_id': lecture_slug,
            'title': lecture_data['title'],
            'timestamp': parse_iso8601(lecture_data.get('time')),
            'description': lecture_data.get('description_wiki'),
            'thumbnail': lecture_data.get('thumb'),
        }

        playlist_entries = []
        lecture_type = lecture_data.get('type')
        parts = [str(video) for video in cfg.get('videos', [])]
        if parts:
            multipart = len(parts) > 1

            def extract_part(part_id):
                smil_url = f'{base_url}/{lecture_slug}/video/{part_id}/smil.xml'
                smil = self._download_smil(smil_url, lecture_id)
                info = self._parse_smil(smil, smil_url, lecture_id)
                info['id'] = lecture_id if not multipart else f'{lecture_id}_part{part_id}'
                info['display_id'] = lecture_slug if not multipart else f'{lecture_slug}_part{part_id}'
                if multipart:
                    info['title'] += f' (Part {part_id})'
                switch = smil.find('.//switch')
                if switch is not None:
                    info['duration'] = parse_duration(switch.attrib.get('dur'))
                item_info = lecture_info.copy()
                item_info.update(info)
                return item_info

            if explicit_part_id or not multipart:
                result = extract_part(explicit_part_id or parts[0])
            else:
                result = {
                    '_type': 'multi_video',
                    'entries': [extract_part(part) for part in parts],
                }
                result.update(lecture_info)

            # Immediately return explicitly requested part or non event item
            if explicit_part_id or lecture_type != 'evt':
                return result

            playlist_entries.append(result)

        # It's probably a playlist
        if not parts or lecture_type == 'evt':
            playlist_webpage = self._download_webpage(
                f'{base_url}/site/ajax/drilldown/?id={lecture_id}', lecture_id)
            entries = [
                self.url_result(urllib.parse.urljoin(url, video_url), 'Viidea')
                for _, video_url in re.findall(
                    r'<a[^>]+href=(["\'])(.+?)\1[^>]+id=["\']lec=\d+', playlist_webpage)]
            playlist_entries.extend(entries)

        playlist = self.playlist_result(playlist_entries, lecture_id)
        playlist.update(lecture_info)
        return playlist
Commit	Line	Data
0c996b9f	1	import re
add96eb9	2	import urllib.parse
0c996b9f	3
64e7ad60	4	from .common import InfoExtractor
3d2623a8	5	from ..networking.exceptions import HTTPError
fb97809e	6	from ..utils import (
64f0e30b	7	ExtractorError,
ee4337d1	8	js_to_json,
64f0e30b	9	parse_duration,
ee4337d1	10	parse_iso8601,
fb97809e	11	)
64e7ad60 PH	12
64e7ad60 PH	13
a06bf87a	14	class ViideaIE(InfoExtractor):
5886b38d	15	_VALID_URL = r'''(?x)https?://(?:www\.)?(?:
a06bf87a	16	videolectures\.net\|
	17	flexilearn\.viidea\.net\|
	18	presentations\.ocwconsortium\.org\|
	19	video\.travel-zoom\.si\|
	20	video\.pomp-forum\.si\|
	21	tv\.nil\.si\|
	22	video\.hekovnik.com\|
	23	video\.szko\.si\|
	24	kpk\.viidea\.com\|
	25	inside\.viidea\.net\|
	26	video\.kiberpipa\.org\|
	27	bvvideo\.si\|
	28	kongres\.viidea\.net\|
	29	edemokracija\.viidea\.com
8e3a2bd6	30	)(?:/lecture)?/(?P<id>[^/]+)(?:/video/(?P<part>\d+))?/(?:[#?].)?$'''
64e7ad60	31
6edaa0e2	32	_TESTS = [{
64e7ad60 PH	33	'url': 'http://videolectures.net/promogram_igor_mekjavic_eng/',
64e7ad60 PH	34	'info_dict': {
e8ce2375 S	35	'id': '20171',
e8ce2375 S	36	'display_id': 'promogram_igor_mekjavic_eng',
64e7ad60 PH	37	'ext': 'mp4',
	38	'title': 'Automatics, robotics and biocybernetics',
	39	'description': 'md5:815fc1deb6b3a2bff99de2d5325be482',
ec85ded8	40	'thumbnail': r're:http://.*\.jpg',
e8ce2375	41	'timestamp': 1372349289,
64e7ad60 PH	42	'upload_date': '20130627',
64e7ad60 PH	43	'duration': 565,
64e7ad60	44	},
60ad3eb9 YCH	45	'params': {
	46	# m3u8 download
	47	'skip_download': True,
	48	},
06c6efa9 S	49	}, {
	50	# video with invalid direct format links (HTTP 403)
	51	'url': 'http://videolectures.net/russir2010_filippova_nlp/',
	52	'info_dict': {
e8ce2375 S	53	'id': '14891',
e8ce2375 S	54	'display_id': 'russir2010_filippova_nlp',
06c6efa9 S	55	'ext': 'flv',
	56	'title': 'NLP at Google',
	57	'description': 'md5:fc7a6d9bf0302d7cc0e53f7ca23747b3',
ec85ded8	58	'thumbnail': r're:http://.*\.jpg',
e8ce2375 S	59	'timestamp': 1284375600,
	60	'upload_date': '20100913',
	61	'duration': 5352,
06c6efa9 S	62	},
	63	'params': {
	64	# rtmp download
	65	'skip_download': True,
	66	},
6edaa0e2	67	}, {
e8ce2375	68	# event playlist
6edaa0e2 S	69	'url': 'http://videolectures.net/deeplearning2015_montreal/',
6edaa0e2 S	70	'info_dict': {
ee4337d1	71	'id': '23181',
6edaa0e2	72	'title': 'Deep Learning Summer School, Montreal 2015',
ee4337d1	73	'description': 'md5:0533a85e4bd918df52a01f0e1ebe87b7',
ec85ded8	74	'thumbnail': r're:http://.*\.jpg',
ee4337d1	75	'timestamp': 1438560000,
6edaa0e2 S	76	},
6edaa0e2 S	77	'playlist_count': 30,
ee4337d1	78	}, {
	79	# multi part lecture
	80	'url': 'http://videolectures.net/mlss09uk_bishop_ibi/',
	81	'info_dict': {
	82	'id': '9737',
e8ce2375	83	'display_id': 'mlss09uk_bishop_ibi',
ee4337d1	84	'title': 'Introduction To Bayesian Inference',
ec85ded8	85	'thumbnail': r're:http://.*\.jpg',
ee4337d1	86	'timestamp': 1251622800,
	87	},
	88	'playlist': [{
	89	'info_dict': {
	90	'id': '9737_part1',
e8ce2375	91	'display_id': 'mlss09uk_bishop_ibi_part1',
ee4337d1	92	'ext': 'wmv',
e8ce2375	93	'title': 'Introduction To Bayesian Inference (Part 1)',
ec85ded8	94	'thumbnail': r're:http://.*\.jpg',
e8ce2375 S	95	'duration': 4622,
	96	'timestamp': 1251622800,
	97	'upload_date': '20090830',
ee4337d1	98	},
	99	}, {
	100	'info_dict': {
	101	'id': '9737_part2',
e8ce2375	102	'display_id': 'mlss09uk_bishop_ibi_part2',
ee4337d1	103	'ext': 'wmv',
e8ce2375	104	'title': 'Introduction To Bayesian Inference (Part 2)',
ec85ded8	105	'thumbnail': r're:http://.*\.jpg',
e8ce2375 S	106	'duration': 5641,
	107	'timestamp': 1251622800,
	108	'upload_date': '20090830',
ee4337d1	109	},
	110	}],
	111	'playlist_count': 2,
6edaa0e2	112	}]
64e7ad60 PH	113
64e7ad60 PH	114	def _real_extract(self, url):
5ad28e7f	115	lecture_slug, explicit_part_id = self._match_valid_url(url).groups()
ee4337d1	116
ee4337d1	117	webpage = self._download_webpage(url, lecture_slug)
64e7ad60	118
e8ce2375 S	119	cfg = self._parse_json(self._search_regex(
	120	[r'cfg\s:\s({.+?})\s,\s[\da-zA-Z_]+\s:\s\(?\s*function',
	121	r'cfg\s:\s({[^}]+})'],
	122	webpage, 'cfg'), lecture_slug, js_to_json)
fb97809e	123
add96eb9	124	lecture_id = str(cfg['obj_id'])
64e7ad60	125
a06bf87a	126	base_url = self._proto_relative_url(cfg['livepipe'], 'http:')
a06bf87a	127
64f0e30b S	128	try:
64f0e30b S	129	lecture_data = self._download_json(
add96eb9	130	f'{base_url}/site/api/lecture/{lecture_id}?format=json',
64f0e30b S	131	lecture_id)['lecture'][0]
64f0e30b S	132	except ExtractorError as e:
3d2623a8	133	if isinstance(e.cause, HTTPError) and e.cause.status == 403:
64f0e30b	134	msg = self._parse_json(
3d2623a8	135	e.cause.response.read().decode('utf-8'), lecture_id)
64f0e30b S	136	raise ExtractorError(msg['detail'], expected=True)
64f0e30b S	137	raise
64e7ad60	138
ee4337d1	139	lecture_info = {
	140	'id': lecture_id,
	141	'display_id': lecture_slug,
	142	'title': lecture_data['title'],
	143	'timestamp': parse_iso8601(lecture_data.get('time')),
	144	'description': lecture_data.get('description_wiki'),
	145	'thumbnail': lecture_data.get('thumb'),
	146	}
64e7ad60	147
e8ce2375 S	148	playlist_entries = []
e8ce2375 S	149	lecture_type = lecture_data.get('type')
add96eb9	150	parts = [str(video) for video in cfg.get('videos', [])]
ee4337d1	151	if parts:
e8ce2375 S	152	multipart = len(parts) > 1
	153
	154	def extract_part(part_id):
add96eb9	155	smil_url = f'{base_url}/{lecture_slug}/video/{part_id}/smil.xml'
ee4337d1	156	smil = self._download_smil(smil_url, lecture_id)
ee4337d1	157	info = self._parse_smil(smil, smil_url, lecture_id)
add96eb9	158	info['id'] = lecture_id if not multipart else f'{lecture_id}_part{part_id}'
add96eb9	159	info['display_id'] = lecture_slug if not multipart else f'{lecture_slug}_part{part_id}'
e8ce2375	160	if multipart:
add96eb9	161	info['title'] += f' (Part {part_id})'
ee4337d1	162	switch = smil.find('.//switch')
	163	if switch is not None:
	164	info['duration'] = parse_duration(switch.attrib.get('dur'))
e8ce2375 S	165	item_info = lecture_info.copy()
	166	item_info.update(info)
	167	return item_info
	168
	169	if explicit_part_id or not multipart:
	170	result = extract_part(explicit_part_id or parts[0])
ee4337d1	171	else:
e8ce2375 S	172	result = {
	173	'_type': 'multi_video',
	174	'entries': [extract_part(part) for part in parts],
	175	}
	176	result.update(lecture_info)
	177
	178	# Immediately return explicitly requested part or non event item
	179	if explicit_part_id or lecture_type != 'evt':
	180	return result
	181
	182	playlist_entries.append(result)
	183
	184	# It's probably a playlist
	185	if not parts or lecture_type == 'evt':
	186	playlist_webpage = self._download_webpage(
add96eb9	187	f'{base_url}/site/ajax/drilldown/?id={lecture_id}', lecture_id)
ee4337d1	188	entries = [
add96eb9	189	self.url_result(urllib.parse.urljoin(url, video_url), 'Viidea')
e8ce2375 S	190	for _, video_url in re.findall(
	191	r'<a[^>]+href=(["\'])(.+?)\1[^>]+id=["\']lec=\d+', playlist_webpage)]
	192	playlist_entries.extend(entries)
64e7ad60	193
e8ce2375 S	194	playlist = self.playlist_result(playlist_entries, lecture_id)
	195	playlist.update(lecture_info)
	196	return playlist