]> jfr.im git - yt-dlp.git/blame - yt_dlp/extractor/on24.py
[youtube:comments] Add more options for limiting number of comments extracted (#1626)
[yt-dlp.git] / yt_dlp / extractor / on24.py
CommitLineData
693ec744
DA
1# coding: utf-8
2from __future__ import unicode_literals
3
4from .common import InfoExtractor
5from ..utils import (
6 int_or_none,
7 strip_or_none,
8 try_get,
9 urljoin,
10)
11
12
13class On24IE(InfoExtractor):
14 IE_NAME = 'on24'
15 IE_DESC = 'ON24'
16
17 _VALID_URL = r'''(?x)
18 https?://event\.on24\.com/(?:
19 wcc/r/(?P<id_1>\d{7})/(?P<key_1>[0-9A-F]{32})|
20 eventRegistration/(?:console/EventConsoleApollo|EventLobbyServlet\?target=lobby30)
21 \.jsp\?(?:[^/#?]*&)?eventid=(?P<id_2>\d{7})[^/#?]*&key=(?P<key_2>[0-9A-F]{32})
22 )'''
23
24 _TESTS = [{
25 'url': 'https://event.on24.com/eventRegistration/console/EventConsoleApollo.jsp?uimode=nextgeneration&eventid=2197467&sessionid=1&key=5DF57BE53237F36A43B478DD36277A84&contenttype=A&eventuserid=305999&playerwidth=1000&playerheight=650&caller=previewLobby&text_language_id=en&format=fhaudio&newConsole=false',
26 'info_dict': {
27 'id': '2197467',
28 'ext': 'wav',
29 'title': 'Pearson Test of English General/Pearson English International Certificate Teacher Training Guide',
30 'upload_date': '20200219',
31 'timestamp': 1582149600.0,
32 'view_count': int,
33 }
34 }, {
35 'url': 'https://event.on24.com/wcc/r/2639291/82829018E813065A122363877975752E?mode=login&email=johnsmith@gmail.com',
36 'only_matching': True,
37 }, {
38 'url': 'https://event.on24.com/eventRegistration/console/EventConsoleApollo.jsp?&eventid=2639291&sessionid=1&username=&partnerref=&format=fhvideo1&mobile=&flashsupportedmobiledevice=&helpcenter=&key=82829018E813065A122363877975752E&newConsole=true&nxChe=true&newTabCon=true&text_language_id=en&playerwidth=748&playerheight=526&eventuserid=338788762&contenttype=A&mediametricsessionid=384764716&mediametricid=3558192&usercd=369267058&mode=launch',
39 'only_matching': True,
40 }]
41
42 def _real_extract(self, url):
43 mobj = self._match_valid_url(url)
44 event_id = mobj.group('id_1') or mobj.group('id_2')
45 event_key = mobj.group('key_1') or mobj.group('key_2')
46
47 event_data = self._download_json(
48 'https://event.on24.com/apic/utilApp/EventConsoleCachedServlet',
49 event_id, query={
50 'eventId': event_id,
51 'displayProfile': 'player',
52 'key': event_key,
53 'contentType': 'A'
54 })
55 event_id = str(try_get(event_data, lambda x: x['presentationLogInfo']['eventid'])) or event_id
56 language = event_data.get('localelanguagecode')
57
58 formats = []
59 for media in event_data.get('mediaUrlInfo', []):
60 media_url = urljoin('https://event.on24.com/media/news/corporatevideo/events/', str(media.get('url')))
61 if not media_url:
62 continue
63 media_type = media.get('code')
64 if media_type == 'fhvideo1':
65 formats.append({
66 'format_id': 'video',
67 'url': media_url,
68 'language': language,
69 'ext': 'mp4',
70 'vcodec': 'avc1.640020',
71 'acodec': 'mp4a.40.2',
72 })
73 elif media_type == 'audio':
74 formats.append({
75 'format_id': 'audio',
76 'url': media_url,
77 'language': language,
78 'ext': 'wav',
79 'vcodec': 'none',
80 'acodec': 'wav'
81 })
82 self._sort_formats(formats)
83
84 return {
85 'id': event_id,
86 'title': strip_or_none(event_data.get('description')),
87 'timestamp': int_or_none(try_get(event_data, lambda x: x['session']['startdate']), 1000),
88 'webpage_url': f'https://event.on24.com/wcc/r/{event_id}/{event_key}',
89 'view_count': event_data.get('registrantcount'),
90 'formats': formats,
91 }