[ant1newsgr] Add extractor (#1982)

[yt-dlp.git] / yt_dlp / extractor / common.py
diff --git a/yt_dlp/extractor/common.py b/yt_dlp/extractor/common.py

index e289a4ef82149782d10a3a1da7bd8df6f4eef0c2..f86e7cb3e9fba9510bd981686bebdbca40c12a6f 100644 (file)
--- a/yt_dlp/extractor/common.py
+++ b/yt_dlp/extractor/common.py
@@ -75,6 +75,7 @@
      str_to_int,
      strip_or_none,
      traverse_obj,
+    try_get,
      unescapeHTML,
      UnsupportedError,
      unified_strdate,
@@ -239,6 +240,7 @@ class InfoExtractor(object):
                          * "resolution" (optional, string "{width}x{height}",
                                          deprecated)
                          * "filesize" (optional, int)
+                        * "http_headers" (dict) - HTTP headers for the request
      thumbnail:      Full URL to a video thumbnail image.
      description:    Full video description.
      uploader:       Full name of the video uploader.
@@ -272,6 +274,8 @@ class InfoExtractor(object):
                          * "url": A URL pointing to the subtitles file
                      It can optionally also have:
                          * "name": Name or description of the subtitles
+                        * http_headers: A dictionary of additional HTTP headers
+                                  to add to the request.
                      "ext" will be calculated from URL if missing
      automatic_captions: Like 'subtitles'; contains automatically generated
                      captions instead of normal subtitles
@@ -635,7 +639,7 @@ def extract(self, url):
              }
              if hasattr(e, 'countries'):
                  kwargs['countries'] = e.countries
-            raise type(e)(e.msg, **kwargs)
+            raise type(e)(e.orig_msg, **kwargs)
          except compat_http_client.IncompleteRead as e:
              raise ExtractorError('A network error has occurred.', cause=e, expected=True, video_id=self.get_temp_id(url))
          except (KeyError, StopIteration) as e:
@@ -1097,6 +1101,7 @@ def raise_login_required(
          if metadata_available and (
                  self.get_param('ignore_no_formats_error') or self.get_param('wait_for_video')):
              self.report_warning(msg)
+            return
          if method is not None:
              msg = '%s. %s' % (msg, self._LOGIN_HINTS[method])
          raise ExtractorError(msg, expected=True)
@@ -1135,8 +1140,8 @@ def url_result(url, ie=None, video_id=None, video_title=None, *, url_transparent
              'url': url,
          }
  
-    def playlist_from_matches(self, matches, playlist_id=None, playlist_title=None, getter=None, ie=None, **kwargs):
-        urls = (self.url_result(self._proto_relative_url(m), ie)
+    def playlist_from_matches(self, matches, playlist_id=None, playlist_title=None, getter=None, ie=None, video_kwargs=None, **kwargs):
+        urls = (self.url_result(self._proto_relative_url(m), ie, **(video_kwargs or {}))
                  for m in orderedSet(map(getter, matches) if getter else matches))
          return self.playlist_result(urls, playlist_id, playlist_title, **kwargs)
  
@@ -1291,6 +1296,7 @@ def _og_search_description(self, html, **kargs):
          return self._og_search_property('description', html, fatal=False, **kargs)
  
      def _og_search_title(self, html, **kargs):
+        kargs.setdefault('fatal', False)
          return self._og_search_property('title', html, **kargs)
  
      def _og_search_video_url(self, html, name='video url', secure=True, **kargs):
@@ -1302,6 +1308,10 @@ def _og_search_video_url(self, html, name='video url', secure=True, **kargs):
      def _og_search_url(self, html, **kargs):
          return self._og_search_property('url', html, **kargs)
  
+    def _html_extract_title(self, html, name, **kwargs):
+        return self._html_search_regex(
+            r'(?s)<title>(.*?)</title>', html, name, **kwargs)
+
      def _html_search_meta(self, name, html, display_name=None, fatal=False, **kwargs):
          name = variadic(name)
          if display_name is None:
@@ -1447,7 +1457,7 @@ def extract_chapter_information(e):
                  'title': part.get('name'),
                  'start_time': part.get('startOffset'),
                  'end_time': part.get('endOffset'),
-            } for part in e.get('hasPart', []) if part.get('@type') == 'Clip']
+            } for part in variadic(e.get('hasPart') or []) if part.get('@type') == 'Clip']
              for idx, (last_c, current_c, next_c) in enumerate(zip(
                      [{'end_time': 0}] + chapters, chapters, chapters[1:])):
                  current_c['end_time'] = current_c['end_time'] or next_c['start_time']
@@ -1528,6 +1538,8 @@ def traverse_json_ld(json_ld, at_top_level=True):
                          'title': unescapeHTML(e.get('headline')),
                          'description': unescapeHTML(e.get('articleBody') or e.get('description')),
                      })
+                    if traverse_obj(e, ('video', 0, '@type')) == 'VideoObject':
+                        extract_video_object(e['video'][0])
                  elif item_type == 'VideoObject':
                      extract_video_object(e)
                      if expected_type is None:
@@ -1606,7 +1618,7 @@ class FormatSort:
              'vcodec': {'type': 'ordered', 'regex': True,
                         'order': ['av0?1', 'vp0?9.2', 'vp0?9', '[hx]265|he?vc?', '[hx]264|avc', 'vp0?8', 'mp4v|h263', 'theora', '', None, 'none']},
              'acodec': {'type': 'ordered', 'regex': True,
-                       'order': ['[af]lac', 'wav|aiff', 'opus', 'vorbis', 'aac', 'mp?4a?', 'mp3', 'e-?a?c-?3', 'ac-?3', 'dts', '', None, 'none']},
+                       'order': ['[af]lac', 'wav|aiff', 'opus', 'vorbis|ogg', 'aac', 'mp?4a?', 'mp3', 'e-?a?c-?3', 'ac-?3', 'dts', '', None, 'none']},
              'hdr': {'type': 'ordered', 'regex': True, 'field': 'dynamic_range',
                      'order': ['dv', '(hdr)?12', r'(hdr)?10\+', '(hdr)?10', 'hlg', '', 'sdr', None]},
              'proto': {'type': 'ordered', 'regex': True, 'field': 'protocol',
@@ -2872,7 +2884,8 @@ def location_key(location):
                              segment_duration = None
                              if 'total_number' not in representation_ms_info and 'segment_duration' in representation_ms_info:
                                  segment_duration = float_or_none(representation_ms_info['segment_duration'], representation_ms_info['timescale'])
-                                representation_ms_info['total_number'] = int(math.ceil(float(period_duration) / segment_duration))
+                                representation_ms_info['total_number'] = int(math.ceil(
+                                    float_or_none(period_duration, segment_duration, default=0)))
                              representation_ms_info['fragments'] = [{
                                  media_location_key: media_template % {
                                      'Number': segment_number,
@@ -2963,6 +2976,10 @@ def add_segment_url():
                                  f['url'] = initialization_url
                              f['fragments'].append({location_key(initialization_url): initialization_url})
                          f['fragments'].extend(representation_ms_info['fragments'])
+                        if not period_duration:
+                            period_duration = try_get(
+                                representation_ms_info,
+                                lambda r: sum(frag['duration'] for frag in r['fragments']), float)
                      else:
                          # Assuming direct URL to unfragmented media.
                          f['url'] = base_url
@@ -3105,7 +3122,7 @@ def _parse_ism_formats_and_subtitles(self, ism_doc, ism_url, ism_id=None):
                      })
          return formats, subtitles
  
-    def _parse_html5_media_entries(self, base_url, webpage, video_id, m3u8_id=None, m3u8_entry_protocol='m3u8', mpd_id=None, preference=None, quality=None):
+    def _parse_html5_media_entries(self, base_url, webpage, video_id, m3u8_id=None, m3u8_entry_protocol='m3u8_native', mpd_id=None, preference=None, quality=None):
          def absolute_url(item_url):
              return urljoin(base_url, item_url)
  
@@ -3504,8 +3521,6 @@ def _live_title(self, name):
  
      def _int(self, v, name, fatal=False, **kwargs):
          res = int_or_none(v, **kwargs)
-        if 'get_attr' in kwargs:
-            print(getattr(v, kwargs['get_attr']))
          if res is None:
              msg = 'Failed to extract %s: Could not parse value %r' % (name, v)
              if fatal:
@@ -3664,7 +3679,7 @@ def _get_automatic_captions(self, *args, **kwargs):
      def mark_watched(self, *args, **kwargs):
          if not self.get_param('mark_watched', False):
              return
-        if (self._get_login_info()[0] is not None
+        if (hasattr(self, '_NETRC_MACHINE') and self._get_login_info()[0] is not None
                  or self.get_param('cookiefile')
                  or self.get_param('cookiesfrombrowser')):
              self._mark_watched(*args, **kwargs)
@@ -3712,6 +3727,22 @@ def _configuration_arg(self, key, default=NO_DEFAULT, *, ie_key=None, casesense=
              return [] if default is NO_DEFAULT else default
          return list(val) if casesense else [x.lower() for x in val]
  
+    def _yes_playlist(self, playlist_id, video_id, smuggled_data=None, *, playlist_label='playlist', video_label='video'):
+        if not playlist_id or not video_id:
+            return not video_id
+
+        no_playlist = (smuggled_data or {}).get('force_noplaylist')
+        if no_playlist is not None:
+            return not no_playlist
+
+        video_id = '' if video_id is True else f' {video_id}'
+        playlist_id = '' if playlist_id is True else f' {playlist_id}'
+        if self.get_param('noplaylist'):
+            self.to_screen(f'Downloading just the {video_label}{video_id} because of --no-playlist')
+            return False
+        self.to_screen(f'Downloading {playlist_label}{playlist_id} - add --no-playlist to download just the {video_label}{video_id}')
+        return True
+
  
  class SearchInfoExtractor(InfoExtractor):
      """