[extractor] Support multiple archive ids for one video (#4307)

[yt-dlp.git] / yt_dlp / extractor / youtube.py
diff --git a/yt_dlp/extractor/youtube.py b/yt_dlp/extractor/youtube.py

index c60e5ca53385de153d950bf33b34a7c23505e7a5..4dc8e79ac1ba22576cb46293052ba88dad835778 100644 (file)
--- a/yt_dlp/extractor/youtube.py
+++ b/yt_dlp/extractor/youtube.py
@@ -2266,6 +2266,42 @@ class YoutubeIE(YoutubeBaseInfoExtractor):
          }
      ]
  
+    _WEBPAGE_TESTS = [
+        # YouTube <object> embed
+        {
+            'url': 'http://www.improbable.com/2017/04/03/untrained-modern-youths-and-ancient-masters-in-selfie-portraits/',
+            'md5': '873c81d308b979f0e23ee7e620b312a3',
+            'info_dict': {
+                'id': 'msN87y-iEx0',
+                'ext': 'mp4',
+                'title': 'Feynman: Mirrors FUN TO IMAGINE 6',
+                'upload_date': '20080526',
+                'description': 'md5:873c81d308b979f0e23ee7e620b312a3',
+                'uploader': 'Christopher Sykes',
+                'uploader_id': 'ChristopherJSykes',
+                'age_limit': 0,
+                'tags': ['feynman', 'mirror', 'science', 'physics', 'imagination', 'fun', 'cool', 'puzzle'],
+                'channel_id': 'UCCeo--lls1vna5YJABWAcVA',
+                'playable_in_embed': True,
+                'thumbnail': 'https://i.ytimg.com/vi/msN87y-iEx0/hqdefault.jpg',
+                'like_count': int,
+                'comment_count': int,
+                'channel': 'Christopher Sykes',
+                'live_status': 'not_live',
+                'channel_url': 'https://www.youtube.com/channel/UCCeo--lls1vna5YJABWAcVA',
+                'availability': 'public',
+                'duration': 195,
+                'view_count': int,
+                'categories': ['Science & Technology'],
+                'channel_follower_count': int,
+                'uploader_url': 'http://www.youtube.com/user/ChristopherJSykes',
+            },
+            'params': {
+                'skip_download': True,
+            }
+        },
+    ]
+
      @classmethod
      def suitable(cls, url):
          from ..utils import parse_qs
@@ -2298,7 +2334,7 @@ def refetch_manifest(format_id, delay):
              microformats = traverse_obj(
                  prs, (..., 'microformat', 'playerMicroformatRenderer'),
                  expected_type=dict, default=[])
-            _, is_live, _, formats = self._list_formats(video_id, microformats, video_details, prs, player_url)
+            _, is_live, _, formats, _ = self._list_formats(video_id, microformats, video_details, prs, player_url)
              start_time = time.time()
  
          def mpd_feed(format_id, delay):
@@ -3136,7 +3172,7 @@ def append_client(*client_names):
              self.report_warning(last_error)
          return prs, player_url
  
-    def _extract_formats(self, streaming_data, video_id, player_url, is_live, duration):
+    def _extract_formats_and_subtitles(self, streaming_data, video_id, player_url, is_live, duration):
          itags, stream_ids = {}, []
          itag_qualities, res_qualities = {}, {}
          q = qualities([
@@ -3293,17 +3329,22 @@ def process_manifest_format(f, proto, itag):
                  if val in qdict), -1)
              return True
  
+        subtitles = {}
          for sd in streaming_data:
              hls_manifest_url = get_hls and sd.get('hlsManifestUrl')
              if hls_manifest_url:
-                for f in self._extract_m3u8_formats(hls_manifest_url, video_id, 'mp4', fatal=False):
+                fmts, subs = self._extract_m3u8_formats_and_subtitles(hls_manifest_url, video_id, 'mp4', fatal=False, live=is_live)
+                subtitles = self._merge_subtitles(subs, subtitles)
+                for f in fmts:
                      if process_manifest_format(f, 'hls', self._search_regex(
                              r'/itag/(\d+)', f['url'], 'itag', default=None)):
                          yield f
  
              dash_manifest_url = get_dash and sd.get('dashManifestUrl')
              if dash_manifest_url:
-                for f in self._extract_mpd_formats(dash_manifest_url, video_id, fatal=False):
+                formats, subs = self._extract_mpd_formats_and_subtitles(dash_manifest_url, video_id, fatal=False)
+                subtitles = self._merge_subtitles(subs, subtitles)  # Prioritize HLS subs over DASH
+                for f in formats:
                      if process_manifest_format(f, 'dash', f['format_id']):
                          f['filesize'] = int_or_none(self._search_regex(
                              r'/clen/(\d+)', f.get('fragment_base_url') or f['url'], 'file size', default=None))
@@ -3311,6 +3352,7 @@ def process_manifest_format(f, proto, itag):
                              f['is_from_start'] = True
  
                          yield f
+        yield subtitles
  
      def _extract_storyboard(self, player_responses, duration):
          spec = get_first(
@@ -3371,9 +3413,9 @@ def _list_formats(self, video_id, microformats, video_details, player_responses,
              is_live = get_first(live_broadcast_details, 'isLiveNow')
  
          streaming_data = traverse_obj(player_responses, (..., 'streamingData'), default=[])
-        formats = list(self._extract_formats(streaming_data, video_id, player_url, is_live, duration))
+        *formats, subtitles = self._extract_formats_and_subtitles(streaming_data, video_id, player_url, is_live, duration)
  
-        return live_broadcast_details, is_live, streaming_data, formats
+        return live_broadcast_details, is_live, streaming_data, formats, subtitles
  
      def _real_extract(self, url):
          url, smuggled_data = unsmuggle_url(url, {})
@@ -3457,15 +3499,8 @@ def feed_entry(name):
              or get_first(microformats, 'lengthSeconds')
              or parse_duration(search_meta('duration'))) or None
  
-        if get_first(video_details, 'isPostLiveDvr'):
-            self.write_debug('Video is in Post-Live Manifestless mode')
-            if (duration or 0) > 4 * 3600:
-                self.report_warning(
-                    'The livestream has not finished processing. Only 4 hours of the video can be currently downloaded. '
-                    'This is a known issue and patches are welcome')
-
-        live_broadcast_details, is_live, streaming_data, formats = self._list_formats(
-            video_id, microformats, video_details, player_responses, player_url, duration)
+        live_broadcast_details, is_live, streaming_data, formats, automatic_captions = \
+            self._list_formats(video_id, microformats, video_details, player_responses, player_url)
  
          if not formats:
              if not self.get_param('allow_unplayable_formats') and traverse_obj(streaming_data, (..., 'licenseInfos')):
@@ -3556,8 +3591,7 @@ def feed_entry(name):
  
          formats.extend(self._extract_storyboard(player_responses, duration))
  
-        # Source is given priority since formats that throttle are given lower source_preference
-        # When throttling issue is fully fixed, remove this
+        # source_preference is lower for throttled/potentially damaged formats
          self._sort_formats(formats, ('quality', 'res', 'fps', 'hdr:12', 'source', 'codec:vp9.2', 'lang', 'proto'))
  
          info = {
@@ -3595,6 +3629,15 @@ def feed_entry(name):
              'release_timestamp': live_start_time,
          }
  
+        if get_first(video_details, 'isPostLiveDvr'):
+            self.write_debug('Video is in Post-Live Manifestless mode')
+            info['live_status'] = 'post_live'
+            if (duration or 0) > 4 * 3600:
+                self.report_warning(
+                    'The livestream has not finished processing. Only 4 hours of the video can be currently downloaded. '
+                    'This is a known issue and patches are welcome')
+
+        subtitles = {}
          pctr = traverse_obj(player_responses, (..., 'captions', 'playerCaptionsTracklistRenderer'), expected_type=dict)
          if pctr:
              def get_lang_code(track):
@@ -3621,7 +3664,9 @@ def process_language(container, base_url, lang_code, sub_name, query):
                          'name': sub_name,
                      })
  
-            subtitles, automatic_captions = {}, {}
+            # NB: Constructing the full subtitle dictionary is slow
+            get_translated_subs = 'translated_subs' not in self._configuration_arg('skip') and (
+                self.get_param('writeautomaticsub', False) or self.get_param('listsubtitles'))
              for lang_code, caption_track in captions.items():
                  base_url = caption_track.get('baseUrl')
                  orig_lang = parse_qs(base_url).get('lang', [None])[-1]
@@ -3640,7 +3685,7 @@ def process_language(container, base_url, lang_code, sub_name, query):
                          continue
                      orig_trans_code = trans_code
                      if caption_track.get('kind') != 'asr':
-                        if 'translated_subs' in self._configuration_arg('skip'):
+                        if not get_translated_subs:
                              continue
                          trans_code += f'-{lang_code}'
                          trans_name += format_field(lang_name, None, ' from %s')
@@ -3652,8 +3697,9 @@ def process_language(container, base_url, lang_code, sub_name, query):
                      # Setting tlang=lang returns damaged subtitles.
                      process_language(automatic_captions, base_url, trans_code, trans_name,
                                       {} if orig_lang == orig_trans_code else {'tlang': trans_code})
-            info['automatic_captions'] = automatic_captions
-            info['subtitles'] = subtitles
+
+        info['automatic_captions'] = automatic_captions
+        info['subtitles'] = subtitles
  
          parsed_url = urllib.parse.urlparse(url)
          for component in [parsed_url.fragment, parsed_url.query]: