[extractor/youtube] Parse translated subtitles only when requested

[yt-dlp.git] / yt_dlp / YoutubeDL.py
diff --git a/yt_dlp/YoutubeDL.py b/yt_dlp/YoutubeDL.py

index 4162727c49465aee70bdb240366b2db2ea27484a..38a8bb6c1a55b7a02e2670b036632e7a20abb21b 100644 (file)
--- a/yt_dlp/YoutubeDL.py
+++ b/yt_dlp/YoutubeDL.py
@@ -1,4 +1,3 @@
-#!/usr/bin/env python3
  import collections
  import contextlib
  import datetime
@@ -11,7 +10,6 @@
  import locale
  import operator
  import os
-import platform
  import random
  import re
  import shutil
@@ -26,15 +24,7 @@
  from string import ascii_letters
  
  from .cache import Cache
-from .compat import (
-    HAS_LEGACY as compat_has_legacy,
-    compat_get_terminal_size,
-    compat_os_name,
-    compat_shlex_quote,
-    compat_str,
-    compat_urllib_error,
-    compat_urllib_request,
-)
+from .compat import compat_os_name, compat_shlex_quote
  from .cookies import load_cookies
  from .downloader import FFmpegFD, get_suitable_downloader, shorten_protocol_name
  from .downloader.rtmp import rtmpdump_version
@@ -52,12 +42,15 @@
      FFmpegFixupTimestampPP,
      FFmpegMergerPP,
      FFmpegPostProcessor,
+    FFmpegVideoConvertorPP,
      MoveFilesAfterDownloadPP,
      get_postprocessor,
  )
+from .postprocessor.ffmpeg import resolve_mapping as resolve_recode_mapping
  from .update import detect_variant
  from .utils import (
      DEFAULT_OUTTMPL,
+    IDENTITY,
      LINK_TEMPLATES,
      NO_DEFAULT,
      NUMBER_RE,
@@ -87,17 +80,20 @@
      RejectedVideoReached,
      SameFileError,
      UnavailableVideoError,
+    UserNotLive,
      YoutubeDLCookieProcessor,
      YoutubeDLHandler,
      YoutubeDLRedirectHandler,
      age_restricted,
      args_to_str,
+    bug_reports_message,
      date_from_str,
      determine_ext,
      determine_protocol,
      encode_compat_str,
      encodeFilename,
      error_to_compat_str,
+    escapeHTML,
      expand_path,
      filter_dict,
      float_or_none,
@@ -117,7 +113,6 @@
      number_of_digits,
      orderedSet,
      parse_filesize,
-    platform_name,
      preferredencoding,
      prepend_extension,
      register_socks_protocols,
@@ -133,6 +128,7 @@
      strftime_or_none,
      subtitles_filename,
      supports_terminal_sequences,
+    system_identifier,
      timetuple_from_msec,
      to_high_limit_path,
      traverse_obj,
@@ -242,11 +238,9 @@ class YoutubeDL:
                         and don't overwrite any file if False
                         For compatibility with youtube-dl,
                         "nooverwrites" may also be used instead
-    playliststart:     Playlist item to start at.
-    playlistend:       Playlist item to end at.
      playlist_items:    Specific indices of playlist to download.
-    playlistreverse:   Download playlist items in reverse order.
      playlistrandom:    Download playlist items in random order.
+    lazy_playlist:     Process playlist entries as they are received.
      matchtitle:        Download only matching titles.
      rejecttitle:       Reject downloads for matching titles.
      logger:            Log messages to a logging.Logger instance.
@@ -313,7 +307,7 @@ class YoutubeDL:
      client_certificate_password:  Password for client certificate private key, if encrypted.
                          If not provided and the key is encrypted, yt-dlp will ask interactively
      prefer_insecure:   Use HTTP instead of HTTPS to retrieve information.
-                       At the moment, this is only supported by YouTube.
+                       (Only supported by some extractors)
      http_headers:      A dictionary of custom headers to be used for all requests
      proxy:             URL of the proxy server to use
      geo_verification_proxy:  URL of the proxy to use for IP address verification
@@ -325,9 +319,14 @@ class YoutubeDL:
      default_search:    Prepend this string if an input url is not valid.
                         'auto' for elaborate guessing
      encoding:          Use this encoding instead of the system-specified.
-    extract_flat:      Do not resolve URLs, return the immediate result.
-                       Pass in 'in_playlist' to only show this behavior for
-                       playlist items.
+    extract_flat:      Whether to resolve and process url_results further
+                       * False:     Always process (default)
+                       * True:      Never process
+                       * 'in_playlist': Do not process inside playlist/multi_video
+                       * 'discard': Always process, but don't return the result
+                                    from inside playlist/multi_video
+                       * 'discard_in_playlist': Same as "discard", but only for
+                                    playlists (not multi_video)
      wait_for_video:    If given, wait for scheduled streams to become available.
                         The value should be a tuple containing the range
                         (min_secs, max_secs) to wait between retries
@@ -431,19 +430,22 @@ class YoutubeDL:
      retry_sleep_functions: Dictionary of functions that takes the number of attempts
                         as argument and returns the time to sleep in seconds.
                         Allowed keys are 'http', 'fragment', 'file_access'
-    download_ranges:   A function that gets called for every video with the signature
-                       (info_dict, *, ydl) -> Iterable[Section].
-                       Only the returned sections will be downloaded. Each Section contains:
+    download_ranges:   A callback function that gets called for every video with
+                       the signature (info_dict, ydl) -> Iterable[Section].
+                       Only the returned sections will be downloaded.
+                       Each Section is a dict with the following keys:
                         * start_time: Start time of the section in seconds
                         * end_time: End time of the section in seconds
                         * title: Section title (Optional)
                         * index: Section number (Optional)
+    force_keyframes_at_cuts: Re-encode the video when downloading ranges to get precise cuts
+    noprogress:        Do not print the progress bar
  
      The following parameters are not used by YoutubeDL itself, they are used by
      the downloader (see yt_dlp/downloader/common.py):
      nopart, updatetime, buffersize, ratelimit, throttledratelimit, min_filesize,
      max_filesize, test, noresizebuffer, retries, file_access_retries, fragment_retries,
-    continuedl, noprogress, xattr_set_filesize, hls_use_mpegts, http_chunk_size,
+    continuedl, xattr_set_filesize, hls_use_mpegts, http_chunk_size,
      external_downloader_args, concurrent_fragment_downloads.
  
      The following options are used by the post processors:
@@ -469,6 +471,12 @@ class YoutubeDL:
  
      The following options are deprecated and may be removed in the future:
  
+    playliststart:     - Use playlist_items
+                       Playlist item to start at.
+    playlistend:       - Use playlist_items
+                       Playlist item to end at.
+    playlistreverse:   - Use playlist_items
+                       Download playlist items in reverse order.
      forceurl:          - Use forceprint
                         Force printing final URL.
      forcetitle:        - Use forceprint
@@ -577,9 +585,17 @@ def __init__(self, params=None, auto_init=True):
              for type_, stream in self._out_files.items_ if type_ != 'console'
          })
  
-        if sys.version_info < (3, 6):
-            self.report_warning(
-                'Python version %d.%d is not supported! Please update to Python 3.6 or above' % sys.version_info[:2])
+        # The code is left like this to be reused for future deprecations
+        MIN_SUPPORTED, MIN_RECOMMENDED = (3, 7), (3, 7)
+        current_version = sys.version_info[:2]
+        if current_version < MIN_RECOMMENDED:
+            msg = ('Support for Python version %d.%d has been deprecated. '
+                   'See  https://github.com/yt-dlp/yt-dlp/issues/3764  for more details.'
+                   '\n                    You will no longer receive updates on this version')
+            if current_version < MIN_SUPPORTED:
+                msg = 'Python version %d.%d is no longer supported'
+            self.deprecation_warning(
+                f'{msg}! Please update to Python %d.%d or above' % (*current_version, *MIN_RECOMMENDED))
  
          if self.params.get('allow_unplayable_formats'):
              self.report_warning(
@@ -608,8 +624,6 @@ def check_deprecated(param, option, suggestion):
              self.deprecation_warning(msg)
  
          self.params['compat_opts'] = set(self.params.get('compat_opts', ()))
-        if not compat_has_legacy:
-            self.params['compat_opts'].add('no-compat-legacy')
          if 'list-formats' in self.params['compat_opts']:
              self.params['listformats_table'] = False
  
@@ -634,7 +648,7 @@ def check_deprecated(param, option, suggestion):
              try:
                  import pty
                  master, slave = pty.openpty()
-                width = compat_get_terminal_size().columns
+                width = shutil.get_terminal_size().columns
                  width_args = [] if width is None else ['-w', str(width)]
                  sp_kwargs = {'stdin': subprocess.PIPE, 'stdout': slave, 'stderr': self._out_files.error}
                  try:
@@ -665,7 +679,7 @@ def check_deprecated(param, option, suggestion):
                  'Set the LC_ALL environment variable to fix this.')
              self.params['restrictfilenames'] = True
  
-        self.outtmpl_dict = self.parse_outtmpl()
+        self._parse_outtmpl()
  
          # Creating format selector here allows us to catch syntax errors before the extraction
          self.format_selector = (
@@ -765,6 +779,7 @@ def add_default_info_extractors(self):
  
      def add_post_processor(self, pp, when='post_process'):
          """Add a PostProcessor object to the end of the chain."""
+        assert when in POSTPROCESS_WHEN, f'Invalid when={when}'
          self._pps[when].append(pp)
          pp.set_downloader(self)
  
@@ -788,7 +803,7 @@ def _bidi_workaround(self, message):
              return message
  
          assert hasattr(self, '_output_process')
-        assert isinstance(message, compat_str)
+        assert isinstance(message, str)
          line_count = message.count('\n') + 1
          self._output_process.stdin.write((message + '\n').encode())
          self._output_process.stdin.flush()
@@ -824,7 +839,7 @@ def to_screen(self, message, skip_eol=False, quiet=None):
  
      def to_stderr(self, message, only_once=False):
          """Print message to stderr"""
-        assert isinstance(message, compat_str)
+        assert isinstance(message, str)
          if self.params.get('logger'):
              self.params['logger'].error(message)
          else:
@@ -992,21 +1007,19 @@ def raise_no_formats(self, info, forced=False, *, msg=None):
              self.report_warning(msg)
  
      def parse_outtmpl(self):
-        outtmpl_dict = self.params.get('outtmpl', {})
-        if not isinstance(outtmpl_dict, dict):
-            outtmpl_dict = {'default': outtmpl_dict}
-        # Remove spaces in the default template
-        if self.params.get('restrictfilenames'):
+        self.deprecation_warning('"YoutubeDL.parse_outtmpl" is deprecated and may be removed in a future version')
+        self._parse_outtmpl()
+        return self.params['outtmpl']
+
+    def _parse_outtmpl(self):
+        sanitize = IDENTITY
+        if self.params.get('restrictfilenames'):  # Remove spaces in the default template
              sanitize = lambda x: x.replace(' - ', ' ').replace(' ', '-')
-        else:
-            sanitize = lambda x: x
-        outtmpl_dict.update({
-            k: sanitize(v) for k, v in DEFAULT_OUTTMPL.items()
-            if outtmpl_dict.get(k) is None})
-        for _, val in outtmpl_dict.items():
-            if isinstance(val, bytes):
-                self.report_warning('Parameter outtmpl is bytes, but should be a unicode string')
-        return outtmpl_dict
+
+        outtmpl = self.params.setdefault('outtmpl', {})
+        if not isinstance(outtmpl, dict):
+            self.params['outtmpl'] = outtmpl = {'default': outtmpl}
+        outtmpl.update({k: sanitize(v) for k, v in DEFAULT_OUTTMPL.items() if outtmpl.get(k) is None})
  
      def get_output_path(self, dir_type='', filename=None):
          paths = self.params.get('paths', {})
@@ -1044,7 +1057,7 @@ def escape_outtmpl(outtmpl):
      def validate_outtmpl(cls, outtmpl):
          ''' @return None or Exception object '''
          outtmpl = re.sub(
-            STR_FORMAT_RE_TMPL.format('[^)]*', '[ljqBUDS]'),
+            STR_FORMAT_RE_TMPL.format('[^)]*', '[ljhqBUDS]'),
              lambda mobj: f'{mobj.group(0)[:-1]}s',
              cls._outtmpl_expandpath(outtmpl))
          try:
@@ -1087,7 +1100,7 @@ def prepare_outtmpl(self, outtmpl, info_dict, sanitize=False):
          }
  
          TMPL_DICT = {}
-        EXTERNAL_FORMAT_RE = re.compile(STR_FORMAT_RE_TMPL.format('[^)]*', f'[{STR_FORMAT_TYPES}ljqBUDS]'))
+        EXTERNAL_FORMAT_RE = re.compile(STR_FORMAT_RE_TMPL.format('[^)]*', f'[{STR_FORMAT_TYPES}ljhqBUDS]'))
          MATH_FUNCTIONS = {
              '+': float.__add__,
              '-': float.__sub__,
@@ -1196,6 +1209,8 @@ def create_key(outer_mobj):
                  value, fmt = delim.join(map(str, variadic(value, allowed_types=(str, bytes)))), str_fmt
              elif fmt[-1] == 'j':  # json
                  value, fmt = json.dumps(value, default=_dumpjson_default, indent=4 if '#' in flags else None), str_fmt
+            elif fmt[-1] == 'h':  # html
+                value, fmt = escapeHTML(value), str_fmt
              elif fmt[-1] == 'q':  # quoted
                  value = map(str, variadic(value) if '#' in flags else [value])
                  value, fmt = ' '.join(map(compat_shlex_quote, value)), str_fmt
@@ -1244,7 +1259,7 @@ def evaluate_outtmpl(self, outtmpl, info_dict, *args, **kwargs):
      def _prepare_filename(self, info_dict, *, outtmpl=None, tmpl_type=None):
          assert None in (outtmpl, tmpl_type), 'outtmpl and tmpl_type are mutually exclusive'
          if outtmpl is None:
-            outtmpl = self.outtmpl_dict.get(tmpl_type or 'default', self.outtmpl_dict['default'])
+            outtmpl = self.params['outtmpl'].get(tmpl_type or 'default', self.params['outtmpl']['default'])
          try:
              outtmpl = self._outtmpl_expandpath(outtmpl)
              filename = self.evaluate_outtmpl(outtmpl, info_dict, True)
@@ -1295,7 +1310,7 @@ def prepare_filename(self, info_dict, dir_type='', *, outtmpl=None, warn=False):
      def _match_entry(self, info_dict, incomplete=False, silent=False):
          """ Returns None if the file should be downloaded """
  
-        video_title = info_dict.get('title', info_dict.get('id', 'video'))
+        video_title = info_dict.get('title', info_dict.get('id', 'entry'))
  
          def check_filter():
              if 'title' in info_dict:
@@ -1442,7 +1457,7 @@ def wrapper(self, *args, **kwargs):
                  break
          return wrapper
  
-    def _wait_for_video(self, ie_result):
+    def _wait_for_video(self, ie_result={}):
          if (not self.params.get('wait_for_video')
                  or ie_result.get('_type', 'video') != 'video'
                  or ie_result.get('formats') or ie_result.get('url')):
@@ -1453,7 +1468,12 @@ def _wait_for_video(self, ie_result):
  
          def progress(msg):
              nonlocal last_msg
-            self.to_screen(msg + ' ' * (len(last_msg) - len(msg)) + '\r', skip_eol=True)
+            full_msg = f'{msg}\n'
+            if not self.params.get('noprogress'):
+                full_msg = msg + ' ' * (len(last_msg) - len(msg)) + '\r'
+            elif last_msg:
+                return
+            self.to_screen(full_msg, skip_eol=True)
              last_msg = msg
  
          min_wait, max_wait = self.params.get('wait_for_video')
@@ -1461,7 +1481,7 @@ def progress(msg):
          if diff is None and ie_result.get('live_status') == 'is_upcoming':
              diff = round(random.uniform(min_wait, max_wait) if (max_wait and min_wait) else (max_wait or min_wait), 0)
              self.report_warning('Release time of video is not known')
-        elif (diff or 0) <= 0:
+        elif ie_result and (diff or 0) <= 0:
              self.report_warning('Video should already be available according to extracted info')
          diff = min(max(diff or 0, min_wait or 0), max_wait or float('inf'))
          self.to_screen(f'[wait] Waiting for {format_dur(diff)} - Press Ctrl+C to try now')
@@ -1485,8 +1505,16 @@ def progress(msg):
  
      @_handle_extraction_exceptions
      def __extract_info(self, url, ie, download, extra_info, process):
-        ie_result = ie.extract(url)
+        try:
+            ie_result = ie.extract(url)
+        except UserNotLive as e:
+            if process:
+                if self.params.get('wait_for_video'):
+                    self.report_warning(e)
+                self._wait_for_video()
+            raise
          if ie_result is None:  # Finished already (backwards compatibility; listformats and friends should be moved here)
+            self.report_warning(f'Extractor {ie.IE_NAME} returned nothing{bug_reports_message()}')
              return
          if isinstance(ie_result, list):
              # Backwards compatibility: old IE result format
@@ -1561,7 +1589,7 @@ def process_ie_result(self, ie_result, download=True, extra_info=None):
              additional_urls = (ie_result or {}).get('additional_urls')
              if additional_urls:
                  # TODO: Improve MetadataParserPP to allow setting a list
-                if isinstance(additional_urls, compat_str):
+                if isinstance(additional_urls, str):
                      additional_urls = [additional_urls]
                  self.to_screen(
                      '[info] %s: %d additional URL(s) requested' % (ie_result['id'], len(additional_urls)))
@@ -1592,9 +1620,13 @@ def process_ie_result(self, ie_result, download=True, extra_info=None):
              if not info:
                  return info
  
+            exempted_fields = {'_type', 'url', 'ie_key'}
+            if not ie_result.get('section_end') and ie_result.get('section_start') is None:
+                # For video clips, the id etc of the clip extractor should be used
+                exempted_fields |= {'id', 'extractor', 'extractor_key'}
+
              new_result = info.copy()
-            new_result.update(filter_dict(ie_result, lambda k, v: (
-                v is not None and k not in {'_type', 'url', 'id', 'extractor', 'extractor_key', 'ie_key'})))
+            new_result.update(filter_dict(ie_result, lambda k, v: v is not None and k not in exempted_fields))
  
              # Extracted info may not be a video result (i.e.
              # info.get('_type', 'video') != video) but rather an url or
@@ -1653,34 +1685,62 @@ def _ensure_dir_exists(self, path):
          return make_dir(path, self.report_error)
  
      @staticmethod
-    def _playlist_infodict(ie_result, **kwargs):
-        return {
-            **ie_result,
+    def _playlist_infodict(ie_result, strict=False, **kwargs):
+        info = {
+            'playlist_count': ie_result.get('playlist_count'),
              'playlist': ie_result.get('title') or ie_result.get('id'),
              'playlist_id': ie_result.get('id'),
              'playlist_title': ie_result.get('title'),
              'playlist_uploader': ie_result.get('uploader'),
              'playlist_uploader_id': ie_result.get('uploader_id'),
-            'playlist_index': 0,
              **kwargs,
          }
+        if strict:
+            return info
+        return {
+            **info,
+            'playlist_index': 0,
+            '__last_playlist_index': max(ie_result['requested_entries'] or (0, 0)),
+            'extractor': ie_result['extractor'],
+            'webpage_url': ie_result['webpage_url'],
+            'webpage_url_basename': url_basename(ie_result['webpage_url']),
+            'webpage_url_domain': get_domain(ie_result['webpage_url']),
+            'extractor_key': ie_result['extractor_key'],
+        }
  
      def __process_playlist(self, ie_result, download):
          """Process each entry in the playlist"""
-        title = ie_result.get('title') or ie_result.get('id') or '<Untitled>'
-        self.to_screen(f'[download] Downloading playlist: {title}')
+        assert ie_result['_type'] in ('playlist', 'multi_video')
+
+        common_info = self._playlist_infodict(ie_result, strict=True)
+        title = common_info.get('playlist') or '<Untitled>'
+        if self._match_entry(common_info, incomplete=True) is not None:
+            return
+        self.to_screen(f'[download] Downloading {ie_result["_type"]}: {title}')
  
          all_entries = PlaylistEntries(self, ie_result)
-        entries = orderedSet(all_entries.get_requested_items())
-        ie_result['requested_entries'], ie_result['entries'] = tuple(zip(*entries)) or ([], [])
-        n_entries, ie_result['playlist_count'] = len(entries), all_entries.full_count
+        entries = orderedSet(all_entries.get_requested_items(), lazy=True)
+
+        lazy = self.params.get('lazy_playlist')
+        if lazy:
+            resolved_entries, n_entries = [], 'N/A'
+            ie_result['requested_entries'], ie_result['entries'] = None, None
+        else:
+            entries = resolved_entries = list(entries)
+            n_entries = len(resolved_entries)
+            ie_result['requested_entries'], ie_result['entries'] = tuple(zip(*resolved_entries)) or ([], [])
+        if not ie_result.get('playlist_count'):
+            # Better to do this after potentially exhausting entries
+            ie_result['playlist_count'] = all_entries.get_full_count()
+
+        ie_copy = collections.ChainMap(
+            ie_result, self._playlist_infodict(ie_result, n_entries=int_or_none(n_entries)))
  
          _infojson_written = False
          write_playlist_files = self.params.get('allow_playlist_files', True)
          if write_playlist_files and self.params.get('list_thumbnails'):
              self.list_thumbnails(ie_result)
          if write_playlist_files and not self.params.get('simulate'):
-            ie_copy = self._playlist_infodict(ie_result, n_entries=n_entries)
              _infojson_written = self._write_info_json(
                  'playlist', ie_result, self.prepare_filename(ie_copy, 'pl_infojson'))
              if _infojson_written is None:
@@ -1689,56 +1749,62 @@ def __process_playlist(self, ie_result, download):
                                         self.prepare_filename(ie_copy, 'pl_description')) is None:
                  return
              # TODO: This should be passed to ThumbnailsConvertor if necessary
-            self._write_thumbnails('playlist', ie_copy, self.prepare_filename(ie_copy, 'pl_thumbnail'))
-
-        if self.params.get('playlistreverse', False):
-            entries = entries[::-1]
-        if self.params.get('playlistrandom', False):
+            self._write_thumbnails('playlist', ie_result, self.prepare_filename(ie_copy, 'pl_thumbnail'))
+
+        if lazy:
+            if self.params.get('playlistreverse') or self.params.get('playlistrandom'):
+                self.report_warning('playlistreverse and playlistrandom are not supported with lazy_playlist', only_once=True)
+        elif self.params.get('playlistreverse'):
+            entries.reverse()
+        elif self.params.get('playlistrandom'):
              random.shuffle(entries)
  
          self.to_screen(f'[{ie_result["extractor"]}] Playlist {title}: Downloading {n_entries} videos'
                         f'{format_field(ie_result, "playlist_count", " of %s")}')
  
+        keep_resolved_entries = self.params.get('extract_flat') != 'discard'
+        if self.params.get('extract_flat') == 'discard_in_playlist':
+            keep_resolved_entries = ie_result['_type'] != 'playlist'
+        if keep_resolved_entries:
+            self.write_debug('The information of all playlist entries will be held in memory')
+
          failures = 0
          max_failures = self.params.get('skip_playlist_after_errors') or float('inf')
-        for i, (playlist_index, entry) in enumerate(entries, 1):
-            # TODO: Add auto-generated fields
-            if self._match_entry(entry, incomplete=True) is not None:
+        for i, (playlist_index, entry) in enumerate(entries):
+            if lazy:
+                resolved_entries.append((playlist_index, entry))
+            if not entry:
                  continue
  
-            if 'playlist-index' in self.params.get('compat_opts', []):
-                playlist_index = ie_result['requested_entries'][i - 1]
-            self.to_screen('[download] Downloading video %s of %s' % (
-                self._format_screen(i, self.Styles.ID), self._format_screen(n_entries, self.Styles.EMPHASIS)))
-
              entry['__x_forwarded_for_ip'] = ie_result.get('__x_forwarded_for_ip')
-            entry_result = self.__process_iterable_entry(entry, download, {
-                'n_entries': n_entries,
-                '__last_playlist_index': max(ie_result['requested_entries']),
-                'playlist_count': ie_result.get('playlist_count'),
+            if not lazy and 'playlist-index' in self.params.get('compat_opts', []):
+                playlist_index = ie_result['requested_entries'][i]
+
+            extra = {
+                **common_info,
+                'n_entries': int_or_none(n_entries),
                  'playlist_index': playlist_index,
-                'playlist_autonumber': i,
-                'playlist': title,
-                'playlist_id': ie_result.get('id'),
-                'playlist_title': ie_result.get('title'),
-                'playlist_uploader': ie_result.get('uploader'),
-                'playlist_uploader_id': ie_result.get('uploader_id'),
-                'extractor': ie_result['extractor'],
-                'webpage_url': ie_result['webpage_url'],
-                'webpage_url_basename': url_basename(ie_result['webpage_url']),
-                'webpage_url_domain': get_domain(ie_result['webpage_url']),
-                'extractor_key': ie_result['extractor_key'],
-            })
+                'playlist_autonumber': i + 1,
+            }
+
+            if self._match_entry(collections.ChainMap(entry, extra), incomplete=True) is not None:
+                continue
+
+            self.to_screen('[download] Downloading video %s of %s' % (
+                self._format_screen(i + 1, self.Styles.ID), self._format_screen(n_entries, self.Styles.EMPHASIS)))
+
+            entry_result = self.__process_iterable_entry(entry, download, extra)
              if not entry_result:
                  failures += 1
              if failures >= max_failures:
                  self.report_error(
                      f'Skipping the remaining entries in playlist "{title}" since {failures} items failed extraction')
                  break
-            entries[i - 1] = (playlist_index, entry_result)
+            if keep_resolved_entries:
+                resolved_entries[i] = (playlist_index, entry_result)
  
          # Update with processed data
-        ie_result['requested_entries'], ie_result['entries'] = tuple(zip(*entries)) or ([], [])
+        ie_result['requested_entries'], ie_result['entries'] = tuple(zip(*resolved_entries)) or ([], [])
  
          # Write the updated info to json
          if _infojson_written is True and self._write_info_json(
@@ -1857,7 +1923,7 @@ def can_merge():
              and (
                  not can_merge()
                  or info_dict.get('is_live') and not self.params.get('live_from_start')
-                or self.outtmpl_dict['default'] == '-'))
+                or self.params['outtmpl']['default'] == '-'))
          compat = (
              prefer_best
              or self.params.get('allow_multiple_audio_streams', False)
@@ -2333,10 +2399,10 @@ def report_force_conversion(field, field_not, conversion):
  
          def sanitize_string_field(info, string_field):
              field = info.get(string_field)
-            if field is None or isinstance(field, compat_str):
+            if field is None or isinstance(field, str):
                  return
              report_force_conversion(string_field, 'a string', 'string')
-            info[string_field] = compat_str(field)
+            info[string_field] = str(field)
  
          def sanitize_numeric_fields(info):
              for numeric_field in self._NUMERIC_FIELDS:
@@ -2348,9 +2414,25 @@ def sanitize_numeric_fields(info):
  
          sanitize_string_field(info_dict, 'id')
          sanitize_numeric_fields(info_dict)
+        if info_dict.get('section_end') and info_dict.get('section_start') is not None:
+            info_dict['duration'] = round(info_dict['section_end'] - info_dict['section_start'], 3)
          if (info_dict.get('duration') or 0) <= 0 and info_dict.pop('duration', None):
              self.report_warning('"duration" field is negative, there is an error in extractor')
  
+        chapters = info_dict.get('chapters') or []
+        if chapters and chapters[0].get('start_time'):
+            chapters.insert(0, {'start_time': 0})
+
+        dummy_chapter = {'end_time': 0, 'start_time': info_dict.get('duration')}
+        for idx, (prev, current, next_) in enumerate(zip(
+                (dummy_chapter, *chapters), chapters, (*chapters[1:], dummy_chapter)), 1):
+            if current.get('start_time') is None:
+                current['start_time'] = prev.get('end_time')
+            if not current.get('end_time'):
+                current['end_time'] = next_.get('start_time')
+            if not current.get('title'):
+                current['title'] = f'<Untitled Chapter {idx}>'
+
          if 'playlist' not in info_dict:
              # It isn't part of a playlist
              info_dict['playlist'] = None
@@ -2437,7 +2519,7 @@ def is_wellformed(f):
              sanitize_numeric_fields(format)
              format['url'] = sanitize_url(format['url'])
              if not format.get('format_id'):
-                format['format_id'] = compat_str(i)
+                format['format_id'] = str(i)
              else:
                  # Sanitize format_id from characters used in format selector expression
                  format['format_id'] = re.sub(r'[\s,/+\[\]()]', '_', format['format_id'])
@@ -2583,10 +2665,11 @@ def to_screen(*msg):
              for fmt, chapter in itertools.product(formats_to_download, requested_ranges or [{}]):
                  new_info = self._copy_infodict(info_dict)
                  new_info.update(fmt)
-                if chapter:
+                offset, duration = info_dict.get('section_start') or 0, info_dict.get('duration') or float('inf')
+                if chapter or offset:
                      new_info.update({
-                        'section_start': chapter.get('start_time'),
-                        'section_end': chapter.get('end_time', 0),
+                        'section_start': offset + chapter.get('start_time', 0),
+                        'section_end': offset + min(chapter.get('end_time', duration), duration),
                          'section_title': chapter.get('title'),
                          'section_number': chapter.get('index'),
                      })
@@ -2963,13 +3046,12 @@ def existing_video_file(*filepaths):
                          info_dict['ext'] = os.path.splitext(file)[1][1:]
                      return file
  
-                success = True
-                merger, fd = FFmpegMergerPP(self), None
+                fd, success = None, True
                  if info_dict.get('protocol') or info_dict.get('url'):
                      fd = get_suitable_downloader(info_dict, self.params, to_stdout=temp_filename == '-')
                      if fd is not FFmpegFD and (
                              info_dict.get('section_start') or info_dict.get('section_end')):
-                        msg = ('This format cannot be partially downloaded' if merger.available
+                        msg = ('This format cannot be partially downloaded' if FFmpegFD.available()
                                 else 'You have requested downloading the video partially, but ffmpeg is not installed')
                          self.report_error(f'{msg}. Aborting')
                          return
@@ -3028,6 +3110,7 @@ def correct_ext(filename, ext=new_ext):
                      dl_filename = existing_video_file(full_filename, temp_filename)
                      info_dict['__real_download'] = False
  
+                    merger = FFmpegMergerPP(self)
                      downloaded = []
                      if dl_filename is not None:
                          self.report_file_already_downloaded(dl_filename)
@@ -3138,22 +3221,23 @@ def ffmpeg_fixup(cndn, msg, cls):
                              self.report_warning(f'{vid}: {msg}. Install ffmpeg to fix this automatically')
  
                      stretched_ratio = info_dict.get('stretched_ratio')
-                    ffmpeg_fixup(
-                        stretched_ratio not in (1, None),
-                        f'Non-uniform pixel ratio {stretched_ratio}',
-                        FFmpegFixupStretchedPP)
-
-                    ffmpeg_fixup(
-                        (info_dict.get('requested_formats') is None
-                         and info_dict.get('container') == 'm4a_dash'
-                         and info_dict.get('ext') == 'm4a'),
-                        'writing DASH m4a. Only some players support this container',
-                        FFmpegFixupM4aPP)
+                    ffmpeg_fixup(stretched_ratio not in (1, None),
+                                 f'Non-uniform pixel ratio {stretched_ratio}',
+                                 FFmpegFixupStretchedPP)
  
                      downloader = get_suitable_downloader(info_dict, self.params) if 'protocol' in info_dict else None
                      downloader = downloader.FD_NAME if downloader else None
  
-                    if info_dict.get('requested_formats') is None:  # Not necessary if doing merger
+                    ext = info_dict.get('ext')
+                    postprocessed_by_ffmpeg = info_dict.get('requested_formats') or any((
+                        isinstance(pp, FFmpegVideoConvertorPP)
+                        and resolve_recode_mapping(ext, pp.mapping)[0] not in (ext, None)
+                    ) for pp in self._pps['post_process'])
+
+                    if not postprocessed_by_ffmpeg:
+                        ffmpeg_fixup(ext == 'm4a' and info_dict.get('container') == 'm4a_dash',
+                                     'writing DASH m4a. Only some players support this container',
+                                     FFmpegFixupM4aPP)
                          ffmpeg_fixup(downloader == 'hlsnative' and not self.params.get('hls_use_mpegts')
                                       or info_dict.get('is_live') and self.params.get('hls_use_mpegts') is None,
                                       'Possible MPEG-TS in MP4 container or malformed AAC timestamps',
@@ -3203,7 +3287,7 @@ def wrapper(*args, **kwargs):
      def download(self, url_list):
          """Download a given list of URLs."""
          url_list = variadic(url_list)  # Passing a single URL is a common mistake
-        outtmpl = self.outtmpl_dict['default']
+        outtmpl = self.params['outtmpl']['default']
          if (len(url_list) > 1
                  and outtmpl != '-'
                  and '%' not in outtmpl
@@ -3477,28 +3561,39 @@ def render_formats_table(self, info_dict):
                  ] for f in formats if f.get('preference') is None or f['preference'] >= -1000]
              return render_table(['format code', 'extension', 'resolution', 'note'], table, extra_gap=1)
  
+        def simplified_codec(f, field):
+            assert field in ('acodec', 'vcodec')
+            codec = f.get(field, 'unknown')
+            if not codec:
+                return 'unknown'
+            elif codec != 'none':
+                return '.'.join(codec.split('.')[:4])
+
+            if field == 'vcodec' and f.get('acodec') == 'none':
+                return 'images'
+            elif field == 'acodec' and f.get('vcodec') == 'none':
+                return ''
+            return self._format_out('audio only' if field == 'vcodec' else 'video only',
+                                    self.Styles.SUPPRESS)
+
          delim = self._format_out('\u2502', self.Styles.DELIM, '|', test_encoding=True)
          table = [
              [
                  self._format_out(format_field(f, 'format_id'), self.Styles.ID),
                  format_field(f, 'ext'),
                  format_field(f, func=self.format_resolution, ignore=('audio only', 'images')),
-                format_field(f, 'fps', '\t%d'),
+                format_field(f, 'fps', '\t%d', func=round),
                  format_field(f, 'dynamic_range', '%s', ignore=(None, 'SDR')).replace('HDR', ''),
                  delim,
                  format_field(f, 'filesize', ' \t%s', func=format_bytes) + format_field(f, 'filesize_approx', '~\t%s', func=format_bytes),
-                format_field(f, 'tbr', '\t%dk'),
+                format_field(f, 'tbr', '\t%dk', func=round),
                  shorten_protocol_name(f.get('protocol', '')),
                  delim,
-                format_field(f, 'vcodec', default='unknown').replace(
-                    'none', 'images' if f.get('acodec') == 'none'
-                            else self._format_out('audio only', self.Styles.SUPPRESS)),
-                format_field(f, 'vbr', '\t%dk'),
-                format_field(f, 'acodec', default='unknown').replace(
-                    'none', '' if f.get('vcodec') == 'none'
-                            else self._format_out('video only', self.Styles.SUPPRESS)),
-                format_field(f, 'abr', '\t%dk'),
-                format_field(f, 'asr', '\t%dHz'),
+                simplified_codec(f, 'vcodec'),
+                format_field(f, 'vbr', '\t%dk', func=round),
+                simplified_codec(f, 'acodec'),
+                format_field(f, 'abr', '\t%dk', func=round),
+                format_field(f, 'asr', '\t%s', func=format_decimal_suffix),
                  join_nonempty(
                      self._format_out('UNSUPPORTED', 'light red') if f.get('ext') in ('f4f', 'f4m') else None,
                      format_field(f, 'language', '[%s]'),
@@ -3622,17 +3717,7 @@ def get_encoding(stream):
                  with contextlib.suppress(Exception):
                      sys.exc_clear()
  
-        def python_implementation():
-            impl_name = platform.python_implementation()
-            if impl_name == 'PyPy' and hasattr(sys, 'pypy_version_info'):
-                return impl_name + ' version %d.%d.%d' % sys.pypy_version_info[:3]
-            return impl_name
-
-        write_debug('Python version %s (%s %s) - %s' % (
-            platform.python_version(),
-            python_implementation(),
-            platform.architecture()[0],
-            platform_name()))
+        write_debug(system_identifier())
  
          exe_versions, ffmpeg_features = FFmpegPostProcessor.get_versions_and_features(self)
          ffmpeg_features = {key for key, val in ffmpeg_features.items() if val}
@@ -3691,7 +3776,7 @@ def _setup_opener(self):
              else:
                  proxies = {'http': opts_proxy, 'https': opts_proxy}
          else:
-            proxies = compat_urllib_request.getproxies()
+            proxies = urllib.request.getproxies()
              # Set HTTPS proxy to HTTP one if given (https://github.com/ytdl-org/youtube-dl/issues/805)
              if 'http' in proxies and 'https' not in proxies:
                  proxies['https'] = proxies['http']
@@ -3707,13 +3792,13 @@ def _setup_opener(self):
          # default FileHandler and allows us to disable the file protocol, which
          # can be used for malicious purposes (see
          # https://github.com/ytdl-org/youtube-dl/issues/8227)
-        file_handler = compat_urllib_request.FileHandler()
+        file_handler = urllib.request.FileHandler()
  
          def file_open(*args, **kwargs):
-            raise compat_urllib_error.URLError('file:// scheme is explicitly disabled in yt-dlp for security reasons')
+            raise urllib.error.URLError('file:// scheme is explicitly disabled in yt-dlp for security reasons')
          file_handler.file_open = file_open
  
-        opener = compat_urllib_request.build_opener(
+        opener = urllib.request.build_opener(
              proxy_handler, https_handler, cookie_processor, ydlh, redirect_handler, data_handler, file_handler)
  
          # Delete the default user-agent header, which would otherwise apply in