Let `--match-filter` reject entries early

[yt-dlp.git] / yt_dlp / YoutubeDL.py
diff --git a/yt_dlp/YoutubeDL.py b/yt_dlp/YoutubeDL.py

index ac99dd45b58662ad28b6e5a8e734f9b1f8d36f7b..eef3f8b4ca42b2021cc0aa288b9eae6066fe62e6 100644 (file)
--- a/yt_dlp/YoutubeDL.py
+++ b/yt_dlp/YoutubeDL.py
@@ -220,7 +220,7 @@ class YoutubeDL(object):
                         'temp' and the keys of OUTTMPL_TYPES (in utils.py)
      outtmpl:           Dictionary of templates for output names. Allowed keys
                         are 'default' and the keys of OUTTMPL_TYPES (in utils.py).
-                       A string a also accepted for backward compatibility
+                       For compatibility with youtube-dl, a single string can also be used
      outtmpl_na_placeholder: Placeholder for unavailable meta fields.
      restrictfilenames: Do not allow "&" and spaces in file names
      trim_file_name:    Limit length of filename (extension excluded)
@@ -234,6 +234,8 @@ class YoutubeDL(object):
      overwrites:        Overwrite all video and metadata files if True,
                         overwrite only non-video files if None
                         and don't overwrite any file if False
+                       For compatibility with youtube-dl,
+                       "nooverwrites" may also be used instead
      playliststart:     Playlist item to start at.
      playlistend:       Playlist item to end at.
      playlist_items:    Specific indices of playlist to download.
@@ -246,7 +248,7 @@ class YoutubeDL(object):
      writedescription:  Write the video description to a .description file
      writeinfojson:     Write the video description to a .info.json file
      clean_infojson:    Remove private fields from the infojson
-    writecomments:     Extract video comments. This will not be written to disk
+    getcomments:       Extract video comments. This will not be written to disk
                         unless writeinfojson is also given
      writeannotations:  Write the video annotations to a .annotations.xml file
      writethumbnail:    Write the thumbnail image to a file
@@ -420,10 +422,12 @@ class YoutubeDL(object):
      ffmpeg_location:   Location of the ffmpeg/avconv binary; either the path
                         to the binary or its containing directory.
      postprocessor_args: A dictionary of postprocessor/executable keys (in lower case)
-                        and a list of additional command-line arguments for the
-                        postprocessor/executable. The dict can also have "PP+EXE" keys
-                        which are used when the given exe is used by the given PP.
-                        Use 'default' as the name for arguments to passed to all PP
+                       and a list of additional command-line arguments for the
+                       postprocessor/executable. The dict can also have "PP+EXE" keys
+                       which are used when the given exe is used by the given PP.
+                       Use 'default' as the name for arguments to passed to all PP
+                       For compatibility with youtube-dl, a single list of args
+                       can also be used
  
      The following options are used by the extractors:
      extractor_retries: Number of times to retry for known errors
@@ -515,8 +519,15 @@ def check_deprecated(param, option, suggestion):
                  self.report_warning('--merge-output-format will be ignored since --remux-video or --recode-video is given')
              self.params['merge_output_format'] = self.params['final_ext']
  
-        if 'overwrites' in self.params and self.params['overwrites'] is None:
-            del self.params['overwrites']
+        if self.params.get('overwrites') is None:
+            self.params.pop('overwrites', None)
+        elif self.params.get('nooverwrites') is not None:
+            # nooverwrites was unnecessarily changed to overwrites
+            # in 0c3d0f51778b153f65c21906031c2e091fcfb641
+            # This ensures compatibility with both keys
+            self.params['overwrites'] = not self.params['nooverwrites']
+        else:
+            self.params['nooverwrites'] = not self.params['overwrites']
  
          if params.get('bidi_workaround', False):
              try:
@@ -889,7 +900,6 @@ def validate_outtmpl(cls, outtmpl):
      def prepare_outtmpl(self, outtmpl, info_dict, sanitize=None):
          """ Make the template and info_dict suitable for substitution : ydl.outtmpl_escape(outtmpl) % info_dict """
          info_dict.setdefault('epoch', int(time.time()))  # keep epoch consistent once set
-        na = self.params.get('outtmpl_na_placeholder', 'NA')
  
          info_dict = dict(info_dict)  # Do not sanitize so as not to consume LazyList
          for key in ('__original_infodict', '__postprocessors'):
@@ -970,6 +980,8 @@ def get_value(mdict):
  
              return value
  
+        na = self.params.get('outtmpl_na_placeholder', 'NA')
+
          def _dumpjson_default(obj):
              if isinstance(obj, (set, LazyList)):
                  return list(obj)
@@ -978,10 +990,7 @@ def _dumpjson_default(obj):
          def create_key(outer_mobj):
              if not outer_mobj.group('has_key'):
                  return f'%{outer_mobj.group(0)}'
-
-            prefix = outer_mobj.group('prefix')
              key = outer_mobj.group('key')
-            original_fmt = fmt = outer_mobj.group('format')
              mobj = re.match(INTERNAL_FORMAT_RE, key)
              if mobj is None:
                  value, default, mobj = None, na, {'fields': ''}
@@ -990,6 +999,7 @@ def create_key(outer_mobj):
                  default = mobj['default'] if mobj['default'] is not None else na
                  value = get_value(mobj)
  
+            fmt = outer_mobj.group('format')
              if fmt == 's' and value is not None and key in field_size_compat_map.keys():
                  fmt = '0{:d}d'.format(field_size_compat_map[key])
  
@@ -1021,9 +1031,9 @@ def create_key(outer_mobj):
                  if fmt[-1] in 'csr':
                      value = sanitize(mobj['fields'].split('.')[-1], value)
  
-            key = '%s\0%s' % (key.replace('%', '%\0'), original_fmt)
+            key = '%s\0%s' % (key.replace('%', '%\0'), outer_mobj.group('format'))
              TMPL_DICT[key] = value
-            return f'{prefix}%({key}){fmt}'
+            return '{prefix}%({key}){fmt}'.format(key=key, fmt=fmt, prefix=outer_mobj.group('prefix'))
  
          return EXTERNAL_FORMAT_RE.sub(create_key, outtmpl), TMPL_DICT
  
@@ -1069,7 +1079,6 @@ def prepare_filename(self, info_dict, dir_type='', warn=False):
                  self.report_warning('--paths is ignored when an outputting to stdout', only_once=True)
              elif os.path.isabs(filename):
                  self.report_warning('--paths is ignored since an absolute path is given in output template', only_once=True)
-            self.__prepare_filename_warned = True
          if filename == '-' or not filename:
              return filename
  
@@ -1108,12 +1117,15 @@ def check_filter():
              if age_restricted(info_dict.get('age_limit'), self.params.get('age_limit')):
                  return 'Skipping "%s" because it is age restricted' % video_title
  
-            if not incomplete:
-                match_filter = self.params.get('match_filter')
-                if match_filter is not None:
-                    ret = match_filter(info_dict)
-                    if ret is not None:
-                        return ret
+            match_filter = self.params.get('match_filter')
+            if match_filter is not None:
+                try:
+                    ret = match_filter(info_dict, incomplete=incomplete)
+                except TypeError:
+                    # For backward compatibility
+                    ret = None if incomplete else match_filter(info_dict)
+                if ret is not None:
+                    return ret
              return None
  
          if self.in_download_archive(info_dict):
@@ -1272,7 +1284,7 @@ def process_ie_result(self, ie_result, download=True, extra_info={}):
              ie_result = self.process_video_result(ie_result, download=download)
              additional_urls = (ie_result or {}).get('additional_urls')
              if additional_urls:
-                # TODO: Improve MetadataFromFieldPP to allow setting a list
+                # TODO: Improve MetadataParserPP to allow setting a list
                  if isinstance(additional_urls, compat_str):
                      additional_urls = [additional_urls]
                  self.to_screen(
@@ -1348,15 +1360,12 @@ def process_ie_result(self, ie_result, download=True, extra_info={}):
                  'It needs to be updated.' % ie_result.get('extractor'))
  
              def _fixup(r):
-                self.add_extra_info(
-                    r,
-                    {
-                        'extractor': ie_result['extractor'],
-                        'webpage_url': ie_result['webpage_url'],
-                        'webpage_url_basename': url_basename(ie_result['webpage_url']),
-                        'extractor_key': ie_result['extractor_key'],
-                    }
-                )
+                self.add_extra_info(r, {
+                    'extractor': ie_result['extractor'],
+                    'webpage_url': ie_result['webpage_url'],
+                    'webpage_url_basename': url_basename(ie_result['webpage_url']),
+                    'extractor_key': ie_result['extractor_key'],
+                })
                  return r
              ie_result['entries'] = [
                  self.process_ie_result(_fixup(r), download, extra_info)
@@ -2193,7 +2202,7 @@ def is_wellformed(f):
                  format['format'] = '{id} - {res}{note}'.format(
                      id=format['format_id'],
                      res=self.format_resolution(format),
-                    note=' ({0})'.format(format['format_note']) if format.get('format_note') is not None else '',
+                    note=format_field(format, 'format_note', ' (%s)'),
                  )
              # Automatically determine file extension if missing
              if format.get('ext') is None:
@@ -2225,7 +2234,7 @@ def is_wellformed(f):
          if self.params.get('list_thumbnails'):
              self.list_thumbnails(info_dict)
          if self.params.get('listformats'):
-            if not info_dict.get('formats'):
+            if not info_dict.get('formats') and not info_dict.get('url'):
                  raise ExtractorError('No video formats found', expected=True)
              self.list_formats(info_dict)
          if self.params.get('listsubtitles'):
@@ -2333,7 +2342,8 @@ def process_subtitles(self, video_id, normal_subtitles, automatic_captions):
              requested_langs = ['en']
          else:
              requested_langs = [list(all_sub_langs)[0]]
-        self.write_debug('Downloading subtitles: %s' % ', '.join(requested_langs))
+        if requested_langs:
+            self.write_debug('Downloading subtitles: %s' % ', '.join(requested_langs))
  
          formats_query = self.params.get('subtitlesformat', 'best')
          formats_preference = formats_query.split('/') if formats_query else []
@@ -2395,7 +2405,7 @@ def print_optional(field):
          print_optional('thumbnail')
          print_optional('description')
          print_optional('filename')
-        if self.params.get('forceduration', False) and info_dict.get('duration') is not None:
+        if self.params.get('forceduration') and info_dict.get('duration') is not None:
              self.to_stdout(formatSeconds(info_dict['duration']))
          print_mandatory('format')
  
@@ -2435,8 +2445,6 @@ def process_info(self, info_dict):
  
          assert info_dict.get('_type', 'video') == 'video'
  
-        info_dict.setdefault('__postprocessors', [])
-
          max_downloads = self.params.get('max_downloads')
          if max_downloads is not None:
              if self._num_downloads >= int(max_downloads):
@@ -2637,6 +2645,7 @@ def _write_link_file(extension, template, newline, embed_filename):
              info_dict = self.run_pp(MoveFilesAfterDownloadPP(self, False), info_dict)
          else:
              # Download
+            info_dict.setdefault('__postprocessors', [])
              try:
  
                  def existing_file(*filepaths):
@@ -2657,7 +2666,6 @@ def existing_file(*filepaths):
                              os.remove(encodeFilename(file))
                          return None
  
-                    self.report_file_already_downloaded(existing_files[0])
                      info_dict['ext'] = os.path.splitext(existing_files[0])[1][1:]
                      return existing_files[0]
  
@@ -2712,7 +2720,7 @@ def correct_ext(filename, ext=new_ext):
                          info_dict['protocol'] = _protocols.pop()
                      directly_mergable = FFmpegFD.can_merge_formats(info_dict)
                      if dl_filename is not None:
-                        pass
+                        self.report_file_already_downloaded(dl_filename)
                      elif (directly_mergable and get_suitable_downloader(
                              info_dict, self.params, to_stdout=(temp_filename == '-')) == FFmpegFD):
                          info_dict['url'] = '\n'.join(f['url'] for f in requested_formats)
@@ -2764,9 +2772,13 @@ def correct_ext(filename, ext=new_ext):
                  else:
                      # Just a single file
                      dl_filename = existing_file(full_filename, temp_filename)
-                    if dl_filename is None:
+                    if dl_filename is None or dl_filename == temp_filename:
+                        # dl_filename == temp_filename could mean that the file was partially downloaded with --no-part.
+                        # So we should try to resume the download
                          success, real_download = self.dl(temp_filename, info_dict)
                          info_dict['__real_download'] = real_download
+                    else:
+                        self.report_file_already_downloaded(dl_filename)
  
                  dl_filename = dl_filename or temp_filename
                  info_dict['__finaldir'] = os.path.dirname(os.path.abspath(encodeFilename(full_filename)))
@@ -2864,13 +2876,13 @@ def download(self, url_list):
              except UnavailableVideoError:
                  self.report_error('unable to download video')
              except MaxDownloadsReached:
-                self.to_screen('[info] Maximum number of downloaded files reached')
+                self.to_screen('[info] Maximum number of downloads reached')
                  raise
              except ExistingVideoReached:
-                self.to_screen('[info] Encountered a file that is already in the archive, stopping due to --break-on-existing')
+                self.to_screen('[info] Encountered a video that is already in the archive, stopping due to --break-on-existing')
                  raise
              except RejectedVideoReached:
-                self.to_screen('[info] Encountered a file that did not match filter, stopping due to --break-on-reject')
+                self.to_screen('[info] Encountered a video that did not match filter, stopping due to --break-on-reject')
                  raise
              else:
                  if self.params.get('dump_single_json', False):
@@ -2899,6 +2911,8 @@ def download_with_info_file(self, info_filename):
      @staticmethod
      def sanitize_info(info_dict, remove_private_keys=False):
          ''' Sanitize the infodict for converting to json '''
+        if info_dict is None:
+            return info_dict
          info_dict.setdefault('epoch', int(time.time()))
          remove_keys = {'__original_infodict'}  # Always remove this since this may contain a copy of the entire dict
          keep_keys = ['_type'],  # Always keep this to facilitate load-info-json
@@ -3187,11 +3201,6 @@ def print_debug_header(self):
          if not self.params.get('verbose'):
              return
  
-        if type('') is not compat_str:
-            # Python 2.6 on SLES11 SP1 (https://github.com/ytdl-org/youtube-dl/issues/3326)
-            self.report_warning(
-                'Your Python is broken! Update to a newer and supported version')
-
          stdout_encoding = getattr(
              sys.stdout, 'encoding', 'missing (%s)' % type(sys.stdout).__name__)
          encoding_str = (
@@ -3247,14 +3256,24 @@ def python_implementation():
          exe_versions['rtmpdump'] = rtmpdump_version()
          exe_versions['phantomjs'] = PhantomJSwrapper._version()
          exe_str = ', '.join(
-            '%s %s' % (exe, v)
-            for exe, v in sorted(exe_versions.items())
-            if v
-        )
-        if not exe_str:
-            exe_str = 'none'
+            f'{exe} {v}' for exe, v in sorted(exe_versions.items()) if v
+        ) or 'none'
          self._write_string('[debug] exe versions: %s\n' % exe_str)
  
+        from .downloader.fragment import can_decrypt_frag
+        from .downloader.websocket import has_websockets
+        from .postprocessor.embedthumbnail import has_mutagen
+        from .cookies import SQLITE_AVAILABLE, KEYRING_AVAILABLE
+
+        lib_str = ', '.join(sorted(filter(None, (
+            can_decrypt_frag and 'pycryptodome',
+            has_websockets and 'websockets',
+            has_mutagen and 'mutagen',
+            SQLITE_AVAILABLE and 'sqlite',
+            KEYRING_AVAILABLE and 'keyring',
+        )))) or 'none'
+        self._write_string('[debug] Optional libraries: %s\n' % lib_str)
+
          proxy_map = {}
          for handler in self._opener.handlers:
              if hasattr(handler, 'proxies'):