Let `--match-filter` reject entries early

[yt-dlp.git] / yt_dlp / YoutubeDL.py
diff --git a/yt_dlp/YoutubeDL.py b/yt_dlp/YoutubeDL.py

index 7edae6fa246e38ef2bd3a0ce4f93ea3b136487f9..eef3f8b4ca42b2021cc0aa288b9eae6066fe62e6 100644 (file)
--- a/yt_dlp/YoutubeDL.py
+++ b/yt_dlp/YoutubeDL.py
@@ -1117,12 +1117,15 @@ def check_filter():
              if age_restricted(info_dict.get('age_limit'), self.params.get('age_limit')):
                  return 'Skipping "%s" because it is age restricted' % video_title
  
-            if not incomplete:
-                match_filter = self.params.get('match_filter')
-                if match_filter is not None:
-                    ret = match_filter(info_dict)
-                    if ret is not None:
-                        return ret
+            match_filter = self.params.get('match_filter')
+            if match_filter is not None:
+                try:
+                    ret = match_filter(info_dict, incomplete=incomplete)
+                except TypeError:
+                    # For backward compatibility
+                    ret = None if incomplete else match_filter(info_dict)
+                if ret is not None:
+                    return ret
              return None
  
          if self.in_download_archive(info_dict):
@@ -2231,7 +2234,7 @@ def is_wellformed(f):
          if self.params.get('list_thumbnails'):
              self.list_thumbnails(info_dict)
          if self.params.get('listformats'):
-            if not info_dict.get('formats'):
+            if not info_dict.get('formats') and not info_dict.get('url'):
                  raise ExtractorError('No video formats found', expected=True)
              self.list_formats(info_dict)
          if self.params.get('listsubtitles'):
@@ -2339,7 +2342,8 @@ def process_subtitles(self, video_id, normal_subtitles, automatic_captions):
              requested_langs = ['en']
          else:
              requested_langs = [list(all_sub_langs)[0]]
-        self.write_debug('Downloading subtitles: %s' % ', '.join(requested_langs))
+        if requested_langs:
+            self.write_debug('Downloading subtitles: %s' % ', '.join(requested_langs))
  
          formats_query = self.params.get('subtitlesformat', 'best')
          formats_preference = formats_query.split('/') if formats_query else []
@@ -2662,7 +2666,6 @@ def existing_file(*filepaths):
                              os.remove(encodeFilename(file))
                          return None
  
-                    self.report_file_already_downloaded(existing_files[0])
                      info_dict['ext'] = os.path.splitext(existing_files[0])[1][1:]
                      return existing_files[0]
  
@@ -2717,7 +2720,7 @@ def correct_ext(filename, ext=new_ext):
                          info_dict['protocol'] = _protocols.pop()
                      directly_mergable = FFmpegFD.can_merge_formats(info_dict)
                      if dl_filename is not None:
-                        pass
+                        self.report_file_already_downloaded(dl_filename)
                      elif (directly_mergable and get_suitable_downloader(
                              info_dict, self.params, to_stdout=(temp_filename == '-')) == FFmpegFD):
                          info_dict['url'] = '\n'.join(f['url'] for f in requested_formats)
@@ -2769,9 +2772,13 @@ def correct_ext(filename, ext=new_ext):
                  else:
                      # Just a single file
                      dl_filename = existing_file(full_filename, temp_filename)
-                    if dl_filename is None:
+                    if dl_filename is None or dl_filename == temp_filename:
+                        # dl_filename == temp_filename could mean that the file was partially downloaded with --no-part.
+                        # So we should try to resume the download
                          success, real_download = self.dl(temp_filename, info_dict)
                          info_dict['__real_download'] = real_download
+                    else:
+                        self.report_file_already_downloaded(dl_filename)
  
                  dl_filename = dl_filename or temp_filename
                  info_dict['__finaldir'] = os.path.dirname(os.path.abspath(encodeFilename(full_filename)))
@@ -2869,13 +2876,13 @@ def download(self, url_list):
              except UnavailableVideoError:
                  self.report_error('unable to download video')
              except MaxDownloadsReached:
-                self.to_screen('[info] Maximum number of downloaded files reached')
+                self.to_screen('[info] Maximum number of downloads reached')
                  raise
              except ExistingVideoReached:
-                self.to_screen('[info] Encountered a file that is already in the archive, stopping due to --break-on-existing')
+                self.to_screen('[info] Encountered a video that is already in the archive, stopping due to --break-on-existing')
                  raise
              except RejectedVideoReached:
-                self.to_screen('[info] Encountered a file that did not match filter, stopping due to --break-on-reject')
+                self.to_screen('[info] Encountered a video that did not match filter, stopping due to --break-on-reject')
                  raise
              else:
                  if self.params.get('dump_single_json', False):
@@ -2904,6 +2911,8 @@ def download_with_info_file(self, info_filename):
      @staticmethod
      def sanitize_info(info_dict, remove_private_keys=False):
          ''' Sanitize the infodict for converting to json '''
+        if info_dict is None:
+            return info_dict
          info_dict.setdefault('epoch', int(time.time()))
          remove_keys = {'__original_infodict'}  # Always remove this since this may contain a copy of the entire dict
          keep_keys = ['_type'],  # Always keep this to facilitate load-info-json
@@ -3256,13 +3265,13 @@ def python_implementation():
          from .postprocessor.embedthumbnail import has_mutagen
          from .cookies import SQLITE_AVAILABLE, KEYRING_AVAILABLE
  
-        lib_str = ', '.join(filter(None, (
+        lib_str = ', '.join(sorted(filter(None, (
              can_decrypt_frag and 'pycryptodome',
              has_websockets and 'websockets',
              has_mutagen and 'mutagen',
              SQLITE_AVAILABLE and 'sqlite',
              KEYRING_AVAILABLE and 'keyring',
-        ))) or 'none'
+        )))) or 'none'
          self._write_string('[debug] Optional libraries: %s\n' % lib_str)
  
          proxy_map = {}