[ExtractAudio] Allow conditional conversion

[yt-dlp.git] / yt_dlp / postprocessor / modify_chapters.py
diff --git a/yt_dlp/postprocessor/modify_chapters.py b/yt_dlp/postprocessor/modify_chapters.py

index a0818c41ba8ed32e5921440379f60e2cbfcf84b1..de3505e11b83b6ddd14ebc9990fe779d88fea386 100644 (file)
--- a/yt_dlp/postprocessor/modify_chapters.py
+++ b/yt_dlp/postprocessor/modify_chapters.py
@@ -3,17 +3,9 @@
  import os
  
  from .common import PostProcessor
-from .ffmpeg import (
-    FFmpegPostProcessor,
-    FFmpegSubtitlesConvertorPP
-)
+from .ffmpeg import FFmpegPostProcessor, FFmpegSubtitlesConvertorPP
  from .sponsorblock import SponsorBlockPP
-from ..utils import (
-    orderedSet,
-    PostProcessingError,
-    prepend_extension,
-)
-
+from ..utils import PostProcessingError, orderedSet, prepend_extension
  
  _TINY_CHAPTER_DURATION = 1
  DEFAULT_SPONSORBLOCK_CHAPTER_TITLE = '[SponsorBlock]: %(category_names)l'
@@ -24,27 +16,29 @@ def __init__(self, downloader, remove_chapters_patterns=None, remove_sponsor_seg
                   *, sponsorblock_chapter_title=DEFAULT_SPONSORBLOCK_CHAPTER_TITLE, force_keyframes=False):
          FFmpegPostProcessor.__init__(self, downloader)
          self._remove_chapters_patterns = set(remove_chapters_patterns or [])
-        self._remove_sponsor_segments = set(remove_sponsor_segments or [])
+        self._remove_sponsor_segments = set(remove_sponsor_segments or []) - set(SponsorBlockPP.POI_CATEGORIES.keys())
          self._ranges_to_remove = set(remove_ranges or [])
          self._sponsorblock_chapter_title = sponsorblock_chapter_title
          self._force_keyframes = force_keyframes
  
      @PostProcessor._restrict_to(images=False)
      def run(self, info):
+        # Chapters must be preserved intact when downloading multiple formats of the same video.
          chapters, sponsor_chapters = self._mark_chapters_to_remove(
-            info.get('chapters') or [], info.get('sponsorblock_chapters') or [])
+            copy.deepcopy(info.get('chapters')) or [],
+            copy.deepcopy(info.get('sponsorblock_chapters')) or [])
          if not chapters and not sponsor_chapters:
              return [], info
  
-        real_duration = self._get_real_video_duration(info)
+        real_duration = self._get_real_video_duration(info['filepath'])
          if not chapters:
-            chapters = [{'start_time': 0, 'end_time': real_duration, 'title': info['title']}]
+            chapters = [{'start_time': 0, 'end_time': info.get('duration') or real_duration, 'title': info['title']}]
  
          info['chapters'], cuts = self._remove_marked_arrange_sponsors(chapters + sponsor_chapters)
          if not cuts:
              return [], info
  
-        if self._duration_mismatch(real_duration, info.get('duration')):
+        if self._duration_mismatch(real_duration, info.get('duration'), 1):
              if not self._duration_mismatch(real_duration, info['chapters'][-1]['end_time']):
                  self.to_screen(f'Skipping {self.pp_key()} since the video appears to be already cut')
                  return [], info
@@ -55,6 +49,7 @@ def run(self, info):
                  self.write_debug('Expected and actual durations mismatch')
  
          concat_opts = self._make_concat_opts(cuts, real_duration)
+        self.write_debug('Concat spec = %s' % ', '.join(f'{c.get("inpoint", 0.0)}-{c.get("outpoint", "inf")}' for c in concat_opts))
  
          def remove_chapters(file, is_sub):
              return file, self.remove_chapters(file, cuts, concat_opts, self._force_keyframes and not is_sub)
@@ -65,12 +60,13 @@ def remove_chapters(file, is_sub):
          # Renaming should only happen after all files are processed
          files_to_remove = []
          for in_file, out_file in in_out_files:
+            mtime = os.stat(in_file).st_mtime
              uncut_file = prepend_extension(in_file, 'uncut')
              os.replace(in_file, uncut_file)
              os.replace(out_file, in_file)
+            self.try_utime(in_file, mtime, mtime)
              files_to_remove.append(uncut_file)
  
-        info['_real_duration'] = info['chapters'][-1]['end_time']
          return files_to_remove, info
  
      def _mark_chapters_to_remove(self, chapters, sponsor_chapters):
@@ -126,7 +122,7 @@ def _remove_marked_arrange_sponsors(self, chapters):
          cuts = []
  
          def append_cut(c):
-            assert 'remove' in c
+            assert 'remove' in c, 'Not a cut is appended to cuts'
              last_to_cut = cuts[-1] if cuts else None
              if last_to_cut and last_to_cut['end_time'] >= c['start_time']:
                  last_to_cut['end_time'] = max(last_to_cut['end_time'], c['end_time'])
@@ -154,7 +150,7 @@ def excess_duration(c):
          new_chapters = []
  
          def append_chapter(c):
-            assert 'remove' not in c
+            assert 'remove' not in c, 'Cut is appended to chapters'
              length = c['end_time'] - c['start_time'] - excess_duration(c)
              # Chapter is completely covered by cuts or sponsors.
              if length <= 0:
@@ -237,7 +233,7 @@ def append_chapter(c):
                      heapq.heappush(chapters, (c['start_time'], i, c))
              # (normal, sponsor) and (sponsor, sponsor)
              else:
-                assert '_categories' in c
+                assert '_categories' in c, 'Normal chapters overlap'
                  cur_chapter['_was_cut'] = True
                  c['_was_cut'] = True
                  # Push the part after the sponsor to PQ.
@@ -301,7 +297,7 @@ def _remove_tiny_rename_sponsors(self, chapters):
                      'name': SponsorBlockPP.CATEGORIES[category],
                      'category_names': [SponsorBlockPP.CATEGORIES[c] for c in cats]
                  })
-                c['title'] = self._downloader.evaluate_outtmpl(self._sponsorblock_chapter_title, c)
+                c['title'] = self._downloader.evaluate_outtmpl(self._sponsorblock_chapter_title, c.copy())
                  # Merge identically named sponsors.
                  if (new_chapters and 'categories' in new_chapters[-1]
                          and new_chapters[-1]['title'] == c['title']):
@@ -318,7 +314,7 @@ def remove_chapters(self, filename, ranges_to_cut, concat_opts, force_keyframes=
          self.to_screen(f'Removing chapters from {filename}')
          self.concat_files([in_file] * len(concat_opts), out_file, concat_opts)
          if in_file != filename:
-            os.remove(in_file)
+            self._delete_downloaded_files(in_file, msg=None)
          return out_file
  
      @staticmethod
@@ -331,6 +327,6 @@ def _make_concat_opts(chapters_to_remove, duration):
                  continue
              opts[-1]['outpoint'] = f'{s["start_time"]:.6f}'
              # Do not create 0 duration chunk at the end.
-            if s['end_time'] != duration:
+            if s['end_time'] < duration:
                  opts.append({'inpoint': f'{s["end_time"]:.6f}'})
          return opts