[cleanup, utils] Split into submodules (#7090)

[yt-dlp.git] / yt_dlp / YoutubeDL.py
diff --git a/yt_dlp/YoutubeDL.py b/yt_dlp/YoutubeDL.py

index dce6cf928c2ff0a95424f0b6d55169228cff458b..b8f1a05a0954b2fe9de955a196eff3fd6eb53580 100644 (file)
--- a/yt_dlp/YoutubeDL.py
+++ b/yt_dlp/YoutubeDL.py
@@ -13,6 +13,7 @@
  import random
  import re
  import shutil
+import string
  import subprocess
  import sys
  import tempfile
@@ -21,7 +22,6 @@
  import traceback
  import unicodedata
  import urllib.request
-from string import Formatter, ascii_letters
  
  from .cache import Cache
  from .compat import compat_os_name, compat_shlex_quote
@@ -124,7 +124,6 @@
      parse_filesize,
      preferredencoding,
      prepend_extension,
-    register_socks_protocols,
      remove_terminal_sequences,
      render_table,
      replace_extension,
@@ -190,6 +189,7 @@ class YoutubeDL:
      ap_username:       Multiple-system operator account username.
      ap_password:       Multiple-system operator account password.
      usenetrc:          Use netrc for authentication instead.
+    netrc_location:    Location of the netrc file. Defaults to ~/.netrc.
      verbose:           Print additional info to stdout.
      quiet:             Do not print messages to stdout.
      no_warnings:       Do not print out anything for warnings.
@@ -738,7 +738,6 @@ def check_deprecated(param, option, suggestion):
                  when=when)
  
          self._setup_opener()
-        register_socks_protocols()
  
          def preload_download_archive(fn):
              """Preload the archive, if any is specified"""
@@ -1078,7 +1077,7 @@ def _outtmpl_expandpath(outtmpl):
          # correspondingly that is not what we want since we need to keep
          # '%%' intact for template dict substitution step. Working around
          # with boundary-alike separator hack.
-        sep = ''.join(random.choices(ascii_letters, k=32))
+        sep = ''.join(random.choices(string.ascii_letters, k=32))
          outtmpl = outtmpl.replace('%%', f'%{sep}%').replace('$$', f'${sep}$')
  
          # outtmpl should be expand_path'ed before template dict substitution
@@ -1237,7 +1236,7 @@ def _dumpjson_default(obj):
                  return list(obj)
              return repr(obj)
  
-        class _ReplacementFormatter(Formatter):
+        class _ReplacementFormatter(string.Formatter):
              def get_field(self, field_name, args, kwargs):
                  if field_name.isdigit():
                      return args[0], -1
@@ -1677,7 +1676,7 @@ def process_ie_result(self, ie_result, download=True, extra_info=None):
                  self.add_extra_info(info_copy, extra_info)
                  info_copy, _ = self.pre_process(info_copy)
                  self._fill_common_fields(info_copy, False)
-                self.__forced_printings(info_copy, self.prepare_filename(info_copy), incomplete=True)
+                self.__forced_printings(info_copy)
                  self._raise_pending_errors(info_copy)
                  if self.params.get('force_write_download_archive', False):
                      self.record_download_archive(info_copy)
@@ -2067,86 +2066,86 @@ def syntax_error(note, start):
  
          def _parse_filter(tokens):
              filter_parts = []
-            for type, string, start, _, _ in tokens:
-                if type == tokenize.OP and string == ']':
+            for type, string_, start, _, _ in tokens:
+                if type == tokenize.OP and string_ == ']':
                      return ''.join(filter_parts)
                  else:
-                    filter_parts.append(string)
+                    filter_parts.append(string_)
  
          def _remove_unused_ops(tokens):
              # Remove operators that we don't use and join them with the surrounding strings.
              # E.g. 'mp4' '-' 'baseline' '-' '16x9' is converted to 'mp4-baseline-16x9'
              ALLOWED_OPS = ('/', '+', ',', '(', ')')
              last_string, last_start, last_end, last_line = None, None, None, None
-            for type, string, start, end, line in tokens:
-                if type == tokenize.OP and string == '[':
+            for type, string_, start, end, line in tokens:
+                if type == tokenize.OP and string_ == '[':
                      if last_string:
                          yield tokenize.NAME, last_string, last_start, last_end, last_line
                          last_string = None
-                    yield type, string, start, end, line
+                    yield type, string_, start, end, line
                      # everything inside brackets will be handled by _parse_filter
-                    for type, string, start, end, line in tokens:
-                        yield type, string, start, end, line
-                        if type == tokenize.OP and string == ']':
+                    for type, string_, start, end, line in tokens:
+                        yield type, string_, start, end, line
+                        if type == tokenize.OP and string_ == ']':
                              break
-                elif type == tokenize.OP and string in ALLOWED_OPS:
+                elif type == tokenize.OP and string_ in ALLOWED_OPS:
                      if last_string:
                          yield tokenize.NAME, last_string, last_start, last_end, last_line
                          last_string = None
-                    yield type, string, start, end, line
+                    yield type, string_, start, end, line
                  elif type in [tokenize.NAME, tokenize.NUMBER, tokenize.OP]:
                      if not last_string:
-                        last_string = string
+                        last_string = string_
                          last_start = start
                          last_end = end
                      else:
-                        last_string += string
+                        last_string += string_
              if last_string:
                  yield tokenize.NAME, last_string, last_start, last_end, last_line
  
          def _parse_format_selection(tokens, inside_merge=False, inside_choice=False, inside_group=False):
              selectors = []
              current_selector = None
-            for type, string, start, _, _ in tokens:
+            for type, string_, start, _, _ in tokens:
                  # ENCODING is only defined in python 3.x
                  if type == getattr(tokenize, 'ENCODING', None):
                      continue
                  elif type in [tokenize.NAME, tokenize.NUMBER]:
-                    current_selector = FormatSelector(SINGLE, string, [])
+                    current_selector = FormatSelector(SINGLE, string_, [])
                  elif type == tokenize.OP:
-                    if string == ')':
+                    if string_ == ')':
                          if not inside_group:
                              # ')' will be handled by the parentheses group
                              tokens.restore_last_token()
                          break
-                    elif inside_merge and string in ['/', ',']:
+                    elif inside_merge and string_ in ['/', ',']:
                          tokens.restore_last_token()
                          break
-                    elif inside_choice and string == ',':
+                    elif inside_choice and string_ == ',':
                          tokens.restore_last_token()
                          break
-                    elif string == ',':
+                    elif string_ == ',':
                          if not current_selector:
                              raise syntax_error('"," must follow a format selector', start)
                          selectors.append(current_selector)
                          current_selector = None
-                    elif string == '/':
+                    elif string_ == '/':
                          if not current_selector:
                              raise syntax_error('"/" must follow a format selector', start)
                          first_choice = current_selector
                          second_choice = _parse_format_selection(tokens, inside_choice=True)
                          current_selector = FormatSelector(PICKFIRST, (first_choice, second_choice), [])
-                    elif string == '[':
+                    elif string_ == '[':
                          if not current_selector:
                              current_selector = FormatSelector(SINGLE, 'best', [])
                          format_filter = _parse_filter(tokens)
                          current_selector.filters.append(format_filter)
-                    elif string == '(':
+                    elif string_ == '(':
                          if current_selector:
                              raise syntax_error('Unexpected "("', start)
                          group = _parse_format_selection(tokens, inside_group=True)
                          current_selector = FormatSelector(GROUP, group, [])
-                    elif string == '+':
+                    elif string_ == '+':
                          if not current_selector:
                              raise syntax_error('Unexpected "+"', start)
                          selector_1 = current_selector
@@ -2155,7 +2154,7 @@ def _parse_format_selection(tokens, inside_merge=False, inside_choice=False, ins
                              raise syntax_error('Expected a selector', start)
                          current_selector = FormatSelector(MERGE, (selector_1, selector_2), [])
                      else:
-                        raise syntax_error(f'Operator not recognized: "{string}"', start)
+                        raise syntax_error(f'Operator not recognized: "{string_}"', start)
                  elif type == tokenize.ENDMARKER:
                      break
              if current_selector:
@@ -2719,7 +2718,7 @@ def is_wellformed(f):
              self.list_formats(info_dict)
          if list_only:
              # Without this printing, -F --print-json will not work
-            self.__forced_printings(info_dict, self.prepare_filename(info_dict), incomplete=True)
+            self.__forced_printings(info_dict)
              return info_dict
  
          format_selector = self.format_selector
@@ -2879,6 +2878,12 @@ def _forceprint(self, key, info_dict):
          if info_dict is None:
              return
          info_copy = info_dict.copy()
+        info_copy.setdefault('filename', self.prepare_filename(info_dict))
+        if info_dict.get('requested_formats') is not None:
+            # For RTMP URLs, also include the playpath
+            info_copy['urls'] = '\n'.join(f['url'] + f.get('play_path', '') for f in info_dict['requested_formats'])
+        elif info_dict.get('url'):
+            info_copy['urls'] = info_dict['url'] + info_dict.get('play_path', '')
          info_copy['formats_table'] = self.render_formats_table(info_dict)
          info_copy['thumbnails_table'] = self.render_thumbnails_table(info_dict)
          info_copy['subtitles_table'] = self.render_subtitles_table(info_dict.get('id'), info_dict.get('subtitles'))
@@ -2891,7 +2896,7 @@ def format_tmpl(tmpl):
  
              fmt = '%({})s'
              if tmpl.startswith('{'):
-                tmpl = f'.{tmpl}'
+                tmpl, fmt = f'.{tmpl}', '%({})j'
              if tmpl.endswith('='):
                  tmpl, fmt = tmpl[:-1], '{0} = %({0})#j'
              return '\n'.join(map(fmt.format, [tmpl] if mobj.group('dict') else tmpl.split(',')))
@@ -2907,43 +2912,34 @@ def format_tmpl(tmpl):
                  with open(filename, 'a', encoding='utf-8', newline='') as f:
                      f.write(self.evaluate_outtmpl(tmpl, info_copy) + os.linesep)
  
-    def __forced_printings(self, info_dict, filename, incomplete):
-        def print_mandatory(field, actual_field=None):
-            if actual_field is None:
-                actual_field = field
-            if (self.params.get('force%s' % field, False)
-                    and (not incomplete or info_dict.get(actual_field) is not None)):
-                self.to_stdout(info_dict[actual_field])
-
-        def print_optional(field):
-            if (self.params.get('force%s' % field, False)
-                    and info_dict.get(field) is not None):
-                self.to_stdout(info_dict[field])
-
-        info_dict = info_dict.copy()
-        if filename is not None:
-            info_dict['filename'] = filename
-        if info_dict.get('requested_formats') is not None:
-            # For RTMP URLs, also include the playpath
-            info_dict['urls'] = '\n'.join(f['url'] + f.get('play_path', '') for f in info_dict['requested_formats'])
-        elif info_dict.get('url'):
-            info_dict['urls'] = info_dict['url'] + info_dict.get('play_path', '')
+        return info_copy
  
+    def __forced_printings(self, info_dict, filename=None, incomplete=True):
          if (self.params.get('forcejson')
                  or self.params['forceprint'].get('video')
                  or self.params['print_to_file'].get('video')):
              self.post_extract(info_dict)
-        self._forceprint('video', info_dict)
-
-        print_mandatory('title')
-        print_mandatory('id')
-        print_mandatory('url', 'urls')
-        print_optional('thumbnail')
-        print_optional('description')
-        print_optional('filename')
-        if self.params.get('forceduration') and info_dict.get('duration') is not None:
-            self.to_stdout(formatSeconds(info_dict['duration']))
-        print_mandatory('format')
+        if filename:
+            info_dict['filename'] = filename
+        info_copy = self._forceprint('video', info_dict)
+
+        def print_field(field, actual_field=None, optional=False):
+            if actual_field is None:
+                actual_field = field
+            if self.params.get(f'force{field}') and (
+                    info_copy.get(field) is not None or (not optional and not incomplete)):
+                self.to_stdout(info_copy[actual_field])
+
+        print_field('title')
+        print_field('id')
+        print_field('url', 'urls')
+        print_field('thumbnail', optional=True)
+        print_field('description', optional=True)
+        if filename:
+            print_field('filename')
+        if self.params.get('forceduration') and info_copy.get('duration') is not None:
+            self.to_stdout(formatSeconds(info_copy['duration']))
+        print_field('format')
  
          if self.params.get('forcejson'):
              self.to_stdout(json.dumps(self.sanitize_info(info_dict)))
@@ -3422,8 +3418,8 @@ def sanitize_info(info_dict, remove_private_keys=False):
          if remove_private_keys:
              reject = lambda k, v: v is None or k.startswith('__') or k in {
                  'requested_downloads', 'requested_formats', 'requested_subtitles', 'requested_entries',
-                'entries', 'filepath', '_filename', 'infojson_filename', 'original_url', 'playlist_autonumber',
-                '_format_sort_fields',
+                'entries', 'filepath', '_filename', 'filename', 'infojson_filename', 'original_url',
+                'playlist_autonumber', '_format_sort_fields',
              }
          else:
              reject = lambda k, v: False
@@ -3998,7 +3994,7 @@ def _write_subtitles(self, info_dict, filename):
              # that way it will silently go on when used with unsupporting IE
              return ret
          elif not subtitles:
-            self.to_screen('[info] There\'s no subtitles for the requested languages')
+            self.to_screen('[info] There are no subtitles for the requested languages')
              return ret
          sub_filename_base = self.prepare_filename(info_dict, 'subtitle')
          if not sub_filename_base:
@@ -4052,7 +4048,7 @@ def _write_thumbnails(self, label, info_dict, filename, thumb_filename_base=None
          if write_all or self.params.get('writethumbnail', False):
              thumbnails = info_dict.get('thumbnails') or []
              if not thumbnails:
-                self.to_screen(f'[info] There\'s no {label} thumbnails to download')
+                self.to_screen(f'[info] There are no {label} thumbnails to download')
                  return ret
          multiple = write_all and len(thumbnails) > 1