[cleanup] Misc fixes (see desc)

[yt-dlp.git] / yt_dlp / utils.py
diff --git a/yt_dlp/utils.py b/yt_dlp/utils.py

index 777b8b3ead3888e812ef9861f2754e6312fdd5f0..e6e6d27594e6cf6d458ad097ece362ee44416ff1 100644 (file)
--- a/yt_dlp/utils.py
+++ b/yt_dlp/utils.py
@@ -34,6 +34,7 @@
  import tempfile
  import time
  import traceback
+import types
  import urllib.parse
  import xml.etree.ElementTree
  import zlib
@@ -397,14 +398,14 @@ def get_element_html_by_attribute(attribute, value, html, **kargs):
  def get_elements_by_class(class_name, html, **kargs):
      """Return the content of all tags with the specified class in the passed HTML document as a list"""
      return get_elements_by_attribute(
-        'class', r'[^\'"]*\b%s\b[^\'"]*' % re.escape(class_name),
+        'class', r'[^\'"]*(?<=[\'"\s])%s(?=[\'"\s])[^\'"]*' % re.escape(class_name),
          html, escape_value=False)
  
  
  def get_elements_html_by_class(class_name, html):
      """Return the html of all tags with the specified class in the passed HTML document as a list"""
      return get_elements_html_by_attribute(
-        'class', r'[^\'"]*\b%s\b[^\'"]*' % re.escape(class_name),
+        'class', r'[^\'"]*(?<=[\'"\s])%s(?=[\'"\s])[^\'"]*' % re.escape(class_name),
          html, escape_value=False)
  
  
@@ -3404,16 +3405,15 @@ def _match_one(filter_part, dct, incomplete):
      else:
          is_incomplete = lambda k: k in incomplete
  
-    operator_rex = re.compile(r'''(?x)\s*
+    operator_rex = re.compile(r'''(?x)
          (?P<key>[a-z_]+)
          \s*(?P<negation>!\s*)?(?P<op>%s)(?P<none_inclusive>\s*\?)?\s*
          (?:
              (?P<quote>["\'])(?P<quotedstrval>.+?)(?P=quote)|
              (?P<strval>.+?)
          )
-        \s*$
          ''' % '|'.join(map(re.escape, COMPARISON_OPERATORS.keys())))
-    m = operator_rex.search(filter_part)
+    m = operator_rex.fullmatch(filter_part.strip())
      if m:
          m = m.groupdict()
          unnegated_op = COMPARISON_OPERATORS[m['op']]
@@ -3449,11 +3449,10 @@ def _match_one(filter_part, dct, incomplete):
          '': lambda v: (v is True) if isinstance(v, bool) else (v is not None),
          '!': lambda v: (v is False) if isinstance(v, bool) else (v is None),
      }
-    operator_rex = re.compile(r'''(?x)\s*
+    operator_rex = re.compile(r'''(?x)
          (?P<op>%s)\s*(?P<key>[a-z_]+)
-        \s*$
          ''' % '|'.join(map(re.escape, UNARY_OPERATORS.keys())))
-    m = operator_rex.search(filter_part)
+    m = operator_rex.fullmatch(filter_part.strip())
      if m:
          op = UNARY_OPERATORS[m.group('op')]
          actual_value = dct.get(m.group('key'))
@@ -3495,6 +3494,23 @@ def _match_func(info_dict, incomplete=False):
      return _match_func
  
  
+def download_range_func(chapters, ranges):
+    def inner(info_dict, ydl):
+        warning = ('There are no chapters matching the regex' if info_dict.get('chapters')
+                   else 'Cannot match chapters since chapter information is unavailable')
+        for regex in chapters or []:
+            for i, chapter in enumerate(info_dict.get('chapters') or []):
+                if re.search(regex, chapter['title']):
+                    warning = None
+                    yield {**chapter, 'index': i}
+        if chapters and warning:
+            ydl.to_screen(f'[info] {info_dict["id"]}: {warning}')
+
+        yield from ({'start_time': start, 'end_time': end} for start, end in ranges or [])
+
+    return inner
+
+
  def parse_dfxp_time_expr(time_expr):
      if not time_expr:
          return
@@ -4886,9 +4902,9 @@ def to_high_limit_path(path):
      return path
  
  
-def format_field(obj, field=None, template='%s', ignore=(None, ''), default='', func=None):
+def format_field(obj, field=None, template='%s', ignore=NO_DEFAULT, default='', func=None):
      val = traverse_obj(obj, *variadic(field))
-    if val in ignore:
+    if (not val and val != 0) if ignore is NO_DEFAULT else val in ignore:
          return default
      return template % (func(val) if func else val)
  
@@ -5378,23 +5394,15 @@ def __get__(self, _, cls):
          return self.func(cls)
  
  
-class Namespace:
+class Namespace(types.SimpleNamespace):
      """Immutable namespace"""
  
-    def __init__(self, **kwargs):
-        self._dict = kwargs
-
-    def __getattr__(self, attr):
-        return self._dict[attr]
-
-    def __contains__(self, item):
-        return item in self._dict.values()
-
      def __iter__(self):
-        return iter(self._dict.items())
+        return iter(self.__dict__.values())
  
-    def __repr__(self):
-        return f'{type(self).__name__}({", ".join(f"{k}={v}" for k, v in self)})'
+    @property
+    def items_(self):
+        return self.__dict__.items()
  
  
  # Deprecated