[cleanup] Lint and misc cleanup

[yt-dlp.git] / yt_dlp / extractor / common.py
diff --git a/yt_dlp/extractor/common.py b/yt_dlp/extractor/common.py

index 3e3e557985ca17d6cdb405e9345318c3feba5234..20ed5221637df16b1eb7eff9f875a28ccf13dc78 100644 (file)
--- a/yt_dlp/extractor/common.py
+++ b/yt_dlp/extractor/common.py
@@ -1,35 +1,32 @@
  import base64
  import collections
+import getpass
  import hashlib
+import http.client
+import http.cookiejar
+import http.cookies
+import inspect
  import itertools
  import json
  import math
  import netrc
  import os
  import random
+import re
  import sys
  import time
+import types
+import urllib.parse
+import urllib.request
  import xml.etree.ElementTree
  
-from ..compat import functools, re  # isort: split
-from ..compat import (
-    compat_cookiejar_Cookie,
-    compat_cookies_SimpleCookie,
-    compat_etree_fromstring,
-    compat_expanduser,
-    compat_getpass,
-    compat_http_client,
-    compat_os_name,
-    compat_str,
-    compat_urllib_error,
-    compat_urllib_parse_unquote,
-    compat_urllib_parse_urlencode,
-    compat_urllib_request,
-    compat_urlparse,
-)
+from ..compat import functools  # isort: split
+from ..compat import compat_etree_fromstring, compat_expanduser, compat_os_name
+from ..cookies import LenientSimpleCookie
  from ..downloader import FileDownloader
  from ..downloader.f4m import get_base_url, remove_encrypted_media
  from ..utils import (
+    IDENTITY,
      JSON_LD_RE,
      NO_DEFAULT,
      ExtractorError,
@@ -37,6 +34,7 @@
      GeoUtils,
      LenientJSONDecoder,
      RegexNotFoundError,
+    RetryManager,
      UnsupportedError,
      age_restricted,
      base_url,
@@ -66,11 +64,14 @@
      parse_m3u8_attributes,
      parse_resolution,
      sanitize_filename,
+    sanitize_url,
      sanitized_Request,
+    smuggle_url,
      str_or_none,
      str_to_int,
      strip_or_none,
      traverse_obj,
+    try_call,
      try_get,
      unescapeHTML,
      unified_strdate,
@@ -156,6 +157,7 @@ class InfoExtractor:
                      * abr        Average audio bitrate in KBit/s
                      * acodec     Name of the audio codec in use
                      * asr        Audio sampling rate in Hertz
+                    * audio_channels  Number of audio channels
                      * vbr        Average video bitrate in KBit/s
                      * fps        Frame rate
                      * vcodec     Name of the video codec in use
@@ -283,6 +285,7 @@ class InfoExtractor:
                      captions instead of normal subtitles
      duration:       Length of the video in seconds, as an integer or float.
      view_count:     How many users have watched the video on the platform.
+    concurrent_view_count: How many users are currently watching the video on the platform.
      like_count:     Number of positive ratings of the video
      dislike_count:  Number of negative ratings of the video
      repost_count:   Number of reposts of the video
@@ -318,7 +321,8 @@ class InfoExtractor:
                      live stream that goes on instead of a fixed-length video.
      was_live:       True, False, or None (=unknown). Whether this video was
                      originally a live stream.
-    live_status:    'is_live', 'is_upcoming', 'was_live', 'not_live' or None (=unknown)
+    live_status:    None (=unknown), 'is_live', 'is_upcoming', 'was_live', 'not_live',
+                    or 'post_live' (was live, but VOD is not yet processed)
                      If absent, automatically set from is_live, was_live
      start_time:     Time in seconds where the reproduction should start, as
                      specified in the URL.
@@ -331,11 +335,12 @@ class InfoExtractor:
      playable_in_embed: Whether this video is allowed to play in embedded
                      players on other sites. Can be True (=always allowed),
                      False (=never allowed), None (=unknown), or a string
-                    specifying the criteria for embedability (Eg: 'whitelist')
+                    specifying the criteria for embedability; e.g. 'whitelist'
      availability:   Under what condition the video is available. One of
                      'private', 'premium_only', 'subscriber_only', 'needs_auth',
                      'unlisted' or 'public'. Use 'InfoExtractor._availability'
                      to set it
+    _old_archive_ids: A list of old archive ids needed for backward compatibility
      __post_extractor: A function to be called just before the metadata is
                      written to either disk, logger or console. The function
                      must return a dict which will be added to the info_dict.
@@ -385,6 +390,15 @@ class InfoExtractor:
      release_year:   Year (YYYY) when the album was released.
      composer:       Composer of the piece
  
+    The following fields should only be set for clips that should be cut from the original video:
+
+    section_start:  Start time of the section in seconds
+    section_end:    End time of the section in seconds
+
+    The following fields should only be set for storyboards:
+    rows:           Number of rows in each storyboard fragment, as an integer
+    columns:        Number of columns in each storyboard fragment, as an integer
+
      Unless mentioned otherwise, the fields should be Unicode strings.
  
      Unless mentioned otherwise, None is equivalent to absence of information.
@@ -394,7 +408,7 @@ class InfoExtractor:
      There must be a key "entries", which is a list, an iterable, or a PagedList
      object, each element of which is a valid dictionary by this specification.
  
-    Additionally, playlists can have "id", "title", and any other relevent
+    Additionally, playlists can have "id", "title", and any other relevant
      attributes with the same semantics as videos (see above).
  
      It can also have the following optional fields:
@@ -427,14 +441,26 @@ class InfoExtractor:
      title, description etc.
  
  
-    Subclasses of this should define a _VALID_URL regexp and, re-define the
-    _real_extract() and (optionally) _real_initialize() methods.
-    Probably, they should also be added to the list of extractors.
+    Subclasses of this should also be added to the list of extractors and
+    should define a _VALID_URL regexp and, re-define the _real_extract() and
+    (optionally) _real_initialize() methods.
  
      Subclasses may also override suitable() if necessary, but ensure the function
      signature is preserved and that this function imports everything it needs
      (except other extractors), so that lazy_extractors works correctly.
  
+    Subclasses can define a list of _EMBED_REGEX, which will be searched for in
+    the HTML of Generic webpages. It may also override _extract_embed_urls
+    or _extract_from_webpage as necessary. While these are normally classmethods,
+    _extract_from_webpage is allowed to be an instance method.
+
+    _extract_from_webpage may raise self.StopExtraction() to stop further
+    processing of the webpage and obtain exclusive rights to it. This is useful
+    when the extractor cannot reliably be matched using just the URL,
+    e.g. invidious/peertube instances
+
+    Embed-only extractors can be defined by setting _VALID_URL = False.
+
      To support username + password (or netrc) login, the extractor must define a
      _NETRC_MACHINE and re-define _perform_login(username, password) and
      (optionally) _initialize_pre_login() methods. The _perform_login method will
@@ -458,6 +484,9 @@ class InfoExtractor:
      will be used by geo restriction bypass mechanism similarly
      to _GEO_COUNTRIES.
  
+    The _ENABLED attribute should be set to False for IEs that
+    are disabled by default and must be explicitly enabled.
+
      The _WORKING attribute should be set to False for broken IEs
      in order to warn the users and skip the tests.
      """
@@ -469,9 +498,12 @@ class InfoExtractor:
      _GEO_COUNTRIES = None
      _GEO_IP_BLOCKS = None
      _WORKING = True
+    _ENABLED = True
      _NETRC_MACHINE = None
      IE_DESC = None
      SEARCH_KEY = None
+    _VALID_URL = None
+    _EMBED_REGEX = []
  
      def _login_hint(self, method=NO_DEFAULT, netrc=None):
          password_hint = f'--username and --password, or --netrc ({netrc or self._NETRC_MACHINE}) to provide account credentials'
@@ -481,7 +513,7 @@ def _login_hint(self, method=NO_DEFAULT, netrc=None):
              'password': f'Use {password_hint}',
              'cookies': (
                  'Use --cookies-from-browser or --cookies for the authentication. '
-                'See  https://github.com/ytdl-org/youtube-dl#how-do-i-pass-cookies-to-youtube-dl  for how to manually pass cookies'),
+                'See  https://github.com/yt-dlp/yt-dlp/wiki/FAQ#how-do-i-pass-cookies-to-yt-dlp  for how to manually pass cookies'),
          }[method if method is not NO_DEFAULT else 'any' if self.supports_login() else 'cookies']
  
      def __init__(self, downloader=None):
@@ -495,12 +527,12 @@ def __init__(self, downloader=None):
  
      @classmethod
      def _match_valid_url(cls, url):
+        if cls._VALID_URL is False:
+            return None
          # This does not use has/getattr intentionally - we want to know whether
          # we have cached the regexp for *this* class, whereas getattr would also
          # match the superclass
          if '_VALID_URL_RE' not in cls.__dict__:
-            if '_VALID_URL' not in cls.__dict__:
-                cls._VALID_URL = cls._make_valid_url()
              cls._VALID_URL_RE = re.compile(cls._VALID_URL)
          return cls._VALID_URL_RE.match(url)
  
@@ -644,10 +676,10 @@ def extract(self, url):
                          return None
                      if self._x_forwarded_for_ip:
                          ie_result['__x_forwarded_for_ip'] = self._x_forwarded_for_ip
-                    subtitles = ie_result.get('subtitles')
-                    if (subtitles and 'live_chat' in subtitles
-                            and 'no-live-chat' in self.get_param('compat_opts', [])):
-                        del subtitles['live_chat']
+                    subtitles = ie_result.get('subtitles') or {}
+                    if 'no-live-chat' in self.get_param('compat_opts'):
+                        for lang in ('live_chat', 'comments', 'danmaku'):
+                            subtitles.pop(lang, None)
                      return ie_result
                  except GeoRestrictedError as e:
                      if self.__maybe_fake_ip_and_retry(e.countries):
@@ -666,7 +698,7 @@ def extract(self, url):
              if hasattr(e, 'countries'):
                  kwargs['countries'] = e.countries
              raise type(e)(e.orig_msg, **kwargs)
-        except compat_http_client.IncompleteRead as e:
+        except http.client.IncompleteRead as e:
              raise ExtractorError('A network error has occurred.', cause=e, expected=True, video_id=self.get_temp_id(url))
          except (KeyError, StopIteration) as e:
              raise ExtractorError('An extractor error has occurred.', cause=e, video_id=self.get_temp_id(url))
@@ -690,8 +722,16 @@ def set_downloader(self, downloader):
          """Sets a YoutubeDL instance as the downloader for this IE."""
          self._downloader = downloader
  
+    @property
+    def cache(self):
+        return self._downloader.cache
+
+    @property
+    def cookiejar(self):
+        return self._downloader.cookiejar
+
      def _initialize_pre_login(self):
-        """ Intialization before login. Redefine in subclasses."""
+        """ Initialization before login. Redefine in subclasses."""
          pass
  
      def _perform_login(self, username, password):
@@ -717,7 +757,7 @@ def IE_NAME(cls):
  
      @staticmethod
      def __can_accept_status_code(err, expected_status):
-        assert isinstance(err, compat_urllib_error.HTTPError)
+        assert isinstance(err, urllib.error.HTTPError)
          if expected_status is None:
              return False
          elif callable(expected_status):
@@ -725,14 +765,14 @@ def __can_accept_status_code(err, expected_status):
          else:
              return err.code in variadic(expected_status)
  
-    def _create_request(self, url_or_request, data=None, headers={}, query={}):
-        if isinstance(url_or_request, compat_urllib_request.Request):
+    def _create_request(self, url_or_request, data=None, headers=None, query=None):
+        if isinstance(url_or_request, urllib.request.Request):
              return update_Request(url_or_request, data=data, headers=headers, query=query)
          if query:
              url_or_request = update_url_query(url_or_request, query)
-        return sanitized_Request(url_or_request, data, headers)
+        return sanitized_Request(url_or_request, data, headers or {})
  
-    def _request_webpage(self, url_or_request, video_id, note=None, errnote=None, fatal=True, data=None, headers={}, query={}, expected_status=None):
+    def _request_webpage(self, url_or_request, video_id, note=None, errnote=None, fatal=True, data=None, headers=None, query=None, expected_status=None):
          """
          Return the response handle.
  
@@ -760,13 +800,13 @@ def _request_webpage(self, url_or_request, video_id, note=None, errnote=None, fa
          # geo unrestricted country. We will do so once we encounter any
          # geo restriction error.
          if self._x_forwarded_for_ip:
-            if 'X-Forwarded-For' not in headers:
-                headers['X-Forwarded-For'] = self._x_forwarded_for_ip
+            headers = (headers or {}).copy()
+            headers.setdefault('X-Forwarded-For', self._x_forwarded_for_ip)
  
          try:
              return self._downloader.urlopen(self._create_request(url_or_request, data, headers, query))
          except network_exceptions as err:
-            if isinstance(err, compat_urllib_error.HTTPError):
+            if isinstance(err, urllib.error.HTTPError):
                  if self.__can_accept_status_code(err, expected_status):
                      # Retain reference to error to prevent file object from
                      # being closed before it can be read. Works around the
@@ -794,7 +834,7 @@ def _download_webpage_handle(self, url_or_request, video_id, note=None, errnote=
  
          Arguments:
          url_or_request -- plain text URL as a string or
-            a compat_urllib_request.Requestobject
+            a urllib.request.Request object
          video_id -- Video/playlist/item identifier (string)
  
          Keyword arguments:
@@ -822,7 +862,7 @@ def _download_webpage_handle(self, url_or_request, video_id, note=None, errnote=
          """
  
          # Strip hashes from the URL (#1038)
-        if isinstance(url_or_request, (compat_str, str)):
+        if isinstance(url_or_request, str):
              url_or_request = url_or_request.partition('#')[0]
  
          urlh = self._request_webpage(url_or_request, video_id, note, errnote, fatal, data=data, headers=headers, query=query, expected_status=expected_status)
@@ -919,39 +959,37 @@ def _webpage_read_content(self, urlh, url_or_request, video_id, note=None, errno
  
          return content
  
-    def _parse_xml(self, xml_string, video_id, transform_source=None, fatal=True):
+    def __print_error(self, errnote, fatal, video_id, err):
+        if fatal:
+            raise ExtractorError(f'{video_id}: {errnote}', cause=err)
+        elif errnote:
+            self.report_warning(f'{video_id}: {errnote}: {err}')
+
+    def _parse_xml(self, xml_string, video_id, transform_source=None, fatal=True, errnote=None):
          if transform_source:
              xml_string = transform_source(xml_string)
          try:
              return compat_etree_fromstring(xml_string.encode('utf-8'))
          except xml.etree.ElementTree.ParseError as ve:
-            errmsg = '%s: Failed to parse XML ' % video_id
-            if fatal:
-                raise ExtractorError(errmsg, cause=ve)
-            else:
-                self.report_warning(errmsg + str(ve))
+            self.__print_error('Failed to parse XML' if errnote is None else errnote, fatal, video_id, ve)
  
-    def _parse_json(self, json_string, video_id, transform_source=None, fatal=True, **parser_kwargs):
+    def _parse_json(self, json_string, video_id, transform_source=None, fatal=True, errnote=None, **parser_kwargs):
          try:
              return json.loads(
                  json_string, cls=LenientJSONDecoder, strict=False, transform_source=transform_source, **parser_kwargs)
          except ValueError as ve:
-            errmsg = f'{video_id}: Failed to parse JSON'
-            if fatal:
-                raise ExtractorError(errmsg, cause=ve)
-            else:
-                self.report_warning(f'{errmsg}: {ve}')
+            self.__print_error('Failed to parse JSON' if errnote is None else errnote, fatal, video_id, ve)
  
-    def _parse_socket_response_as_json(self, data, video_id, transform_source=None, fatal=True):
-        return self._parse_json(
-            data[data.find('{'):data.rfind('}') + 1],
-            video_id, transform_source, fatal)
+    def _parse_socket_response_as_json(self, data, *args, **kwargs):
+        return self._parse_json(data[data.find('{'):data.rfind('}') + 1], *args, **kwargs)
  
      def __create_download_methods(name, parser, note, errnote, return_value):
  
-        def parse(ie, content, *args, **kwargs):
+        def parse(ie, content, *args, errnote=errnote, **kwargs):
              if parser is None:
                  return content
+            if errnote is False:
+                kwargs['errnote'] = errnote
              # parser is fetched by name so subclasses can override it
              return getattr(ie, parser)(content, *args, **kwargs)
  
@@ -963,7 +1001,7 @@ def download_handle(self, url_or_request, video_id, note=note, errnote=errnote,
              if res is False:
                  return res
              content, urlh = res
-            return parse(self, content, video_id, transform_source=transform_source, fatal=fatal), urlh
+            return parse(self, content, video_id, transform_source=transform_source, fatal=fatal, errnote=errnote), urlh
  
          def download_content(self, url_or_request, video_id, note=note, errnote=errnote, transform_source=None,
                               fatal=True, encoding=None, data=None, headers={}, query={}, expected_status=None):
@@ -978,7 +1016,7 @@ def download_content(self, url_or_request, video_id, note=note, errnote=errnote,
                      self.report_warning(f'Unable to load request from disk: {e}')
                  else:
                      content = self.__decode_webpage(webpage_bytes, encoding, url_or_request.headers)
-                    return parse(self, content, video_id, transform_source, fatal)
+                    return parse(self, content, video_id, transform_source=transform_source, fatal=fatal, errnote=errnote)
              kwargs = {
                  'note': note,
                  'errnote': errnote,
@@ -1043,7 +1081,7 @@ def _download_webpage(
          while True:
              try:
                  return self.__download_webpage(url_or_request, video_id, note, errnote, None, fatal, *args, **kwargs)
-            except compat_http_client.IncompleteRead as e:
+            except http.client.IncompleteRead as e:
                  try_count += 1
                  if try_count >= tries:
                      raise e
@@ -1070,7 +1108,9 @@ def get_param(self, name, default=None, *args, **kwargs):
              return self._downloader.params.get(name, default, *args, **kwargs)
          return default
  
-    def report_drm(self, video_id, partial=False):
+    def report_drm(self, video_id, partial=NO_DEFAULT):
+        if partial is not NO_DEFAULT:
+            self._downloader.deprecation_warning('InfoExtractor.report_drm no longer accepts the argument partial')
          self.raise_no_formats('This video is DRM protected', expected=True, video_id=video_id)
  
      def report_extraction(self, id_or_name):
@@ -1133,10 +1173,12 @@ def url_result(url, ie=None, video_id=None, video_title=None, *, url_transparent
              'url': url,
          }
  
-    def playlist_from_matches(self, matches, playlist_id=None, playlist_title=None, getter=None, ie=None, video_kwargs=None, **kwargs):
-        urls = (self.url_result(self._proto_relative_url(m), ie, **(video_kwargs or {}))
-                for m in orderedSet(map(getter, matches) if getter else matches))
-        return self.playlist_result(urls, playlist_id, playlist_title, **kwargs)
+    @classmethod
+    def playlist_from_matches(cls, matches, playlist_id=None, playlist_title=None,
+                              getter=IDENTITY, ie=None, video_kwargs=None, **kwargs):
+        return cls.playlist_result(
+            (cls.url_result(m, ie, **(video_kwargs or {})) for m in orderedSet(map(getter, matches), lazy=True)),
+            playlist_id, playlist_title, **kwargs)
  
      @staticmethod
      def playlist_result(entries, playlist_id=None, playlist_title=None, playlist_description=None, *, multi_video=False, **kwargs):
@@ -1189,7 +1231,7 @@ def _search_regex(self, pattern, string, name, default=NO_DEFAULT, fatal=True, f
              return None
  
      def _search_json(self, start_pattern, string, name, video_id, *, end_pattern='',
-                     contains_pattern='(?s:.+)', fatal=True, default=NO_DEFAULT, **kwargs):
+                     contains_pattern=r'{(?s:.+)}', fatal=True, default=NO_DEFAULT, **kwargs):
          """Searches string for the JSON object specified by start_pattern"""
          # NB: end_pattern is only used to reduce the size of the initial match
          if default is NO_DEFAULT:
@@ -1198,7 +1240,7 @@ def _search_json(self, start_pattern, string, name, video_id, *, end_pattern='',
              fatal, has_default = False, True
  
          json_string = self._search_regex(
-            rf'{start_pattern}\s*(?P<json>{{\s*{contains_pattern}\s*}})\s*{end_pattern}',
+            rf'(?:{start_pattern})\s*(?P<json>{contains_pattern})\s*(?:{end_pattern})',
              string, name, group='json', fatal=fatal, default=None if has_default else NO_DEFAULT)
          if not json_string:
              return default
@@ -1279,7 +1321,7 @@ def _get_tfa_info(self, note='two-factor verification code'):
          if tfa is not None:
              return tfa
  
-        return compat_getpass('Type %s and press [Return]: ' % note)
+        return getpass.getpass('Type %s and press [Return]: ' % note)
  
      # Helper functions for extracting OpenGraph info
      @staticmethod
@@ -1343,12 +1385,20 @@ def _html_search_meta(self, name, html, display_name=None, fatal=False, **kwargs
      def _dc_search_uploader(self, html):
          return self._html_search_meta('dc.creator', html, 'uploader')
  
-    def _rta_search(self, html):
+    @staticmethod
+    def _rta_search(html):
          # See http://www.rtalabel.org/index.php?content=howtofaq#single
          if re.search(r'(?ix)<meta\s+name="rating"\s+'
                       r'     content="RTA-5042-1996-1400-1577-RTA"',
                       html):
              return 18
+
+        # And then there are the jokers who advertise that they use RTA, but actually don't.
+        AGE_LIMIT_MARKERS = [
+            r'Proudly Labeled <a href="http://www\.rtalabel\.org/" title="Restricted to Adults">RTA</a>',
+        ]
+        if any(re.search(marker, html) for marker in AGE_LIMIT_MARKERS):
+            return 18
          return 0
  
      def _media_rating_search(self, html):
@@ -1387,27 +1437,25 @@ def _twitter_search_player(self, html):
          return self._html_search_meta('twitter:player', html,
                                        'twitter card player')
  
-    def _search_json_ld(self, html, video_id, expected_type=None, **kwargs):
-        json_ld_list = list(re.finditer(JSON_LD_RE, html))
-        default = kwargs.get('default', NO_DEFAULT)
-        # JSON-LD may be malformed and thus `fatal` should be respected.
-        # At the same time `default` may be passed that assumes `fatal=False`
-        # for _search_regex. Let's simulate the same behavior here as well.
-        fatal = kwargs.get('fatal', True) if default is NO_DEFAULT else False
-        json_ld = []
-        for mobj in json_ld_list:
-            json_ld_item = self._parse_json(
-                mobj.group('json_ld'), video_id, fatal=fatal)
-            if not json_ld_item:
-                continue
-            if isinstance(json_ld_item, dict):
-                json_ld.append(json_ld_item)
-            elif isinstance(json_ld_item, (list, tuple)):
-                json_ld.extend(json_ld_item)
-        if json_ld:
-            json_ld = self._json_ld(json_ld, video_id, fatal=fatal, expected_type=expected_type)
-        if json_ld:
-            return json_ld
+    def _yield_json_ld(self, html, video_id, *, fatal=True, default=NO_DEFAULT):
+        """Yield all json ld objects in the html"""
+        if default is not NO_DEFAULT:
+            fatal = False
+        for mobj in re.finditer(JSON_LD_RE, html):
+            json_ld_item = self._parse_json(mobj.group('json_ld'), video_id, fatal=fatal)
+            for json_ld in variadic(json_ld_item):
+                if isinstance(json_ld, dict):
+                    yield json_ld
+
+    def _search_json_ld(self, html, video_id, expected_type=None, *, fatal=True, default=NO_DEFAULT):
+        """Search for a video in any json ld in the html"""
+        if default is not NO_DEFAULT:
+            fatal = False
+        info = self._json_ld(
+            list(self._yield_json_ld(html, video_id, fatal=fatal, default=default)),
+            video_id, fatal=fatal, expected_type=expected_type)
+        if info:
+            return info
          if default is not NO_DEFAULT:
              return default
          elif fatal:
@@ -1417,15 +1465,11 @@ def _search_json_ld(self, html, video_id, expected_type=None, **kwargs):
              return {}
  
      def _json_ld(self, json_ld, video_id, fatal=True, expected_type=None):
-        if isinstance(json_ld, compat_str):
+        if isinstance(json_ld, str):
              json_ld = self._parse_json(json_ld, video_id, fatal=fatal)
          if not json_ld:
              return {}
          info = {}
-        if not isinstance(json_ld, (list, tuple, dict)):
-            return info
-        if isinstance(json_ld, dict):
-            json_ld = [json_ld]
  
          INTERACTION_TYPE_MAP = {
              'CommentAction': 'comment',
@@ -1492,13 +1536,13 @@ def extract_chapter_information(e):
                  info['chapters'] = chapters
  
          def extract_video_object(e):
-            assert is_type(e, 'VideoObject')
              author = e.get('author')
              info.update({
-                'url': traverse_obj(e, 'contentUrl', 'embedUrl', expected_type=url_or_none),
+                'url': url_or_none(e.get('contentUrl')),
+                'ext': mimetype2ext(e.get('encodingFormat')),
                  'title': unescapeHTML(e.get('name')),
                  'description': unescapeHTML(e.get('description')),
-                'thumbnails': [{'url': url}
+                'thumbnails': [{'url': unescapeHTML(url)}
                                 for url in variadic(traverse_obj(e, 'thumbnailUrl', 'thumbnailURL'))
                                 if url_or_none(url)],
                  'duration': parse_duration(e.get('duration')),
@@ -1507,23 +1551,32 @@ def extract_video_object(e):
                  # both types can have 'name' property(inherited from 'Thing' type). [1]
                  # however some websites are using 'Text' type instead.
                  # 1. https://schema.org/VideoObject
-                'uploader': author.get('name') if isinstance(author, dict) else author if isinstance(author, compat_str) else None,
+                'uploader': author.get('name') if isinstance(author, dict) else author if isinstance(author, str) else None,
+                'artist': traverse_obj(e, ('byArtist', 'name'), expected_type=str),
                  'filesize': int_or_none(float_or_none(e.get('contentSize'))),
                  'tbr': int_or_none(e.get('bitrate')),
                  'width': int_or_none(e.get('width')),
                  'height': int_or_none(e.get('height')),
                  'view_count': int_or_none(e.get('interactionCount')),
+                'tags': try_call(lambda: e.get('keywords').split(',')),
              })
+            if is_type(e, 'AudioObject'):
+                info.update({
+                    'vcodec': 'none',
+                    'abr': int_or_none(e.get('bitrate')),
+                })
              extract_interaction_statistic(e)
              extract_chapter_information(e)
  
          def traverse_json_ld(json_ld, at_top_level=True):
-            for e in json_ld:
+            for e in variadic(json_ld):
+                if not isinstance(e, dict):
+                    continue
                  if at_top_level and '@context' not in e:
                      continue
                  if at_top_level and set(e.keys()) == {'@context', '@graph'}:
-                    traverse_json_ld(variadic(e['@graph'], allowed_types=(dict,)), at_top_level=False)
-                    break
+                    traverse_json_ld(e['@graph'], at_top_level=False)
+                    continue
                  if expected_type is not None and not is_type(e, expected_type):
                      continue
                  rating = traverse_obj(e, ('aggregateRating', 'ratingValue'), expected_type=float_or_none)
@@ -1564,7 +1617,7 @@ def traverse_json_ld(json_ld, at_top_level=True):
                          extract_video_object(e['video'][0])
                      elif is_type(traverse_obj(e, ('subjectOf', 0)), 'VideoObject'):
                          extract_video_object(e['subjectOf'][0])
-                elif is_type(e, 'VideoObject'):
+                elif is_type(e, 'VideoObject', 'AudioObject'):
                      extract_video_object(e)
                      if expected_type is None:
                          continue
@@ -1577,8 +1630,8 @@ def traverse_json_ld(json_ld, at_top_level=True):
                      continue
                  else:
                      break
-        traverse_json_ld(json_ld)
  
+        traverse_json_ld(json_ld)
          return filter_dict(info)
  
      def _search_nextjs_data(self, webpage, video_id, *, transform_source=None, fatal=True, **kw):
@@ -1631,8 +1684,8 @@ class FormatSort:
          regex = r' *((?P<reverse>\+)?(?P<field>[a-zA-Z0-9_]+)((?P<separator>[~:])(?P<limit>.*?))?)? *$'
  
          default = ('hidden', 'aud_or_vid', 'hasvid', 'ie_pref', 'lang', 'quality',
-                   'res', 'fps', 'hdr:12', 'codec:vp9.2', 'size', 'br', 'asr',
-                   'proto', 'ext', 'hasaud', 'source', 'id')  # These must not be aliases
+                   'res', 'fps', 'hdr:12', 'vcodec:vp9.2', 'channels', 'acodec',
+                   'size', 'br', 'asr', 'proto', 'ext', 'hasaud', 'source', 'id')  # These must not be aliases
          ytdl_default = ('hasaud', 'lang', 'quality', 'tbr', 'filesize', 'vbr',
                          'height', 'width', 'proto', 'vext', 'abr', 'aext',
                          'fps', 'fs_approx', 'source', 'id')
@@ -1651,7 +1704,7 @@ class FormatSort:
                       'order_free': ('webm', 'mp4', 'flv', '', 'none')},
              'aext': {'type': 'ordered', 'field': 'audio_ext',
                       'order': ('m4a', 'aac', 'mp3', 'ogg', 'opus', 'webm', '', 'none'),
-                     'order_free': ('opus', 'ogg', 'webm', 'm4a', 'mp3', 'aac', '', 'none')},
+                     'order_free': ('ogg', 'opus', 'webm', 'mp3', 'm4a', 'aac', '', 'none')},
              'hidden': {'visible': False, 'forced': True, 'type': 'extractor', 'max': -1000},
              'aud_or_vid': {'visible': False, 'forced': True, 'type': 'multiple',
                             'field': ('vcodec', 'acodec'),
@@ -1667,6 +1720,7 @@ class FormatSort:
              'height': {'convert': 'float_none'},
              'width': {'convert': 'float_none'},
              'fps': {'convert': 'float_none'},
+            'channels': {'convert': 'float_none', 'field': 'audio_channels'},
              'tbr': {'convert': 'float_none'},
              'vbr': {'convert': 'float_none'},
              'abr': {'convert': 'float_none'},
@@ -1680,13 +1734,14 @@ class FormatSort:
              'res': {'type': 'multiple', 'field': ('height', 'width'),
                      'function': lambda it: (lambda l: min(l) if l else 0)(tuple(filter(None, it)))},
  
-            # For compatibility with youtube-dl
+            # Actual field names
              'format_id': {'type': 'alias', 'field': 'id'},
              'preference': {'type': 'alias', 'field': 'ie_pref'},
              'language_preference': {'type': 'alias', 'field': 'lang'},
              'source_preference': {'type': 'alias', 'field': 'source'},
              'protocol': {'type': 'alias', 'field': 'proto'},
              'filesize_approx': {'type': 'alias', 'field': 'fs_approx'},
+            'audio_channels': {'type': 'alias', 'field': 'channels'},
  
              # Deprecated
              'dimension': {'type': 'alias', 'field': 'res', 'deprecated': True},
@@ -1722,9 +1777,8 @@ def _get_field_setting(self, field, key):
              if field not in self.settings:
                  if key in ('forced', 'priority'):
                      return False
-                self.ydl.deprecation_warning(
-                    f'Using arbitrary fields ({field}) for format sorting is deprecated '
-                    'and may be removed in a future version')
+                self.ydl.deprecated_feature(f'Using arbitrary fields ({field}) for format sorting is '
+                                            'deprecated and may be removed in a future version')
                  self.settings[field] = {}
              propObj = self.settings[field]
              if key not in propObj:
@@ -1809,9 +1863,8 @@ def add_item(field, reverse, closest, limit_text):
                  if self._get_field_setting(field, 'type') == 'alias':
                      alias, field = field, self._get_field_setting(field, 'field')
                      if self._get_field_setting(alias, 'deprecated'):
-                        self.ydl.deprecation_warning(
-                            f'Format sorting alias {alias} is deprecated '
-                            f'and may be removed in a future version. Please use {field} instead')
+                        self.ydl.deprecated_feature(f'Format sorting alias {alias} is deprecated and may '
+                                                    f'be removed in a future version. Please use {field} instead')
                  reverse = match.group('reverse') is not None
                  closest = match.group('separator') == '~'
                  limit_text = match.group('limit')
@@ -1957,14 +2010,9 @@ def http_scheme(self):
              else 'https:')
  
      def _proto_relative_url(self, url, scheme=None):
-        if url is None:
-            return url
-        if url.startswith('//'):
-            if scheme is None:
-                scheme = self.http_scheme()
-            return scheme + url
-        else:
-            return url
+        scheme = scheme or self.http_scheme()
+        assert scheme.endswith(':')
+        return sanitize_url(url, scheme=scheme[:-1])
  
      def _sleep(self, timeout, video_id, msg_template=None):
          if msg_template is None:
@@ -2156,7 +2204,7 @@ def _parse_m3u8_formats_and_subtitles(
          ]), m3u8_doc)
  
          def format_url(url):
-            return url if re.match(r'^https?://', url) else compat_urlparse.urljoin(m3u8_url, url)
+            return url if re.match(r'^https?://', url) else urllib.parse.urljoin(m3u8_url, url)
  
          if self.get_param('hls_split_discontinuity', False):
              def _extract_m3u8_playlist_indices(manifest_url=None, m3u8_doc=None):
@@ -2332,7 +2380,7 @@ def build_stream_name():
                      audio_group_id = last_stream_inf.get('AUDIO')
                      # As per [1, 4.3.4.1.1] any EXT-X-STREAM-INF tag which
                      # references a rendition group MUST have a CODECS attribute.
-                    # However, this is not always respected, for example, [2]
+                    # However, this is not always respected. E.g. [2]
                      # contains EXT-X-STREAM-INF tag which references AUDIO
                      # rendition group but does not have CODECS and despite
                      # referencing an audio group it represents a complete
@@ -2529,7 +2577,7 @@ def _parse_smil_formats(self, smil, smil_url, video_id, namespace=None, f4m_para
                      })
                  continue
  
-            src_url = src if src.startswith('http') else compat_urlparse.urljoin(base, src)
+            src_url = src if src.startswith('http') else urllib.parse.urljoin(base, src)
              src_url = src_url.strip()
  
              if proto == 'm3u8' or src_ext == 'm3u8':
@@ -2552,7 +2600,7 @@ def _parse_smil_formats(self, smil, smil_url, video_id, namespace=None, f4m_para
                          'plugin': 'flowplayer-3.2.0.1',
                      }
                  f4m_url += '&' if '?' in f4m_url else '?'
-                f4m_url += compat_urllib_parse_urlencode(f4m_params)
+                f4m_url += urllib.parse.urlencode(f4m_params)
                  formats.extend(self._extract_f4m_formats(f4m_url, video_id, f4m_id='hds', fatal=False))
              elif src_ext == 'mpd':
                  formats.extend(self._extract_mpd_formats(
@@ -2817,12 +2865,12 @@ def extract_Initialization(source):
                      base_url = ''
                      for element in (representation, adaptation_set, period, mpd_doc):
                          base_url_e = element.find(_add_ns('BaseURL'))
-                        if base_url_e is not None:
+                        if try_call(lambda: base_url_e.text) is not None:
                              base_url = base_url_e.text + base_url
                              if re.match(r'^https?://', base_url):
                                  break
                      if mpd_base_url and base_url.startswith('/'):
-                        base_url = compat_urlparse.urljoin(mpd_base_url, base_url)
+                        base_url = urllib.parse.urljoin(mpd_base_url, base_url)
                      elif mpd_base_url and not re.match(r'^https?://', base_url):
                          if not mpd_base_url.endswith('/'):
                              mpd_base_url += '/'
@@ -2877,6 +2925,8 @@ def extract_Initialization(source):
  
                      def prepare_template(template_name, identifiers):
                          tmpl = representation_ms_info[template_name]
+                        if representation_id is not None:
+                            tmpl = tmpl.replace('$RepresentationID$', representation_id)
                          # First of, % characters outside $...$ templates
                          # must be escaped by doubling for proper processing
                          # by % operator string formatting used further (see
@@ -2891,8 +2941,6 @@ def prepare_template(template_name, identifiers):
                                  t += c
                          # Next, $...$ templates are translated to their
                          # %(...) counterparts to be used with % operator
-                        if representation_id is not None:
-                            t = t.replace('$RepresentationID$', representation_id)
                          t = re.sub(r'\$(%s)\$' % '|'.join(identifiers), r'%(\1)d', t)
                          t = re.sub(r'\$(%s)%%([^$]+)\$' % '|'.join(identifiers), r'%(\1)\2', t)
                          t.replace('$$', '$')
@@ -2968,8 +3016,8 @@ def add_segment_url():
                                      segment_number += 1
                                  segment_time += segment_d
                      elif 'segment_urls' in representation_ms_info and 's' in representation_ms_info:
-                        # No media template
-                        # Example: https://www.youtube.com/watch?v=iXZV5uAYMJI
+                        # No media template,
+                        # e.g. https://www.youtube.com/watch?v=iXZV5uAYMJI
                          # or any YouTube dashsegments video
                          fragments = []
                          segment_index = 0
@@ -2986,7 +3034,7 @@ def add_segment_url():
                          representation_ms_info['fragments'] = fragments
                      elif 'segment_urls' in representation_ms_info:
                          # Segment URLs with no SegmentTimeline
-                        # Example: https://www.seznam.cz/zpravy/clanek/cesko-zasahne-vitr-o-sile-vichrice-muze-byt-i-zivotu-nebezpecny-39091
+                        # E.g. https://www.seznam.cz/zpravy/clanek/cesko-zasahne-vitr-o-sile-vichrice-muze-byt-i-zivotu-nebezpecny-39091
                          # https://github.com/ytdl-org/youtube-dl/pull/14844
                          fragments = []
                          segment_duration = float_or_none(
@@ -3078,9 +3126,10 @@ def _parse_ism_formats_and_subtitles(self, ism_doc, ism_url, ism_id=None):
              stream_name = stream.get('Name')
              stream_language = stream.get('Language', 'und')
              for track in stream.findall('QualityLevel'):
-                fourcc = track.get('FourCC') or ('AACL' if track.get('AudioTag') == '255' else None)
+                KNOWN_TAGS = {'255': 'AACL', '65534': 'EC-3'}
+                fourcc = track.get('FourCC') or KNOWN_TAGS.get(track.get('AudioTag'))
                  # TODO: add support for WVC1 and WMAP
-                if fourcc not in ('H264', 'AVC1', 'AACL', 'TTML'):
+                if fourcc not in ('H264', 'AVC1', 'AACL', 'TTML', 'EC-3'):
                      self.report_warning('%s is not a supported codec' % fourcc)
                      continue
                  tbr = int(track.attrib['Bitrate']) // 1000
@@ -3092,7 +3141,7 @@ def _parse_ism_formats_and_subtitles(self, ism_doc, ism_url, ism_id=None):
                  sampling_rate = int_or_none(track.get('SamplingRate'))
  
                  track_url_pattern = re.sub(r'{[Bb]itrate}', track.attrib['Bitrate'], url_pattern)
-                track_url_pattern = compat_urlparse.urljoin(ism_url, track_url_pattern)
+                track_url_pattern = urllib.parse.urljoin(ism_url, track_url_pattern)
  
                  fragments = []
                  fragment_ctx = {
@@ -3111,7 +3160,7 @@ def _parse_ism_formats_and_subtitles(self, ism_doc, ism_url, ism_id=None):
                          fragment_ctx['duration'] = (next_fragment_time - fragment_ctx['time']) / fragment_repeat
                      for _ in range(fragment_repeat):
                          fragments.append({
-                            'url': re.sub(r'{start[ _]time}', compat_str(fragment_ctx['time']), track_url_pattern),
+                            'url': re.sub(r'{start[ _]time}', str(fragment_ctx['time']), track_url_pattern),
                              'duration': fragment_ctx['duration'] / stream_timescale,
                          })
                          fragment_ctx['time'] += fragment_ctx['duration']
@@ -3204,7 +3253,7 @@ def _media_formats(src, cur_media_type, type_info=None):
  
          entries = []
          # amp-video and amp-audio are very similar to their HTML5 counterparts
-        # so we wll include them right here (see
+        # so we will include them right here (see
          # https://www.ampproject.org/docs/reference/components/amp-video)
          # For dl8-* tags see https://delight-vr.com/documentation/dl8-video/
          _MEDIA_TAG_NAME_RE = r'(?:(?:amp|dl8(?:-live)?)-)?(video|audio)'
@@ -3214,8 +3263,8 @@ def _media_formats(src, cur_media_type, type_info=None):
          media_tags.extend(re.findall(
              # We only allow video|audio followed by a whitespace or '>'.
              # Allowing more characters may end up in significant slow down (see
-            # https://github.com/ytdl-org/youtube-dl/issues/11979, example URL:
-            # http://www.porntrex.com/maps/videositemap.xml).
+            # https://github.com/ytdl-org/youtube-dl/issues/11979,
+            # e.g. http://www.porntrex.com/maps/videositemap.xml).
              r'(?s)(<(?P<tag>%s)(?:\s+[^>]*)?>)(.*?)</(?P=tag)>' % _MEDIA_TAG_NAME_RE, webpage))
          for media_tag, _, media_type, media_content in media_tags:
              media_info = {
@@ -3223,7 +3272,7 @@ def _media_formats(src, cur_media_type, type_info=None):
                  'subtitles': {},
              }
              media_attributes = extract_attributes(media_tag)
-            src = strip_or_none(media_attributes.get('src'))
+            src = strip_or_none(dict_get(media_attributes, ('src', 'data-video-src', 'data-src', 'data-source')))
              if src:
                  f = parse_content_type(media_attributes.get('type'))
                  _, formats = _media_formats(src, media_type, f)
@@ -3234,7 +3283,7 @@ def _media_formats(src, cur_media_type, type_info=None):
                      s_attr = extract_attributes(source_tag)
                      # data-video-src and data-src are non standard but seen
                      # several times in the wild
-                    src = strip_or_none(dict_get(s_attr, ('src', 'data-video-src', 'data-src')))
+                    src = strip_or_none(dict_get(s_attr, ('src', 'data-video-src', 'data-src', 'data-source')))
                      if not src:
                          continue
                      f = parse_content_type(s_attr.get('type'))
@@ -3355,7 +3404,7 @@ def _extract_akamai_formats_and_subtitles(self, manifest_url, video_id, hosts={}
          return formats, subtitles
  
      def _extract_wowza_formats(self, url, video_id, m3u8_entry_protocol='m3u8_native', skip_protocols=[]):
-        query = compat_urlparse.urlparse(url).query
+        query = urllib.parse.urlparse(url).query
          url = re.sub(r'/(?:manifest|playlist|jwplayer)\.(?:m3u8|f4m|mpd|smil)', '', url)
          mobj = re.search(
              r'(?:(?:http|rtmp|rtsp)(?P<s>s)?:)?(?P<url>//[^?]+)', url)
@@ -3461,7 +3510,7 @@ def _parse_jwplayer_data(self, jwplayer_data, video_id=None, require_title=True,
                      if not isinstance(track, dict):
                          continue
                      track_kind = track.get('kind')
-                    if not track_kind or not isinstance(track_kind, compat_str):
+                    if not track_kind or not isinstance(track_kind, str):
                          continue
                      if track_kind.lower() not in ('captions', 'subtitles'):
                          continue
@@ -3534,13 +3583,14 @@ def _parse_jwplayer_formats(self, jwplayer_sources_data, video_id=None,
                      # Often no height is provided but there is a label in
                      # format like "1080p", "720p SD", or 1080.
                      height = int_or_none(self._search_regex(
-                        r'^(\d{3,4})[pP]?(?:\b|$)', compat_str(source.get('label') or ''),
+                        r'^(\d{3,4})[pP]?(?:\b|$)', str(source.get('label') or ''),
                          'height', default=None))
                  a_format = {
                      'url': source_url,
                      'width': int_or_none(source.get('width')),
                      'height': height,
-                    'tbr': int_or_none(source.get('bitrate')),
+                    'tbr': int_or_none(source.get('bitrate'), scale=1000),
+                    'filesize': int_or_none(source.get('filesize')),
                      'ext': ext,
                  }
                  if source_url.startswith('rtmp'):
@@ -3586,15 +3636,15 @@ def _float(self, v, name, fatal=False, **kwargs):
  
      def _set_cookie(self, domain, name, value, expire_time=None, port=None,
                      path='/', secure=False, discard=False, rest={}, **kwargs):
-        cookie = compat_cookiejar_Cookie(
+        cookie = http.cookiejar.Cookie(
              0, name, value, port, port is not None, domain, True,
              domain.startswith('.'), path, True, secure, expire_time,
              discard, None, None, rest)
-        self._downloader.cookiejar.set_cookie(cookie)
+        self.cookiejar.set_cookie(cookie)
  
      def _get_cookies(self, url):
-        """ Return a compat_cookies_SimpleCookie with the cookies for the url """
-        return compat_cookies_SimpleCookie(self._downloader._calc_cookies(url))
+        """ Return a http.cookies.SimpleCookie with the cookies for the url """
+        return LenientSimpleCookie(self._downloader._calc_cookies(url))
  
      def _apply_first_set_cookie_header(self, url_handle, cookie):
          """
@@ -3635,11 +3685,18 @@ def get_testcases(cls, include_onlymatching=False):
              t['name'] = cls.ie_key()
              yield t
  
+    @classmethod
+    def get_webpage_testcases(cls):
+        tests = getattr(cls, '_WEBPAGE_TESTS', [])
+        for t in tests:
+            t['name'] = cls.ie_key()
+        return tests
+
      @classproperty
      def age_limit(cls):
          """Get age limit from the testcases"""
          return max(traverse_obj(
-            tuple(cls.get_testcases(include_onlymatching=False)),
+            (*cls.get_testcases(include_onlymatching=False), *cls.get_webpage_testcases()),
              (..., (('playlist', 0), None), 'info_dict', 'age_limit')) or [0])
  
      @classmethod
@@ -3664,11 +3721,12 @@ def description(cls, *, markdown=True, search_examples=None):
              desc += f'; "{cls.SEARCH_KEY}:" prefix'
              if search_examples:
                  _COUNTS = ('', '5', '10', 'all')
-                desc += f' (Example: "{cls.SEARCH_KEY}{random.choice(_COUNTS)}:{random.choice(search_examples)}")'
+                desc += f' (e.g. "{cls.SEARCH_KEY}{random.choice(_COUNTS)}:{random.choice(search_examples)}")'
          if not cls.working():
              desc += ' (**Currently broken**)' if markdown else ' (Currently broken)'
  
-        name = f' - **{cls.IE_NAME}**' if markdown else cls.IE_NAME
+        # Escape emojis. Ref: https://github.com/github/markup/issues/1153
+        name = (' - **%s**' % re.sub(r':(\w+:)', ':\u200B\\g<1>', cls.IE_NAME)) if markdown else cls.IE_NAME
          return f'{name}:{desc}' if desc else name
  
      def extract_subtitles(self, *args, **kwargs):
@@ -3759,11 +3817,15 @@ def geo_verification_headers(self):
              headers['Ytdl-request-proxy'] = geo_verification_proxy
          return headers
  
-    def _generic_id(self, url):
-        return compat_urllib_parse_unquote(os.path.splitext(url.rstrip('/').split('/')[-1])[0])
+    @staticmethod
+    def _generic_id(url):
+        return urllib.parse.unquote(os.path.splitext(url.rstrip('/').split('/')[-1])[0])
  
-    def _generic_title(self, url):
-        return compat_urllib_parse_unquote(os.path.splitext(url_basename(url))[0])
+    def _generic_title(self, url='', webpage='', *, default=None):
+        return (self._og_search_title(webpage, default=None)
+                or self._html_extract_title(webpage, default=None)
+                or urllib.parse.unquote(os.path.splitext(url_basename(url))[0])
+                or default)
  
      @staticmethod
      def _availability(is_private=None, needs_premium=None, needs_subscription=None, needs_auth=None, is_unlisted=None):
@@ -3786,8 +3848,8 @@ def _configuration_arg(self, key, default=NO_DEFAULT, *, ie_key=None, casesense=
          @param default      The default value to return when the key is not present (default: [])
          @param casesense    When false, the values are converted to lower case
          '''
-        val = traverse_obj(
-            self._downloader.params, ('extractor_args', (ie_key or self.ie_key()).lower(), key))
+        ie_key = ie_key if isinstance(ie_key, str) else (ie_key or self).ie_key()
+        val = traverse_obj(self._downloader.params, ('extractor_args', ie_key.lower(), key))
          if val is None:
              return [] if default is NO_DEFAULT else default
          return list(val) if casesense else [x.lower() for x in val]
@@ -3808,6 +3870,72 @@ def _yes_playlist(self, playlist_id, video_id, smuggled_data=None, *, playlist_l
          self.to_screen(f'Downloading {playlist_label}{playlist_id} - add --no-playlist to download just the {video_label}{video_id}')
          return True
  
+    def _error_or_warning(self, err, _count=None, _retries=0, *, fatal=True):
+        RetryManager.report_retry(
+            err, _count or int(fatal), _retries,
+            info=self.to_screen, warn=self.report_warning, error=None if fatal else self.report_warning,
+            sleep_func=self.get_param('retry_sleep_functions', {}).get('extractor'))
+
+    def RetryManager(self, **kwargs):
+        return RetryManager(self.get_param('extractor_retries', 3), self._error_or_warning, **kwargs)
+
+    def _extract_generic_embeds(self, url, *args, info_dict={}, note='Extracting generic embeds', **kwargs):
+        display_id = traverse_obj(info_dict, 'display_id', 'id')
+        self.to_screen(f'{format_field(display_id, None, "%s: ")}{note}')
+        return self._downloader.get_info_extractor('Generic')._extract_embeds(
+            smuggle_url(url, {'block_ies': [self.ie_key()]}), *args, **kwargs)
+
+    @classmethod
+    def extract_from_webpage(cls, ydl, url, webpage):
+        ie = (cls if isinstance(cls._extract_from_webpage, types.MethodType)
+              else ydl.get_info_extractor(cls.ie_key()))
+        for info in ie._extract_from_webpage(url, webpage) or []:
+            # url = None since we do not want to set (webpage/original)_url
+            ydl.add_default_extra_info(info, ie, None)
+            yield info
+
+    @classmethod
+    def _extract_from_webpage(cls, url, webpage):
+        for embed_url in orderedSet(
+                cls._extract_embed_urls(url, webpage) or [], lazy=True):
+            yield cls.url_result(embed_url, None if cls._VALID_URL is False else cls)
+
+    @classmethod
+    def _extract_embed_urls(cls, url, webpage):
+        """@returns all the embed urls on the webpage"""
+        if '_EMBED_URL_RE' not in cls.__dict__:
+            assert isinstance(cls._EMBED_REGEX, (list, tuple))
+            for idx, regex in enumerate(cls._EMBED_REGEX):
+                assert regex.count('(?P<url>') == 1, \
+                    f'{cls.__name__}._EMBED_REGEX[{idx}] must have exactly 1 url group\n\t{regex}'
+            cls._EMBED_URL_RE = tuple(map(re.compile, cls._EMBED_REGEX))
+
+        for regex in cls._EMBED_URL_RE:
+            for mobj in regex.finditer(webpage):
+                embed_url = urllib.parse.urljoin(url, unescapeHTML(mobj.group('url')))
+                if cls._VALID_URL is False or cls.suitable(embed_url):
+                    yield embed_url
+
+    class StopExtraction(Exception):
+        pass
+
+    @classmethod
+    def _extract_url(cls, webpage):  # TODO: Remove
+        """Only for compatibility with some older extractors"""
+        return next(iter(cls._extract_embed_urls(None, webpage) or []), None)
+
+    @classmethod
+    def __init_subclass__(cls, *, plugin_name=None, **kwargs):
+        if plugin_name:
+            mro = inspect.getmro(cls)
+            super_class = cls.__wrapped__ = mro[mro.index(cls) + 1]
+            cls.IE_NAME, cls.ie_key = f'{super_class.IE_NAME}+{plugin_name}', super_class.ie_key
+            while getattr(super_class, '__wrapped__', None):
+                super_class = super_class.__wrapped__
+            setattr(sys.modules[super_class.__module__], super_class.__name__, cls)
+
+        return super().__init_subclass__(**kwargs)
+
  
  class SearchInfoExtractor(InfoExtractor):
      """
@@ -3818,8 +3946,8 @@ class SearchInfoExtractor(InfoExtractor):
  
      _MAX_RESULTS = float('inf')
  
-    @classmethod
-    def _make_valid_url(cls):
+    @classproperty
+    def _VALID_URL(cls):
          return r'%s(?P<prefix>|[1-9][0-9]*|all):(?P<query>[\s\S]+)' % cls._SEARCH_KEY
  
      def _real_extract(self, query):
@@ -3851,3 +3979,12 @@ def _search_results(self, query):
      @classproperty
      def SEARCH_KEY(cls):
          return cls._SEARCH_KEY
+
+
+class UnsupportedURLIE(InfoExtractor):
+    _VALID_URL = '.*'
+    _ENABLED = False
+    IE_DESC = False
+
+    def _real_extract(self, url):
+        raise UnsupportedError(url)