[youtube:search] Support hashtag entries (#3265)

[yt-dlp.git] / yt_dlp / extractor / youtube.py
diff --git a/yt_dlp/extractor/youtube.py b/yt_dlp/extractor/youtube.py

index 4fe9cec5b3257e4d99005d31ce4cedd37502d9ef..4e6a809114be58d873b6cbf8ac38f4cfe660de2b 100644 (file)
--- a/yt_dlp/extractor/youtube.py
+++ b/yt_dlp/extractor/youtube.py
@@ -217,15 +217,35 @@
              }
          },
          'INNERTUBE_CONTEXT_CLIENT_NAME': 2
-    }
+    },
+    # This client can access age restricted videos (unless the uploader has disabled the 'allow embedding' option)
+    # See: https://github.com/zerodytrash/YouTube-Internal-Clients
+    'tv_embedded': {
+        'INNERTUBE_API_KEY': 'AIzaSyAO_FJ2SlqU8Q4STEHLGCilw_Y9_11qcW8',
+        'INNERTUBE_CONTEXT': {
+            'client': {
+                'clientName': 'TVHTML5_SIMPLY_EMBEDDED_PLAYER',
+                'clientVersion': '2.0',
+            },
+        },
+        'INNERTUBE_CONTEXT_CLIENT_NAME': 85
+    },
  }
  
  
+def _split_innertube_client(client_name):
+    variant, *base = client_name.rsplit('.', 1)
+    if base:
+        return variant, base[0], variant
+    base, *variant = client_name.split('_', 1)
+    return client_name, base, variant[0] if variant else None
+
+
  def build_innertube_clients():
      THIRD_PARTY = {
-        'embedUrl': 'https://google.com',  # Can be any valid URL
+        'embedUrl': 'https://www.youtube.com/',  # Can be any valid URL
      }
-    BASE_CLIENTS = ('android', 'web', 'ios', 'mweb')
+    BASE_CLIENTS = ('android', 'web', 'tv', 'ios', 'mweb')
      priority = qualities(BASE_CLIENTS[::-1])
  
      for client, ytcfg in tuple(INNERTUBE_CLIENTS.items()):
@@ -234,15 +254,15 @@ def build_innertube_clients():
          ytcfg.setdefault('REQUIRE_JS_PLAYER', True)
          ytcfg['INNERTUBE_CONTEXT']['client'].setdefault('hl', 'en')
  
-        base_client, *variant = client.split('_')
+        _, base_client, variant = _split_innertube_client(client)
          ytcfg['priority'] = 10 * priority(base_client)
  
          if not variant:
-            INNERTUBE_CLIENTS[f'{client}_agegate'] = agegate_ytcfg = copy.deepcopy(ytcfg)
-            agegate_ytcfg['INNERTUBE_CONTEXT']['client']['clientScreen'] = 'EMBED'
-            agegate_ytcfg['INNERTUBE_CONTEXT']['thirdParty'] = THIRD_PARTY
-            agegate_ytcfg['priority'] -= 1
-        elif variant == ['embedded']:
+            INNERTUBE_CLIENTS[f'{client}_embedscreen'] = embedscreen = copy.deepcopy(ytcfg)
+            embedscreen['INNERTUBE_CONTEXT']['client']['clientScreen'] = 'EMBED'
+            embedscreen['INNERTUBE_CONTEXT']['thirdParty'] = THIRD_PARTY
+            embedscreen['priority'] -= 3
+        elif variant == 'embedded':
              ytcfg['INNERTUBE_CONTEXT']['thirdParty'] = THIRD_PARTY
              ytcfg['priority'] -= 2
          else:
@@ -263,7 +283,7 @@ class YoutubeBaseInfoExtractor(InfoExtractor):
  
      _PLAYLIST_ID_RE = r'(?:(?:PL|LL|EC|UU|FL|RD|UL|TL|PU|OLAK5uy_)[0-9A-Za-z-_]{10,}|RDMM|WL|LL|LM)'
  
-    _NETRC_MACHINE = 'youtube'
+    # _NETRC_MACHINE = 'youtube'
  
      # If True it will raise an error if no login info is provided
      _LOGIN_REQUIRED = False
@@ -334,21 +354,6 @@ class YoutubeBaseInfoExtractor(InfoExtractor):
          r'(?:www\.)?hpniueoejy4opn7bc4ftgazyqjoeqwlvh2uiku2xqku6zpoa4bf5ruid\.onion',
      )
  
-    def _login(self):
-        """
-        Attempt to log in to YouTube.
-        If _LOGIN_REQUIRED is set and no authentication was provided, an error is raised.
-        """
-
-        if (self._LOGIN_REQUIRED
-                and self.get_param('cookiefile') is None
-                and self.get_param('cookiesfrombrowser') is None):
-            self.raise_login_required(
-                'Login details are needed to download this content', method='cookies')
-        username, password = self._get_login_info()
-        if username:
-            self.report_warning(f'Cannot login to YouTube using username and password. {self._LOGIN_HINTS["cookies"]}')
-
      def _initialize_consent(self):
          cookies = self._get_cookies('https://www.youtube.com/')
          if cookies.get('__Secure-3PSID'):
@@ -379,7 +384,10 @@ def _initialize_pref(self):
      def _real_initialize(self):
          self._initialize_pref()
          self._initialize_consent()
-        self._login()
+        if (self._LOGIN_REQUIRED
+                and self.get_param('cookiefile') is None
+                and self.get_param('cookiesfrombrowser') is None):
+            self.raise_login_required('Login details are needed to download this content', method='cookies')
  
      _YT_INITIAL_DATA_RE = r'(?:window\s*\[\s*["\']ytInitialData["\']\s*\]|ytInitialData)\s*=\s*({.+?})\s*;'
      _YT_INITIAL_PLAYER_RESPONSE_RE = r'ytInitialPlayerResponse\s*=\s*({.+?})\s*;'
@@ -458,7 +466,7 @@ def _call_api(self, ep, query, video_id, fatal=True, headers=None,
              'https://%s/youtubei/v1/%s' % (api_hostname or self._get_innertube_host(default_client), ep),
              video_id=video_id, fatal=fatal, note=note, errnote=errnote,
              data=json.dumps(data).encode('utf8'), headers=real_headers,
-            query={'key': api_key or self._extract_api_key()})
+            query={'key': api_key or self._extract_api_key(), 'prettyPrint': 'false'})
  
      def extract_yt_initial_data(self, item_id, webpage, fatal=True):
          data = self._search_regex(
@@ -819,6 +827,12 @@ def _extract_video(self, renderer):
          description = self._get_text(renderer, 'descriptionSnippet')
          duration = parse_duration(self._get_text(
              renderer, 'lengthText', ('thumbnailOverlays', ..., 'thumbnailOverlayTimeStatusRenderer', 'text')))
+        if duration is None:
+            duration = parse_duration(self._search_regex(
+                r'(?i)(ago)(?!.*\1)\s+(?P<duration>[a-z0-9 ,]+?)(?:\s+[\d,]+\s+views)?(?:\s+-\s+play\s+short)?$',
+                traverse_obj(renderer, ('title', 'accessibility', 'accessibilityData', 'label'), default='', expected_type=str),
+                video_id, default=None, group='duration'))
+
          view_count = self._get_count(renderer, 'viewCountText')
  
          uploader = self._get_text(renderer, 'ownerText', 'shortBylineText')
@@ -830,12 +844,17 @@ def _extract_video(self, renderer):
              renderer, ('thumbnailOverlays', ..., 'thumbnailOverlayTimeStatusRenderer', 'style'), get_all=False, expected_type=str)
          badges = self._extract_badges(renderer)
          thumbnails = self._extract_thumbnails(renderer, 'thumbnail')
+        navigation_url = urljoin('https://www.youtube.com/', traverse_obj(
+            renderer, ('navigationEndpoint', 'commandMetadata', 'webCommandMetadata', 'url'), expected_type=str))
+        url = f'https://www.youtube.com/watch?v={video_id}'
+        if overlay_style == 'SHORTS' or (navigation_url and '/shorts/' in navigation_url):
+            url = f'https://www.youtube.com/shorts/{video_id}'
  
          return {
              '_type': 'url',
              'ie_key': YoutubeIE.ie_key(),
              'id': video_id,
-            'url': f'https://www.youtube.com/watch?v={video_id}',
+            'url': url,
              'title': title,
              'description': description,
              'duration': duration,
@@ -1297,7 +1316,6 @@ class YoutubeIE(YoutubeBaseInfoExtractor):
              },
              'expected_warnings': [
                  'DASH manifest missing',
-                'Some formats are possibly damaged'
              ]
          },
          # Olympics (https://github.com/ytdl-org/youtube-dl/issues/4431)
@@ -2953,13 +2971,19 @@ def _extract_player_responses(self, clients, video_id, webpage, master_ytcfg):
                  webpage, self._YT_INITIAL_PLAYER_RESPONSE_RE,
                  video_id, 'initial player response')
  
-        original_clients = clients
+        all_clients = set(clients)
          clients = clients[::-1]
          prs = []
  
-        def append_client(client_name):
-            if client_name in INNERTUBE_CLIENTS and client_name not in original_clients:
-                clients.append(client_name)
+        def append_client(*client_names):
+            """ Append the first client name that exists but not already used """
+            for client_name in client_names:
+                actual_client = _split_innertube_client(client_name)[0]
+                if actual_client in INNERTUBE_CLIENTS:
+                    if actual_client not in all_clients:
+                        clients.append(client_name)
+                        all_clients.add(actual_client)
+                        return
  
          # Android player_response does not have microFormats which are needed for
          # extraction of some data. So we return the initial_pr with formats
@@ -2974,7 +2998,7 @@ def append_client(client_name):
          tried_iframe_fallback = False
          player_url = None
          while clients:
-            client = clients.pop()
+            client, base_client, variant = _split_innertube_client(clients.pop())
              player_ytcfg = master_ytcfg if client == 'web' else {}
              if 'configs' not in self._configuration_arg('player_skip'):
                  player_ytcfg = self._extract_player_ytcfg(client, video_id) or player_ytcfg
@@ -3002,10 +3026,13 @@ def append_client(client_name):
                  prs.append(pr)
  
              # creator clients can bypass AGE_VERIFICATION_REQUIRED if logged in
-            if client.endswith('_agegate') and self._is_unplayable(pr) and self.is_authenticated:
-                append_client(client.replace('_agegate', '_creator'))
+            if variant == 'embedded' and self._is_unplayable(pr) and self.is_authenticated:
+                append_client(f'{base_client}_creator')
              elif self._is_agegated(pr):
-                append_client(f'{client}_agegate')
+                if variant == 'tv_embedded':
+                    append_client(f'{base_client}_embedded')
+                elif not variant:
+                    append_client(f'tv_embedded.{base_client}', f'{base_client}_embedded')
  
          if last_error:
              if not len(prs):
@@ -3013,7 +3040,7 @@ def append_client(client_name):
              self.report_warning(last_error)
          return prs, player_url
  
-    def _extract_formats(self, streaming_data, video_id, player_url, is_live):
+    def _extract_formats(self, streaming_data, video_id, player_url, is_live, duration):
          itags, stream_ids = {}, []
          itag_qualities, res_qualities = {}, {}
          q = qualities([
@@ -3024,10 +3051,9 @@ def _extract_formats(self, streaming_data, video_id, player_url, is_live):
              'small', 'medium', 'large', 'hd720', 'hd1080', 'hd1440', 'hd2160', 'hd2880', 'highres'
          ])
          streaming_formats = traverse_obj(streaming_data, (..., ('formats', 'adaptiveFormats'), ...), default=[])
-        approx_duration = max(traverse_obj(streaming_formats, (..., 'approxDurationMs'), expected_type=float_or_none) or [0]) or None
  
          for fmt in streaming_formats:
-            if fmt.get('targetDurationSec') or fmt.get('drmFamilies'):
+            if fmt.get('targetDurationSec'):
                  continue
  
              itag = str_or_none(fmt.get('itag'))
@@ -3091,7 +3117,9 @@ def _extract_formats(self, streaming_data, video_id, player_url, is_live):
                  else -1)
              # Some formats may have much smaller duration than others (possibly damaged during encoding)
              # Eg: 2-nOtRESiUc Ref: https://github.com/yt-dlp/yt-dlp/issues/2823
-            is_damaged = try_get(fmt, lambda x: float(x['approxDurationMs']) < approx_duration - 10000)
+            # Make sure to avoid false positives with small duration differences.
+            # Eg: __2ABJjxzNo, ySuUZEjARPY
+            is_damaged = try_get(fmt, lambda x: float(x['approxDurationMs']) / duration < 500)
              if is_damaged:
                  self.report_warning(f'{video_id}: Some formats are possibly damaged. They will be deprioritized', only_once=True)
              dct = {
@@ -3107,6 +3135,7 @@ def _extract_formats(self, streaming_data, video_id, player_url, is_live):
                  'fps': int_or_none(fmt.get('fps')) or None,
                  'height': height,
                  'quality': q(quality),
+                'has_drm': bool(fmt.get('drmFamilies')),
                  'tbr': tbr,
                  'url': fmt_url,
                  'width': int_or_none(fmt.get('width')),
@@ -3227,14 +3256,14 @@ def _download_player_responses(self, url, smuggled_data, video_id, webpage_url):
  
          return webpage, master_ytcfg, player_responses, player_url
  
-    def _list_formats(self, video_id, microformats, video_details, player_responses, player_url):
+    def _list_formats(self, video_id, microformats, video_details, player_responses, player_url, duration=None):
          live_broadcast_details = traverse_obj(microformats, (..., 'liveBroadcastDetails'))
          is_live = get_first(video_details, 'isLive')
          if is_live is None:
              is_live = get_first(live_broadcast_details, 'isLiveNow')
  
          streaming_data = traverse_obj(player_responses, (..., 'streamingData'), default=[])
-        formats = list(self._extract_formats(streaming_data, video_id, player_url, is_live))
+        formats = list(self._extract_formats(streaming_data, video_id, player_url, is_live, duration))
  
          return live_broadcast_details, is_live, streaming_data, formats
  
@@ -3315,7 +3344,13 @@ def feed_entry(name):
                  return self.playlist_result(
                      entries, video_id, video_title, video_description)
  
-        live_broadcast_details, is_live, streaming_data, formats = self._list_formats(video_id, microformats, video_details, player_responses, player_url)
+        duration = int_or_none(
+            get_first(video_details, 'lengthSeconds')
+            or get_first(microformats, 'lengthSeconds')
+            or parse_duration(search_meta('duration'))) or None
+
+        live_broadcast_details, is_live, streaming_data, formats = self._list_formats(
+            video_id, microformats, video_details, player_responses, player_url, duration)
  
          if not formats:
              if not self.get_param('allow_unplayable_formats') and traverse_obj(streaming_data, (..., 'licenseInfos')):
@@ -3387,10 +3422,6 @@ def feed_entry(name):
              get_first(video_details, 'channelId')
              or get_first(microformats, 'externalChannelId')
              or search_meta('channelId'))
-        duration = int_or_none(
-            get_first(video_details, 'lengthSeconds')
-            or get_first(microformats, 'lengthSeconds')
-            or parse_duration(search_meta('duration'))) or None
          owner_profile_url = get_first(microformats, 'ownerProfileUrl')
  
          live_content = get_first(video_details, 'isLiveContent')
@@ -3478,6 +3509,7 @@ def process_language(container, base_url, lang_code, sub_name, query):
              subtitles, automatic_captions = {}, {}
              for lang_code, caption_track in captions.items():
                  base_url = caption_track.get('baseUrl')
+                orig_lang = parse_qs(base_url).get('lang', [None])[-1]
                  if not base_url:
                      continue
                  lang_name = self._get_text(caption_track, 'name', max_runs=1)
@@ -3491,19 +3523,20 @@ def process_language(container, base_url, lang_code, sub_name, query):
                  for trans_code, trans_name in translation_languages.items():
                      if not trans_code:
                          continue
+                    orig_trans_code = trans_code
                      if caption_track.get('kind') != 'asr':
+                        if 'translated_subs' in self._configuration_arg('skip'):
+                            continue
                          trans_code += f'-{lang_code}'
                          trans_name += format_field(lang_name, template=' from %s')
                      # Add an "-orig" label to the original language so that it can be distinguished.
                      # The subs are returned without "-orig" as well for compatibility
-                    if lang_code == f'a-{trans_code}':
+                    if lang_code == f'a-{orig_trans_code}':
                          process_language(
                              automatic_captions, base_url, f'{trans_code}-orig', f'{trans_name} (Original)', {})
                      # Setting tlang=lang returns damaged subtitles.
-                    # Not using lang_code == f'a-{trans_code}' here for future-proofing
-                    orig_lang = parse_qs(base_url).get('lang', [None])[-1]
                      process_language(automatic_captions, base_url, trans_code, trans_name,
-                                     {} if orig_lang == trans_code else {'tlang': trans_code})
+                                     {} if orig_lang == orig_trans_code else {'tlang': trans_code})
              info['automatic_captions'] = automatic_captions
              info['subtitles'] = subtitles
  
@@ -3870,6 +3903,13 @@ def _video_entry(self, video_renderer):
          if video_id:
              return self._extract_video(video_renderer)
  
+    def _hashtag_tile_entry(self, hashtag_tile_renderer):
+        url = urljoin('https://youtube.com', traverse_obj(
+            hashtag_tile_renderer, ('onTapCommand', 'commandMetadata', 'webCommandMetadata', 'url')))
+        if url:
+            return self.url_result(
+                url, ie=YoutubeTabIE.ie_key(), title=self._get_text(hashtag_tile_renderer, 'hashtag'))
+
      def _post_thread_entries(self, post_thread_renderer):
          post_renderer = try_get(
              post_thread_renderer, lambda x: x['post']['backstagePostRenderer'], dict)
@@ -3926,6 +3966,7 @@ def _rich_grid_entries(self, contents):
                  if entry:
                      yield entry
      '''
+
      def _extract_entries(self, parent_renderer, continuation_list):
          # continuation_list is modified in-place with continuation_list = [continuation_token]
          continuation_list[:] = [None]
@@ -3957,6 +3998,7 @@ def _extract_entries(self, parent_renderer, continuation_list):
                      'videoRenderer': lambda x: [self._video_entry(x)],
                      'playlistRenderer': lambda x: self._grid_entries({'items': [{'playlistRenderer': x}]}),
                      'channelRenderer': lambda x: self._grid_entries({'items': [{'channelRenderer': x}]}),
+                    'hashtagTileRenderer': lambda x: [self._hashtag_tile_entry(x)]
                  }
                  for key, renderer in isr_content.items():
                      if key not in known_renderers:
@@ -4024,6 +4066,7 @@ def _entries(self, tab, item_id, ytcfg, account_syncid, visitor_data):
                  continue
  
              known_renderers = {
+                'videoRenderer': (self._grid_entries, 'items'),  # for membership tab
                  'gridPlaylistRenderer': (self._grid_entries, 'items'),
                  'gridVideoRenderer': (self._grid_entries, 'items'),
                  'gridChannelRenderer': (self._grid_entries, 'items'),
@@ -5485,7 +5528,17 @@ class YoutubeSearchURLIE(YoutubeTabBaseInfoExtractor):
              'id': 'python',
              'title': 'python',
          }
-
+    }, {
+        'url': 'https://www.youtube.com/results?search_query=%23cats',
+        'playlist_mincount': 1,
+        'info_dict': {
+            'id': '#cats',
+            'title': '#cats',
+            'entries': [{
+                'url': r're:https://(www\.)?youtube\.com/hashtag/cats',
+                'title': '#cats',
+            }],
+        },
      }, {
          'url': 'https://www.youtube.com/results?q=test&sp=EgQIBBgB',
          'only_matching': True,