[youtube] Add fallback metadata extraction from videoDetails (closes #18052)

[yt-dlp.git] / youtube_dl / extractor / youtube.py
diff --git a/youtube_dl/extractor/youtube.py b/youtube_dl/extractor/youtube.py

index e80e36f988196dfa2490e99013134008dc2066bd..abadfa5455f95a9270f3a933fd5222d6f9c5bb4a 100644 (file)
--- a/youtube_dl/extractor/youtube.py
+++ b/youtube_dl/extractor/youtube.py
@@ -41,6 +41,7 @@
      remove_quotes,
      remove_start,
      smuggle_url,
+    str_or_none,
      str_to_int,
      try_get,
      unescapeHTML,
@@ -349,6 +350,7 @@ class YoutubeIE(YoutubeBaseInfoExtractor):
                              (?:www\.)?hooktube\.com/|
                              (?:www\.)?yourepeat\.com/|
                              tube\.majestyc\.net/|
+                            (?:www\.)?invidio\.us/|
                              youtube\.googleapis\.com/)                        # the various hostnames, with wildcard subdomains
                           (?:.*?\#/)?                                          # handle anchor (#/) redirect urls
                           (?:                                                  # the various things that can precede the ID:
@@ -500,6 +502,7 @@ class YoutubeIE(YoutubeBaseInfoExtractor):
                  'categories': ['Science & Technology'],
                  'tags': ['youtube-dl'],
                  'duration': 10,
+                'view_count': int,
                  'like_count': int,
                  'dislike_count': int,
                  'start_time': 1,
@@ -582,6 +585,7 @@ class YoutubeIE(YoutubeBaseInfoExtractor):
                  'categories': ['Science & Technology'],
                  'tags': ['youtube-dl'],
                  'duration': 10,
+                'view_count': int,
                  'like_count': int,
                  'dislike_count': int,
              },
@@ -1068,6 +1072,10 @@ class YoutubeIE(YoutubeBaseInfoExtractor):
              'url': 'https://www.youtube.com/watch?v=MuAGGZNfUkU&list=RDMM',
              'only_matching': True,
          },
+        {
+            'url': 'https://invidio.us/watch?v=BaW_jenozKc',
+            'only_matching': True,
+        },
      ]
  
      def __init__(self, *args, **kwargs):
@@ -1533,6 +1541,8 @@ def add_dash_mpd(video_info):
          def extract_view_count(v_info):
              return int_or_none(try_get(v_info, lambda x: x['view_count'][0]))
  
+        player_response = {}
+
          # Get video info
          embed_webpage = None
          if re.search(r'player-age-gate-content">', video_webpage) is not None:
@@ -1575,6 +1585,12 @@ def extract_view_count(v_info):
                  if args.get('livestream') == '1' or args.get('live_playback') == 1:
                      is_live = True
                  sts = ytplayer_config.get('sts')
+                if not player_response:
+                    pl_response = str_or_none(args.get('player_response'))
+                    if pl_response:
+                        pl_response = self._parse_json(pl_response, video_id, fatal=False)
+                        if isinstance(pl_response, dict):
+                            player_response = pl_response
              if not video_info or self._downloader.params.get('youtube_include_dash_manifest', True):
                  # We also try looking in get_video_info since it may contain different dashmpd
                  # URL that points to a DASH manifest with possibly different itag set (some itags
@@ -1603,6 +1619,10 @@ def extract_view_count(v_info):
                      if not video_info_webpage:
                          continue
                      get_video_info = compat_parse_qs(video_info_webpage)
+                    if not player_response:
+                        pl_response = get_video_info.get('player_response', [None])[0]
+                        if isinstance(pl_response, dict):
+                            player_response = pl_response
                      add_dash_mpd(get_video_info)
                      if view_count is None:
                          view_count = extract_view_count(get_video_info)
@@ -1648,9 +1668,14 @@ def extract_unavailable_message():
                      '"token" parameter not in video info for unknown reason',
                      video_id=video_id)
  
+        video_details = try_get(
+            player_response, lambda x: x['videoDetails'], dict) or {}
+
          # title
          if 'title' in video_info:
              video_title = video_info['title'][0]
+        elif 'title' in player_response:
+            video_title = video_details['title']
          else:
              self._downloader.report_warning('Unable to extract video title')
              video_title = '_'
@@ -1713,6 +1738,8 @@ def replace_url(m):
  
          if view_count is None:
              view_count = extract_view_count(video_info)
+        if view_count is None and video_details:
+            view_count = int_or_none(video_details.get('viewCount'))
  
          # Check for "rental" videos
          if 'ypc_video_rental_bar_text' in video_info and 'author' not in video_info:
@@ -1893,7 +1920,9 @@ def _extract_filesize(media_url):
              raise ExtractorError('no conn, hlsvp or url_encoded_fmt_stream_map information found in video info')
  
          # uploader
-        video_uploader = try_get(video_info, lambda x: x['author'][0], compat_str)
+        video_uploader = try_get(
+            video_info, lambda x: x['author'][0],
+            compat_str) or str_or_none(video_details.get('author'))
          if video_uploader:
              video_uploader = compat_urllib_parse_unquote_plus(video_uploader)
          else:
@@ -2006,12 +2035,19 @@ def _extract_count(count_name):
          like_count = _extract_count('like')
          dislike_count = _extract_count('dislike')
  
+        if view_count is None:
+            view_count = str_to_int(self._search_regex(
+                r'<[^>]+class=["\']watch-view-count[^>]+>\s*([\d,\s]+)', video_webpage,
+                'view count', default=None))
+
          # subtitles
          video_subtitles = self.extract_subtitles(video_id, video_webpage)
          automatic_captions = self.extract_automatic_captions(video_id, video_webpage)
  
          video_duration = try_get(
              video_info, lambda x: int_or_none(x['length_seconds'][0]))
+        if not video_duration:
+            video_duration = int_or_none(video_details.get('lengthSeconds'))
          if not video_duration:
              video_duration = parse_duration(self._html_search_meta(
                  'duration', video_webpage, 'video duration'))
@@ -2239,6 +2275,7 @@ class YoutubePlaylistIE(YoutubePlaylistBaseInfoExtractor):
              'description': 'md5:507cdcb5a49ac0da37a920ece610be80',
              'categories': ['People & Blogs'],
              'tags': list,
+            'view_count': int,
              'like_count': int,
              'dislike_count': int,
          },
@@ -2419,7 +2456,7 @@ def _real_extract(self, url):
  
  class YoutubeChannelIE(YoutubePlaylistBaseInfoExtractor):
      IE_DESC = 'YouTube.com channels'
-    _VALID_URL = r'https?://(?:youtu\.be|(?:\w+\.)?youtube(?:-nocookie)?\.com)/channel/(?P<id>[0-9A-Za-z_-]+)'
+    _VALID_URL = r'https?://(?:youtu\.be|(?:\w+\.)?youtube(?:-nocookie)?\.com|(?:www\.)?invidio\.us)/channel/(?P<id>[0-9A-Za-z_-]+)'
      _TEMPLATE_URL = 'https://www.youtube.com/channel/%s/videos'
      _VIDEO_RE = r'(?:title="(?P<title>[^"]+)"[^>]+)?href="/watch\?v=(?P<id>[0-9A-Za-z_-]+)&?'
      IE_NAME = 'youtube:channel'
@@ -2440,6 +2477,9 @@ class YoutubeChannelIE(YoutubePlaylistBaseInfoExtractor):
              'id': 'UUs0ifCMCm1icqRbqhUINa0w',
              'title': 'Uploads from Deus Ex',
          },
+    }, {
+        'url': 'https://invidio.us/channel/UC23qupoDRn9YOAVzeoxjOQA',
+        'only_matching': True,
      }]
  
      @classmethod