[youtube] Update signature function patterns (closes #21469, closes #21476)

[youtube-dl] / youtube_dl / extractor / youtube.py
diff --git a/youtube_dl/extractor/youtube.py b/youtube_dl/extractor/youtube.py

index 438eb5aa7d371f0d0fd9c70b9c8b7d15a8d44737..83b6ac1346580a65221b6c1db90751e7e4141915 100644 (file)
--- a/youtube_dl/extractor/youtube.py
+++ b/youtube_dl/extractor/youtube.py
@@ -16,6 +16,7 @@ from ..jsinterp import JSInterpreter
  from ..swfinterp import SWFInterpreter
  from ..compat import (
      compat_chr,
+    compat_HTTPError,
      compat_kwargs,
      compat_parse_qs,
      compat_urllib_parse_unquote,
@@ -27,6 +28,7 @@ from ..compat import (
  )
  from ..utils import (
      clean_html,
+    dict_get,
      error_to_compat_str,
      ExtractorError,
      float_or_none,
@@ -287,10 +289,25 @@ class YoutubeEntryListBaseInfoExtractor(YoutubeBaseInfoExtractor):
              if not mobj:
                  break
  
-            more = self._download_json(
-                'https://youtube.com/%s' % mobj.group('more'), playlist_id,
-                'Downloading page #%s' % page_num,
-                transform_source=uppercase_escape)
+            count = 0
+            retries = 3
+            while count <= retries:
+                try:
+                    # Downloading page may result in intermittent 5xx HTTP error
+                    # that is usually worked around with a retry
+                    more = self._download_json(
+                        'https://youtube.com/%s' % mobj.group('more'), playlist_id,
+                        'Downloading page #%s%s'
+                        % (page_num, ' (retry #%d)' % count if count else ''),
+                        transform_source=uppercase_escape)
+                    break
+                except ExtractorError as e:
+                    if isinstance(e.cause, compat_HTTPError) and e.cause.code in (500, 503):
+                        count += 1
+                        if count <= retries:
+                            continue
+                    raise
+
              content_html = more['content_html']
              if not content_html.strip():
                  # Some webpages show a "Load more" button but they don't
@@ -483,6 +500,12 @@ class YoutubeIE(YoutubeBaseInfoExtractor):
  
          # RTMP (unnamed)
          '_rtmp': {'protocol': 'rtmp'},
+
+        # av01 video only formats sometimes served with "unknown" codecs
+        '394': {'acodec': 'none', 'vcodec': 'av01.0.05M.08'},
+        '395': {'acodec': 'none', 'vcodec': 'av01.0.05M.08'},
+        '396': {'acodec': 'none', 'vcodec': 'av01.0.05M.08'},
+        '397': {'acodec': 'none', 'vcodec': 'av01.0.05M.08'},
      }
      _SUBTITLE_FORMATS = ('srv1', 'srv2', 'srv3', 'ttml', 'vtt')
  
@@ -908,6 +931,7 @@ class YoutubeIE(YoutubeBaseInfoExtractor):
                  'creator': 'Todd Haberman,  Daniel Law Heath and Aaron Kaplan',
                  'track': 'Dark Walk - Position Music',
                  'artist': 'Todd Haberman,  Daniel Law Heath and Aaron Kaplan',
+                'album': 'Position Music - Production Music Vol. 143 - Dark Walk',
              },
              'params': {
                  'skip_download': True,
@@ -1088,7 +1112,7 @@ class YoutubeIE(YoutubeBaseInfoExtractor):
              },
          },
          {
-            # artist and track fields should return non-null, per issue #20599
+            # Youtube Music Auto-generated description
              'url': 'https://music.youtube.com/watch?v=MgNrAu2pzNs',
              'info_dict': {
                  'id': 'MgNrAu2pzNs',
@@ -1109,11 +1133,9 @@ class YoutubeIE(YoutubeBaseInfoExtractor):
              },
          },
          {
+            # Youtube Music Auto-generated description
              # Retrieve 'artist' field from 'Artist:' in video description
              # when it is present on youtube music video
-            # Some videos have release_date and no release_year -
-            # (release_year should be extracted from release_date)
-            # https://github.com/ytdl-org/youtube-dl/pull/20742#issuecomment-485740932
              'url': 'https://www.youtube.com/watch?v=k0jLE7tTwjY',
              'info_dict': {
                  'id': 'k0jLE7tTwjY',
@@ -1134,6 +1156,7 @@ class YoutubeIE(YoutubeBaseInfoExtractor):
              },
          },
          {
+            # Youtube Music Auto-generated description
              # handle multiple artists on youtube music video
              'url': 'https://www.youtube.com/watch?v=74qn0eJSjpA',
              'info_dict': {
@@ -1155,6 +1178,7 @@ class YoutubeIE(YoutubeBaseInfoExtractor):
              },
          },
          {
+            # Youtube Music Auto-generated description
              # handle youtube music video with release_year and no release_date
              'url': 'https://www.youtube.com/watch?v=-hcAI0g-f5M',
              'info_dict': {
@@ -1288,11 +1312,17 @@ class YoutubeIE(YoutubeBaseInfoExtractor):
  
      def _parse_sig_js(self, jscode):
          funcname = self._search_regex(
-            (r'(["\'])signature\1\s*,\s*(?P<sig>[a-zA-Z0-9$]+)\(',
+            (r'\b[cs]\s*&&\s*[adf]\.set\([^,]+\s*,\s*encodeURIComponent\s*\(\s*(?P<sig>[a-zA-Z0-9$]+)\(',
+             r'\b[a-zA-Z0-9]+\s*&&\s*[a-zA-Z0-9]+\.set\([^,]+\s*,\s*encodeURIComponent\s*\(\s*(?P<sig>[a-zA-Z0-9$]+)\(',
+             # Obsolete patterns
+             r'(["\'])signature\1\s*,\s*(?P<sig>[a-zA-Z0-9$]+)\(',
               r'\.sig\|\|(?P<sig>[a-zA-Z0-9$]+)\(',
-             r'yt\.akamaized\.net/\)\s*\|\|\s*.*?\s*c\s*&&\s*d\.set\([^,]+\s*,\s*(?:encodeURIComponent\s*\()?(?P<sig>[a-zA-Z0-9$]+)\(',
-             r'\bc\s*&&\s*d\.set\([^,]+\s*,\s*(?:encodeURIComponent\s*\()?\s*(?P<sig>[a-zA-Z0-9$]+)\(',
-             r'\bc\s*&&\s*d\.set\([^,]+\s*,\s*\([^)]*\)\s*\(\s*(?P<sig>[a-zA-Z0-9$]+)\('),
+             r'yt\.akamaized\.net/\)\s*\|\|\s*.*?\s*[cs]\s*&&\s*[adf]\.set\([^,]+\s*,\s*(?:encodeURIComponent\s*\()?\s*(?P<sig>[a-zA-Z0-9$]+)\(',
+             r'\b[cs]\s*&&\s*[adf]\.set\([^,]+\s*,\s*(?P<sig>[a-zA-Z0-9$]+)\(',
+             r'\b[a-zA-Z0-9]+\s*&&\s*[a-zA-Z0-9]+\.set\([^,]+\s*,\s*(?P<sig>[a-zA-Z0-9$]+)\(',
+             r'\bc\s*&&\s*a\.set\([^,]+\s*,\s*\([^)]*\)\s*\(\s*(?P<sig>[a-zA-Z0-9$]+)\(',
+             r'\bc\s*&&\s*[a-zA-Z0-9]+\.set\([^,]+\s*,\s*\([^)]*\)\s*\(\s*(?P<sig>[a-zA-Z0-9$]+)\(',
+             r'\bc\s*&&\s*[a-zA-Z0-9]+\.set\([^,]+\s*,\s*\([^)]*\)\s*\(\s*(?P<sig>[a-zA-Z0-9$]+)\('),
              jscode, 'Initial JS player signature function name', group='sig')
  
          jsi = JSInterpreter(jscode)
@@ -1557,8 +1587,15 @@ class YoutubeIE(YoutubeBaseInfoExtractor):
          return video_id
  
      def _extract_annotations(self, video_id):
-        url = 'https://www.youtube.com/annotations_invideo?features=1&legacy=1&video_id=%s' % video_id
-        return self._download_webpage(url, video_id, note='Searching for annotations.', errnote='Unable to download video annotations.')
+        return self._download_webpage(
+            'https://www.youtube.com/annotations_invideo', video_id,
+            note='Downloading annotations',
+            errnote='Unable to download video annotations', fatal=False,
+            query={
+                'features': 1,
+                'legacy': 1,
+                'video_id': video_id,
+            })
  
      @staticmethod
      def _extract_chapters(description, duration):
@@ -1651,6 +1688,9 @@ class YoutubeIE(YoutubeBaseInfoExtractor):
          def extract_view_count(v_info):
              return int_or_none(try_get(v_info, lambda x: x['view_count'][0]))
  
+        def extract_token(v_info):
+            return dict_get(v_info, ('account_playback_token', 'accountPlaybackToken', 'token'))
+
          player_response = {}
  
          # Get video info
@@ -1710,7 +1750,7 @@ class YoutubeIE(YoutubeBaseInfoExtractor):
                  # The general idea is to take a union of itags of both DASH manifests (for example
                  # video with such 'manifest behavior' see https://github.com/ytdl-org/youtube-dl/issues/6093)
                  self.report_video_info_webpage_download(video_id)
-                for el in ('info', 'embedded', 'detailpage', 'vevo', ''):
+                for el in ('embedded', 'detailpage', 'vevo', ''):
                      query = {
                          'video_id': video_id,
                          'ps': 'default',
@@ -1740,7 +1780,7 @@ class YoutubeIE(YoutubeBaseInfoExtractor):
                          view_count = extract_view_count(get_video_info)
                      if not video_info:
                          video_info = get_video_info
-                    get_token = get_video_info.get('token') or get_video_info.get('account_playback_token')
+                    get_token = extract_token(get_video_info)
                      if get_token:
                          # Different get_video_info requests may report different results, e.g.
                          # some may report video unavailability, but some may serve it without
@@ -1751,7 +1791,7 @@ class YoutubeIE(YoutubeBaseInfoExtractor):
                          # due to YouTube measures against IP ranges of hosting providers.
                          # Working around by preferring the first succeeded video_info containing
                          # the token if no such video_info yet was found.
-                        token = video_info.get('token') or video_info.get('account_playback_token')
+                        token = extract_token(video_info)
                          if not token:
                              video_info = get_video_info
                          break
@@ -1768,31 +1808,6 @@ class YoutubeIE(YoutubeBaseInfoExtractor):
              raise ExtractorError(
                  'YouTube said: %s' % unavailable_message, expected=True, video_id=video_id)
  
-        token = video_info.get('token') or video_info.get('account_playback_token')
-        if not token:
-            if 'reason' in video_info:
-                if 'The uploader has not made this video available in your country.' in video_info['reason']:
-                    regions_allowed = self._html_search_meta(
-                        'regionsAllowed', video_webpage, default=None)
-                    countries = regions_allowed.split(',') if regions_allowed else None
-                    self.raise_geo_restricted(
-                        msg=video_info['reason'][0], countries=countries)
-                reason = video_info['reason'][0]
-                if 'Invalid parameters' in reason:
-                    unavailable_message = extract_unavailable_message()
-                    if unavailable_message:
-                        reason = unavailable_message
-                raise ExtractorError(
-                    'YouTube said: %s' % reason,
-                    expected=True, video_id=video_id)
-            else:
-                raise ExtractorError(
-                    '"token" parameter not in video info for unknown reason',
-                    video_id=video_id)
-
-        if video_info.get('license_info'):
-            raise ExtractorError('This video is DRM protected.', expected=True)
-
          video_details = try_get(
              player_response, lambda x: x['videoDetails'], dict) or {}
  
@@ -1928,7 +1943,7 @@ class YoutubeIE(YoutubeBaseInfoExtractor):
              formats = []
              for url_data_str in encoded_url_map.split(','):
                  url_data = compat_parse_qs(url_data_str)
-                if 'itag' not in url_data or 'url' not in url_data:
+                if 'itag' not in url_data or 'url' not in url_data or url_data.get('drm_families'):
                      continue
                  stream_type = int_or_none(try_get(url_data, lambda x: x['stream_type'][0]))
                  # Unsupported FORMAT_STREAM_TYPE_OTF
@@ -1988,7 +2003,8 @@ class YoutubeIE(YoutubeBaseInfoExtractor):
  
                      signature = self._decrypt_signature(
                          encrypted_sig, video_id, player_url, age_gate)
-                    url += '&signature=' + signature
+                    sp = try_get(url_data, lambda x: x['sp'][0], compat_str) or 'signature'
+                    url += '&%s=%s' % (sp, signature)
                  if 'ratebypass' not in url:
                      url += '&ratebypass=yes'
  
@@ -2052,8 +2068,8 @@ class YoutubeIE(YoutubeBaseInfoExtractor):
                  url_or_none(try_get(
                      player_response,
                      lambda x: x['streamingData']['hlsManifestUrl'],
-                    compat_str)) or
-                url_or_none(try_get(
+                    compat_str))
+                or url_or_none(try_get(
                      video_info, lambda x: x['hlsvp'][0], compat_str)))
              if manifest_url:
                  formats = []
@@ -2101,8 +2117,13 @@ class YoutubeIE(YoutubeBaseInfoExtractor):
          else:
              self._downloader.report_warning('unable to extract uploader nickname')
  
-        channel_id = self._html_search_meta(
-            'channelId', video_webpage, 'channel id')
+        channel_id = (
+            str_or_none(video_details.get('channelId'))
+            or self._html_search_meta(
+                'channelId', video_webpage, 'channel id', default=None)
+            or self._search_regex(
+                r'data-channel-external-id=(["\'])(?P<id>(?:(?!\1).)+)\1',
+                video_webpage, 'channel id', default=None, group='id'))
          channel_url = 'http://www.youtube.com/channel/%s' % channel_id if channel_id else None
  
          # thumbnail image
@@ -2161,36 +2182,27 @@ class YoutubeIE(YoutubeBaseInfoExtractor):
  
          track = extract_meta('Song')
          artist = extract_meta('Artist')
-        album = None
-        release_date = None
-        release_year = None
-
-        description_info = video_description.split('\n\n')
-        # If the description of the video has the youtube music auto-generated format, extract additional info
-        if len(description_info) >= 5 and description_info[-1] == 'Auto-generated by YouTube.':
-            track_artist = description_info[1].split(' · ')
-            if len(track_artist) >= 2:
-                if track is None:
-                    track = track_artist[0]
-                if artist is None:
-                    artist = re.search(r'Artist: ([^\n]+)', description_info[-2])
-                    if artist:
-                        artist = artist.group(1)
-                    if artist is None:
-                        artist = track_artist[1]
-                        # handle multiple artists
-                        if len(track_artist) > 2:
-                            for i in range(2, len(track_artist)):
-                                artist += ', %s' % track_artist[i]
-            release_year = re.search(r'℗ ([0-9]+)', video_description)
-            if release_year:
-                release_year = int_or_none(release_year.group(1))
-            album = description_info[2]
-            if description_info[4].startswith('Released on: '):
-                release_date = description_info[4].split(': ')[1].replace('-', '')
-                # extract release_year from release_date if necessary
-                if release_year is None:
-                    release_year = int_or_none(release_date[0:4])
+        album = extract_meta('Album')
+
+        # Youtube Music Auto-generated description
+        release_date = release_year = None
+        if video_description:
+            mobj = re.search(r'(?s)Provided to YouTube by [^\n]+\n+(?P<track>[^·]+)·(?P<artist>[^\n]+)\n+(?P<album>[^\n]+)(?:.+?℗\s*(?P<release_year>\d{4})(?!\d))?(?:.+?Released on\s*:\s*(?P<release_date>\d{4}-\d{2}-\d{2}))?(.+?\nArtist\s*:\s*(?P<clean_artist>[^\n]+))?', video_description)
+            if mobj:
+                if not track:
+                    track = mobj.group('track').strip()
+                if not artist:
+                    artist = mobj.group('clean_artist') or ', '.join(a.strip() for a in mobj.group('artist').split('·'))
+                if not album:
+                    album = mobj.group('album'.strip())
+                release_year = mobj.group('release_year')
+                release_date = mobj.group('release_date')
+                if release_date:
+                    release_date = release_date.replace('-', '')
+                    if not release_year:
+                        release_year = int(release_date[:4])
+                if release_year:
+                    release_year = int(release_year)
  
          m_episode = re.search(
              r'<div[^>]+id="watch7-headline"[^>]*>\s*<span[^>]*>.*?>(?P<series>[^<]+)</a></b>\s*S(?P<season>\d+)\s*•\s*E(?P<episode>\d+)</span>',
@@ -2231,6 +2243,10 @@ class YoutubeIE(YoutubeBaseInfoExtractor):
                  r'<[^>]+class=["\']watch-view-count[^>]+>\s*([\d,\s]+)', video_webpage,
                  'view count', default=None))
  
+        average_rating = (
+            float_or_none(video_details.get('averageRating'))
+            or try_get(video_info, lambda x: float_or_none(x['avg_rating'][0])))
+
          # subtitles
          video_subtitles = self.extract_subtitles(video_id, video_webpage)
          automatic_captions = self.extract_automatic_captions(video_id, video_webpage)
@@ -2304,6 +2320,32 @@ class YoutubeIE(YoutubeBaseInfoExtractor):
                      if f.get('vcodec') != 'none':
                          f['stretched_ratio'] = ratio
  
+        if not formats:
+            token = extract_token(video_info)
+            if not token:
+                if 'reason' in video_info:
+                    if 'The uploader has not made this video available in your country.' in video_info['reason']:
+                        regions_allowed = self._html_search_meta(
+                            'regionsAllowed', video_webpage, default=None)
+                        countries = regions_allowed.split(',') if regions_allowed else None
+                        self.raise_geo_restricted(
+                            msg=video_info['reason'][0], countries=countries)
+                    reason = video_info['reason'][0]
+                    if 'Invalid parameters' in reason:
+                        unavailable_message = extract_unavailable_message()
+                        if unavailable_message:
+                            reason = unavailable_message
+                    raise ExtractorError(
+                        'YouTube said: %s' % reason,
+                        expected=True, video_id=video_id)
+                else:
+                    raise ExtractorError(
+                        '"token" parameter not in video info for unknown reason',
+                        video_id=video_id)
+
+        if not formats and (video_info.get('license_info') or try_get(player_response, lambda x: x['streamingData']['licenseInfos'])):
+            raise ExtractorError('This video is DRM protected.', expected=True)
+
          self._sort_formats(formats)
  
          self.mark_watched(video_id, video_info, player_response)
@@ -2334,7 +2376,7 @@ class YoutubeIE(YoutubeBaseInfoExtractor):
              'view_count': view_count,
              'like_count': like_count,
              'dislike_count': dislike_count,
-            'average_rating': float_or_none(video_info.get('avg_rating', [None])[0]),
+            'average_rating': average_rating,
              'formats': formats,
              'is_live': is_live,
              'start_time': start_time,
@@ -2545,9 +2587,9 @@ class YoutubePlaylistIE(YoutubePlaylistBaseInfoExtractor):
  
          search_title = lambda class_name: get_element_by_attribute('class', class_name, webpage)
          title_span = (
-            search_title('playlist-title') or
-            search_title('title long-title') or
-            search_title('title'))
+            search_title('playlist-title')
+            or search_title('title long-title')
+            or search_title('title'))
          title = clean_html(title_span)
  
          return self.playlist_result(url_results, playlist_id, title)