[ign] improve extraction and extract uploader_id

[youtube-dl] / youtube_dl / extractor / youtube.py
diff --git a/youtube_dl/extractor/youtube.py b/youtube_dl/extractor/youtube.py

index 030ec70ca0b89c0c910e051c84e26d8d408f458f..b252e36e1162406dedfcc531d7d038e6bd357348 100644 (file)
--- a/youtube_dl/extractor/youtube.py
+++ b/youtube_dl/extractor/youtube.py
@@ -26,6 +26,7 @@ from ..compat import (
  )
  from ..utils import (
      clean_html,
+    encode_dict,
      ExtractorError,
      float_or_none,
      get_element_by_attribute,
@@ -111,10 +112,7 @@ class YoutubeBaseInfoExtractor(InfoExtractor):
              'hl': 'en_US',
          }
  
-        # Convert to UTF-8 *before* urlencode because Python 2.x's urlencode
-        # chokes on unicode
-        login_form = dict((k.encode('utf-8'), v.encode('utf-8')) for k, v in login_form_strs.items())
-        login_data = compat_urllib_parse.urlencode(login_form).encode('ascii')
+        login_data = compat_urllib_parse.urlencode(encode_dict(login_form_strs)).encode('ascii')
  
          req = compat_urllib_request.Request(self._LOGIN_URL, login_data)
          login_results = self._download_webpage(
@@ -147,8 +145,7 @@ class YoutubeBaseInfoExtractor(InfoExtractor):
                  'TrustDevice': 'on',
              })
  
-            tfa_form = dict((k.encode('utf-8'), v.encode('utf-8')) for k, v in tfa_form_strs.items())
-            tfa_data = compat_urllib_parse.urlencode(tfa_form).encode('ascii')
+            tfa_data = compat_urllib_parse.urlencode(encode_dict(tfa_form_strs)).encode('ascii')
  
              tfa_req = compat_urllib_request.Request(self._TWOFACTOR_URL, tfa_data)
              tfa_results = self._download_webpage(
@@ -1657,12 +1654,15 @@ class YoutubeChannelIE(InfoExtractor):
          channel_page = self._download_webpage(
              url + '?view=57', channel_id,
              'Downloading channel page', fatal=False)
-        channel_playlist_id = self._html_search_meta(
-            'channelId', channel_page, 'channel id', default=None)
-        if not channel_playlist_id:
-            channel_playlist_id = self._search_regex(
-                r'data-channel-external-id="([^"]+)"',
-                channel_page, 'channel id', default=None)
+        if channel_page is False:
+            channel_playlist_id = False
+        else:
+            channel_playlist_id = self._html_search_meta(
+                'channelId', channel_page, 'channel id', default=None)
+            if not channel_playlist_id:
+                channel_playlist_id = self._search_regex(
+                    r'data-channel-external-id="([^"]+)"',
+                    channel_page, 'channel id', default=None)
          if channel_playlist_id and channel_playlist_id.startswith('UC'):
              playlist_id = 'UU' + channel_playlist_id[2:]
              return self.url_result(
@@ -1838,8 +1838,8 @@ class YoutubeShowIE(InfoExtractor):
      _VALID_URL = r'https?://www\.youtube\.com/show/(?P<id>[^?#]*)'
      IE_NAME = 'youtube:show'
      _TESTS = [{
-        'url': 'http://www.youtube.com/show/airdisasters',
-        'playlist_mincount': 3,
+        'url': 'https://www.youtube.com/show/airdisasters',
+        'playlist_mincount': 5,
          'info_dict': {
              'id': 'airdisasters',
              'title': 'Air Disasters',
@@ -1850,7 +1850,7 @@ class YoutubeShowIE(InfoExtractor):
          mobj = re.match(self._VALID_URL, url)
          playlist_id = mobj.group('id')
          webpage = self._download_webpage(
-            url, playlist_id, 'Downloading show webpage')
+            'https://www.youtube.com/show/%s/playlists' % playlist_id, playlist_id, 'Downloading show webpage')
          # There's one playlist for each season of the show
          m_seasons = list(re.finditer(r'href="(/playlist\?list=.*?)"', webpage))
          self.to_screen('%s: Found %s seasons' % (playlist_id, len(m_seasons)))
@@ -1973,6 +1973,7 @@ class YoutubeTruncatedURLIE(InfoExtractor):
              annotation_id=annotation_[^&]+|
              x-yt-cl=[0-9]+|
              hl=[^&]*|
+            t=[0-9]+
          )?
          |
              attribution_link\?a=[^&]+
@@ -1995,6 +1996,9 @@ class YoutubeTruncatedURLIE(InfoExtractor):
      }, {
          'url': 'https://www.youtube.com/watch?hl=en-GB',
          'only_matching': True,
+    }, {
+        'url': 'https://www.youtube.com/watch?t=2372',
+        'only_matching': True,
      }]
  
      def _real_extract(self, url):