[trilulilu] improve extraction

[youtube-dl] / youtube_dl / extractor / brightcove.py
diff --git a/youtube_dl/extractor/brightcove.py b/youtube_dl/extractor/brightcove.py

index f137ba8c6e9fd73b4e16f91c53d4751d89230281..f5ebae1e68e456c158476a02d9df49feac97e02a 100644 (file)
--- a/youtube_dl/extractor/brightcove.py
+++ b/youtube_dl/extractor/brightcove.py
@@ -11,7 +11,6 @@ from ..compat import (
      compat_str,
      compat_urllib_parse,
      compat_urllib_parse_urlparse,
-    compat_urllib_request,
      compat_urlparse,
      compat_xml_parse_error,
  )
@@ -24,6 +23,7 @@ from ..utils import (
      js_to_json,
      int_or_none,
      parse_iso8601,
+    sanitized_Request,
      unescapeHTML,
      unsmuggle_url,
  )
@@ -250,7 +250,7 @@ class BrightcoveLegacyIE(InfoExtractor):
  
      def _get_video_info(self, video_id, query_str, query, referer=None):
          request_url = self._FEDERATED_URL_TEMPLATE % query_str
-        req = compat_urllib_request.Request(request_url)
+        req = sanitized_Request(request_url)
          linkBase = query.get('linkBaseURL')
          if linkBase is not None:
              referer = linkBase[0]
@@ -356,7 +356,7 @@ class BrightcoveLegacyIE(InfoExtractor):
  class BrightcoveNewIE(InfoExtractor):
      IE_NAME = 'brightcove:new'
      _VALID_URL = r'https?://players\.brightcove\.net/(?P<account_id>\d+)/(?P<player_id>[^/]+)_(?P<embed>[^/]+)/index\.html\?.*videoId=(?P<video_id>\d+)'
-    _TEST = {
+    _TESTS = [{
          'url': 'http://players.brightcove.net/929656772001/e41d32dc-ec74-459e-a845-6c69f7b724ea_default/index.html?videoId=4463358922001',
          'md5': 'c8100925723840d4b0d243f7025703be',
          'info_dict': {
@@ -364,12 +364,30 @@ class BrightcoveNewIE(InfoExtractor):
              'ext': 'mp4',
              'title': 'Meet the man behind Popcorn Time',
              'description': 'md5:eac376a4fe366edc70279bfb681aea16',
+            'duration': 165.768,
              'timestamp': 1441391203,
              'upload_date': '20150904',
-            'duration': 165768,
              'uploader_id': '929656772001',
+            'formats': 'mincount:22',
+        },
+    }, {
+        # with rtmp streams
+        'url': 'http://players.brightcove.net/4036320279001/5d112ed9-283f-485f-a7f9-33f42e8bc042_default/index.html?videoId=4279049078001',
+        'info_dict': {
+            'id': '4279049078001',
+            'ext': 'mp4',
+            'title': 'Titansgrave: Chapter 0',
+            'description': 'Titansgrave: Chapter 0',
+            'duration': 1242.058,
+            'timestamp': 1433556729,
+            'upload_date': '20150606',
+            'uploader_id': '4036320279001',
+            'formats': 'mincount:41',
+        },
+        'params': {
+            'skip_download': True,
          }
-    }
+    }]
  
      @staticmethod
      def _extract_urls(webpage):
@@ -384,10 +402,11 @@ class BrightcoveNewIE(InfoExtractor):
          for _, url in re.findall(
                  r'<iframe[^>]+src=(["\'])((?:https?:)//players\.brightcove\.net/\d+/[^/]+/index\.html.+?)\1', webpage):
              entries.append(url)
+
          # Look for embed_in_page embeds [2]
-        # According to examples from [3] it's unclear whether video id may be optional
-        # and what to do when it is
          for video_id, account_id, player_id, embed in re.findall(
+                # According to examples from [3] it's unclear whether video id
+                # may be optional and what to do when it is
                  r'''(?sx)
                      <video[^>]+
                          data-video-id=["\'](\d+)["\'][^>]*>.*?
@@ -399,6 +418,7 @@ class BrightcoveNewIE(InfoExtractor):
              entries.append(
                  'http://players.brightcove.net/%s/%s_%s/index.html?videoId=%s'
                  % (account_id, player_id, embed, video_id))
+
          return entries
  
      def _real_extract(self, url):
@@ -423,7 +443,7 @@ class BrightcoveNewIE(InfoExtractor):
                  r'policyKey\s*:\s*(["\'])(?P<pk>.+?)\1',
                  webpage, 'policy key', group='pk')
  
-        req = compat_urllib_request.Request(
+        req = sanitized_Request(
              'https://edge.api.brightcove.com/playback/v1/accounts/%s/videos/%s'
              % (account_id, video_id),
              headers={'Accept': 'application/json;pk=%s' % policy_key})