Merge remote-tracking branch 'upstream/master' into bliptv

[youtube-dl] / youtube_dl / extractor / bbc.py
diff --git a/youtube_dl/extractor/bbc.py b/youtube_dl/extractor/bbc.py

index b98db95b90e0c22d987c62f85c2771bb4851a7c9..7fb80aa38fc39825fd7338b40fc994ccc0c8a185 100644 (file)
--- a/youtube_dl/extractor/bbc.py
+++ b/youtube_dl/extractor/bbc.py
@@ -2,7 +2,6 @@
  from __future__ import unicode_literals
  
  import re
-import xml.etree.ElementTree
  
  from .common import InfoExtractor
  from ..utils import (
@@ -14,18 +13,22 @@ from ..utils import (
      remove_end,
      unescapeHTML,
  )
-from ..compat import compat_HTTPError
+from ..compat import (
+    compat_etree_fromstring,
+    compat_HTTPError,
+)
  
  
  class BBCCoUkIE(InfoExtractor):
      IE_NAME = 'bbc.co.uk'
      IE_DESC = 'BBC iPlayer'
-    _VALID_URL = r'https?://(?:www\.)?bbc\.co\.uk/(?:(?:(?:programmes|iplayer(?:/[^/]+)?/(?:episode|playlist))/)|music/clips[/#])(?P<id>[\da-z]{8})'
+    _ID_REGEX = r'[pb][\da-z]{7}'
+    _VALID_URL = r'https?://(?:www\.)?bbc\.co\.uk/(?:(?:programmes/(?!articles/)|iplayer(?:/[^/]+)?/(?:episode/|playlist/))|music/clips[/#])(?P<id>%s)' % _ID_REGEX
  
      _MEDIASELECTOR_URLS = [
          # Provides HQ HLS streams with even better quality that pc mediaset but fails
          # with geolocation in some cases when it's even not geo restricted at all (e.g.
-        # http://www.bbc.co.uk/programmes/b06bp7lf)
+        # http://www.bbc.co.uk/programmes/b06bp7lf). Also may fail with selectionunavailable.
          'http://open.live.bbc.co.uk/mediaselector/5/select/version/2.0/mediaset/iptv-all/vpid/%s',
          'http://open.live.bbc.co.uk/mediaselector/5/select/version/2.0/mediaset/pc/vpid/%s',
      ]
@@ -332,7 +335,7 @@ class BBCCoUkIE(InfoExtractor):
                  return self._download_media_selector_url(
                      mediaselector_url % programme_id, programme_id)
              except BBCCoUkIE.MediaSelectionError as e:
-                if e.id in ('notukerror', 'geolocation'):
+                if e.id in ('notukerror', 'geolocation', 'selectionunavailable'):
                      last_exception = e
                      continue
                  self._raise_extractor_error(e)
@@ -343,8 +346,8 @@ class BBCCoUkIE(InfoExtractor):
              media_selection = self._download_xml(
                  url, programme_id, 'Downloading media selection XML')
          except ExtractorError as ee:
-            if isinstance(ee.cause, compat_HTTPError) and ee.cause.code == 403:
-                media_selection = xml.etree.ElementTree.fromstring(ee.cause.read().decode('utf-8'))
+            if isinstance(ee.cause, compat_HTTPError) and ee.cause.code in (403, 404):
+                media_selection = compat_etree_fromstring(ee.cause.read().decode('utf-8'))
              else:
                  raise
          return self._process_media_selector(media_selection, programme_id)
@@ -421,7 +424,7 @@ class BBCCoUkIE(InfoExtractor):
                  continue
              title = playlist.find('./{%s}title' % self._EMP_PLAYLIST_NS).text
              description_el = playlist.find('./{%s}summary' % self._EMP_PLAYLIST_NS)
-            description = description_el.text if description_el else None
+            description = description_el.text if description_el is not None else None
  
              def get_programme_id(item):
                  def get_from_attributes(item):
@@ -463,7 +466,7 @@ class BBCCoUkIE(InfoExtractor):
  
          if not programme_id:
              programme_id = self._search_regex(
-                r'"vpid"\s*:\s*"([\da-z]{8})"', webpage, 'vpid', fatal=False, default=None)
+                r'"vpid"\s*:\s*"(%s)"' % self._ID_REGEX, webpage, 'vpid', fatal=False, default=None)
  
          if programme_id:
              formats, subtitles = self._download_media_selector(programme_id)
@@ -493,6 +496,9 @@ class BBCIE(BBCCoUkIE):
      _VALID_URL = r'https?://(?:www\.)?bbc\.(?:com|co\.uk)/(?:[^/]+/)+(?P<id>[^/#?]+)'
  
      _MEDIASELECTOR_URLS = [
+        # Provides HQ HLS streams but fails with geolocation in some cases when it's
+        # even not geo restricted at all
+        'http://open.live.bbc.co.uk/mediaselector/5/select/version/2.0/mediaset/iptv-all/vpid/%s',
          # Provides more formats, namely direct mp4 links, but fails on some videos with
          # notukerror for non UK (?) users (e.g.
          # http://www.bbc.com/travel/story/20150625-sri-lankas-spicy-secret)
@@ -534,8 +540,9 @@ class BBCIE(BBCCoUkIE):
          'url': 'http://www.bbc.com/news/world-europe-32041533',
          'info_dict': {
              'id': 'p02mprgb',
-            'ext': 'flv',
+            'ext': 'mp4',
              'title': 'Aerial footage showed the site of the crash in the Alps - courtesy BFM TV',
+            'description': 'md5:2868290467291b37feda7863f7a83f54',
              'duration': 47,
              'timestamp': 1427219242,
              'upload_date': '20150324',
@@ -577,7 +584,7 @@ class BBCIE(BBCCoUkIE):
          'url': 'http://www.bbc.com/news/video_and_audio/must_see/33376376',
          'info_dict': {
              'id': 'p02w6qjc',
-            'ext': 'flv',
+            'ext': 'mp4',
              'title': '''Judge Mindy Glazer: "I'm sorry to see you here... I always wondered what happened to you"''',
              'duration': 56,
          },
@@ -604,7 +611,7 @@ class BBCIE(BBCCoUkIE):
          'url': 'http://www.bbc.com/autos/story/20130513-hyundais-rock-star',
          'info_dict': {
              'id': 'p018zqqg',
-            'ext': 'flv',
+            'ext': 'mp4',
              'title': 'Hyundai Santa Fe Sport: Rock star',
              'description': 'md5:b042a26142c4154a6e472933cf20793d',
              'timestamp': 1415867444,
@@ -619,8 +626,9 @@ class BBCIE(BBCCoUkIE):
          'url': 'http://www.bbc.com/sport/0/football/33653409',
          'info_dict': {
              'id': 'p02xycnp',
-            'ext': 'flv',
+            'ext': 'mp4',
              'title': 'Transfers: Cristiano Ronaldo to Man Utd, Arsenal to spend?',
+            'description': 'BBC Sport\'s David Ornstein has the latest transfer gossip, including rumours of a Manchester United return for Cristiano Ronaldo.',
              'duration': 140,
          },
          'params': {
@@ -647,7 +655,7 @@ class BBCIE(BBCCoUkIE):
  
      @classmethod
      def suitable(cls, url):
-        return False if BBCCoUkIE.suitable(url) else super(BBCIE, cls).suitable(url)
+        return False if BBCCoUkIE.suitable(url) or BBCCoUkArticleIE.suitable(url) else super(BBCIE, cls).suitable(url)
  
      def _extract_from_media_meta(self, media_meta, video_id):
          # Direct links to media in media metadata (e.g.
@@ -713,7 +721,7 @@ class BBCIE(BBCCoUkIE):
              timestamp = parse_iso8601(self._search_regex(
                  [r'<meta[^>]+property="article:published_time"[^>]+content="([^"]+)"',
                   r'itemprop="datePublished"[^>]+datetime="([^"]+)"',
-                 r'"datePublished":\s*"([^"]+)',],
+                 r'"datePublished":\s*"([^"]+)'],
                  webpage, 'date', default=None))
  
          entries = []
@@ -773,8 +781,9 @@ class BBCIE(BBCCoUkIE):
  
          # single video story (e.g. http://www.bbc.com/travel/story/20150625-sri-lankas-spicy-secret)
          programme_id = self._search_regex(
-            [r'data-video-player-vpid="([\da-z]{8})"',
-             r'<param[^>]+name="externalIdentifier"[^>]+value="([\da-z]{8})"'],
+            [r'data-video-player-vpid="(%s)"' % self._ID_REGEX,
+             r'<param[^>]+name="externalIdentifier"[^>]+value="(%s)"' % self._ID_REGEX,
+             r'videoId\s*:\s*["\'](%s)["\']' % self._ID_REGEX],
              webpage, 'vpid', default=None)
  
          if programme_id:
@@ -809,7 +818,7 @@ class BBCIE(BBCCoUkIE):
  
          # Multiple video article (e.g.
          # http://www.bbc.co.uk/blogs/adamcurtis/entries/3662a707-0af9-3149-963f-47bea720b460)
-        EMBED_URL = r'https?://(?:www\.)?bbc\.co\.uk/(?:[^/]+/)+[\da-z]{8}(?:\b[^"]+)?'
+        EMBED_URL = r'https?://(?:www\.)?bbc\.co\.uk/(?:[^/]+/)+%s(?:\b[^"]+)?' % self._ID_REGEX
          entries = []
          for match in extract_all(r'new\s+SMP\(({.+?})\)'):
              embed_url = match.get('playerSettings', {}).get('externalEmbedUrl')
@@ -898,3 +907,33 @@ class BBCIE(BBCCoUkIE):
              })
  
          return self.playlist_result(entries, playlist_id, playlist_title, playlist_description)
+
+
+class BBCCoUkArticleIE(InfoExtractor):
+    _VALID_URL = 'http://www.bbc.co.uk/programmes/articles/(?P<id>[a-zA-Z0-9]+)'
+    IE_NAME = 'bbc.co.uk:article'
+    IE_DESC = 'BBC articles'
+
+    _TEST = {
+        'url': 'http://www.bbc.co.uk/programmes/articles/3jNQLTMrPlYGTBn0WV6M2MS/not-your-typical-role-model-ada-lovelace-the-19th-century-programmer',
+        'info_dict': {
+            'id': '3jNQLTMrPlYGTBn0WV6M2MS',
+            'title': 'Calculating Ada: The Countess of Computing - Not your typical role model: Ada Lovelace the 19th century programmer - BBC Four',
+            'description': 'Hannah Fry reveals some of her surprising discoveries about Ada Lovelace during filming.',
+        },
+        'playlist_count': 4,
+        'add_ie': ['BBCCoUk'],
+    }
+
+    def _real_extract(self, url):
+        playlist_id = self._match_id(url)
+
+        webpage = self._download_webpage(url, playlist_id)
+
+        title = self._og_search_title(webpage)
+        description = self._og_search_description(webpage).strip()
+
+        entries = [self.url_result(programme_url) for programme_url in re.findall(
+            r'<div[^>]+typeof="Clip"[^>]+resource="([^"]+)"', webpage)]
+
+        return self.playlist_result(entries, playlist_id, title, description)