Merge remote-tracking branch 'daohoangson/zing-mp3'

[youtube-dl] / youtube_dl / extractor / ted.py
diff --git a/youtube_dl/extractor/ted.py b/youtube_dl/extractor/ted.py

index a8d8e8b29668f5db59f27565c21ee9064190498b..8550380779168a80b95e526f8921059e2eddf8f4 100644 (file)
--- a/youtube_dl/extractor/ted.py
+++ b/youtube_dl/extractor/ted.py
@@ -27,7 +27,7 @@ class TEDIE(SubtitlesInfoExtractor):
          '''
      _TESTS = [{
          'url': 'http://www.ted.com/talks/dan_dennett_on_our_consciousness.html',
-        'md5': '4ea1dada91e4174b53dac2bb8ace429d',
+        'md5': 'fc94ac279feebbce69f21c0c6ee82810',
          'info_dict': {
              'id': '102',
              'ext': 'mp4',
@@ -37,6 +37,8 @@ class TEDIE(SubtitlesInfoExtractor):
                  'consciousness, but that half the time our brains are '
                  'actively fooling us.'),
              'uploader': 'Dan Dennett',
+            'width': 854,
+            'duration': 1308,
          }
      }, {
          'url': 'http://www.ted.com/watch/ted-institute/ted-bcg/vishal-sikka-the-beauty-and-power-of-algorithms',
@@ -48,12 +50,45 @@ class TEDIE(SubtitlesInfoExtractor):
              'thumbnail': 're:^https?://.+\.jpg',
              'description': 'Adaptive, intelligent, and consistent, algorithms are emerging as the ultimate app for everything from matching consumers to products to assessing medical diagnoses. Vishal Sikka shares his appreciation for the algorithm, charting both its inherent beauty and its growing power.',
          }
+    }, {
+        'url': 'http://www.ted.com/talks/gabby_giffords_and_mark_kelly_be_passionate_be_courageous_be_your_best',
+        'info_dict': {
+            'id': '1972',
+            'ext': 'mp4',
+            'title': 'Be passionate. Be courageous. Be your best.',
+            'uploader': 'Gabby Giffords and Mark Kelly',
+            'description': 'md5:5174aed4d0f16021b704120360f72b92',
+            'duration': 1128,
+        },
+    }, {
+        'url': 'http://www.ted.com/playlists/who_are_the_hackers',
+        'info_dict': {
+            'id': '10',
+            'title': 'Who are the hackers?',
+        },
+        'playlist_mincount': 6,
+    }, {
+        # contains a youtube video
+        'url': 'https://www.ted.com/talks/douglas_adams_parrots_the_universe_and_everything',
+        'add_ie': ['Youtube'],
+        'info_dict': {
+            'id': '_ZG8HBuDjgc',
+            'ext': 'mp4',
+            'title': 'Douglas Adams: Parrots the Universe and Everything',
+            'description': 'md5:01ad1e199c49ac640cb1196c0e9016af',
+            'uploader': 'University of California Television (UCTV)',
+            'uploader_id': 'UCtelevision',
+            'upload_date': '20080522',
+        },
+        'params': {
+            'skip_download': True,
+        },
      }]
  
-    _FORMATS_PREFERENCE = {
-        'low': 1,
-        'medium': 2,
-        'high': 3,
+    _NATIVE_FORMATS = {
+        'low': {'preference': 1, 'width': 320, 'height': 180},
+        'medium': {'preference': 2, 'width': 512, 'height': 288},
+        'high': {'preference': 3, 'width': 854, 'height': 480},
      }
  
      def _extract_info(self, webpage):
@@ -83,7 +118,7 @@ class TEDIE(SubtitlesInfoExtractor):
          playlist_info = info['playlist']
  
          playlist_entries = [
-            self.url_result(u'http://www.ted.com/talks/' + talk['slug'], self.ie_key())
+            self.url_result('http://www.ted.com/talks/' + talk['slug'], self.ie_key())
              for talk in info['talks']
          ]
          return self.playlist_result(
@@ -97,13 +132,34 @@ class TEDIE(SubtitlesInfoExtractor):
  
          talk_info = self._extract_info(webpage)['talks'][0]
  
+        if talk_info.get('external') is not None:
+            self.to_screen('Found video from %s' % talk_info['external']['service'])
+            return {
+                '_type': 'url',
+                'url': talk_info['external']['uri'],
+            }
+
          formats = [{
-            'ext': 'mp4',
              'url': format_url,
              'format_id': format_id,
              'format': format_id,
-            'preference': self._FORMATS_PREFERENCE.get(format_id, -1),
-        } for (format_id, format_url) in talk_info['nativeDownloads'].items()]
+        } for (format_id, format_url) in talk_info['nativeDownloads'].items() if format_url is not None]
+        if formats:
+            for f in formats:
+                finfo = self._NATIVE_FORMATS.get(f['format_id'])
+                if finfo:
+                    f.update(finfo)
+        else:
+            # Use rtmp downloads
+            formats = [{
+                'format_id': f['name'],
+                'url': talk_info['streamer'],
+                'play_path': f['file'],
+                'ext': 'flv',
+                'width': f['width'],
+                'height': f['height'],
+                'tbr': f['bitrate'],
+            } for f in talk_info['resources']['rtmp']]
          self._sort_formats(formats)
  
          video_id = compat_str(talk_info['id'])
@@ -118,12 +174,13 @@ class TEDIE(SubtitlesInfoExtractor):
              thumbnail = 'http://' + thumbnail
          return {
              'id': video_id,
-            'title': talk_info['title'],
+            'title': talk_info['title'].strip(),
              'uploader': talk_info['speaker'],
              'thumbnail': thumbnail,
              'description': self._og_search_description(webpage),
              'subtitles': video_subtitles,
              'formats': formats,
+            'duration': talk_info.get('duration'),
          }
  
      def _get_available_subtitles(self, video_id, talk_info):
@@ -135,7 +192,7 @@ class TEDIE(SubtitlesInfoExtractor):
                  sub_lang_list[l] = url
              return sub_lang_list
          else:
-            self._downloader.report_warning(u'video doesn\'t have subtitles')
+            self._downloader.report_warning('video doesn\'t have subtitles')
              return {}
  
      def _watch_info(self, url, name):
@@ -150,7 +207,10 @@ class TEDIE(SubtitlesInfoExtractor):
          title = self._html_search_regex(
              r"(?s)<h1(?:\s+class='[^']+')?>(.+?)</h1>", webpage, 'title')
          description = self._html_search_regex(
-            r'(?s)<h4 class="[^"]+" id="h3--about-this-talk">.*?</h4>(.*?)</div>',
+            [
+                r'(?s)<h4 class="[^"]+" id="h3--about-this-talk">.*?</h4>(.*?)</div>',
+                r'(?s)<p><strong>About this talk:</strong>\s+(.*?)</p>',
+            ],
              webpage, 'description', fatal=False)
  
          return {