[extractor/common] Properly extract audio only formats in master m3u8 playlists

[youtube-dl] / youtube_dl / extractor / common.py
diff --git a/youtube_dl/extractor/common.py b/youtube_dl/extractor/common.py

index 6b40a8323463c72d1a223d427ade2a978b763d72..51351fb57c95cc18637ed6c126d4ff894680784e 100644 (file)
--- a/youtube_dl/extractor/common.py
+++ b/youtube_dl/extractor/common.py
@@ -46,6 +46,7 @@ from ..utils import (
      xpath_with_ns,
      determine_protocol,
      parse_duration,
+    mimetype2ext,
  )
  
  
@@ -636,7 +637,7 @@ class InfoExtractor(object):
          downloader_params = self._downloader.params
  
          # Attempt to use provided username and password or .netrc data
-        if downloader_params.get('username', None) is not None:
+        if downloader_params.get('username') is not None:
              username = downloader_params['username']
              password = downloader_params['password']
          elif downloader_params.get('usenetrc', False):
@@ -663,7 +664,7 @@ class InfoExtractor(object):
              return None
          downloader_params = self._downloader.params
  
-        if downloader_params.get('twofactor', None) is not None:
+        if downloader_params.get('twofactor') is not None:
              return downloader_params['twofactor']
  
          return compat_getpass('Type %s and press [Return]: ' % note)
@@ -744,7 +745,7 @@ class InfoExtractor(object):
              'mature': 17,
              'restricted': 19,
          }
-        return RATING_TABLE.get(rating.lower(), None)
+        return RATING_TABLE.get(rating.lower())
  
      def _family_friendly_search(self, html):
          # See http://schema.org/VideoObject
@@ -759,7 +760,7 @@ class InfoExtractor(object):
              '0': 18,
              'false': 18,
          }
-        return RATING_TABLE.get(family_friendly.lower(), None)
+        return RATING_TABLE.get(family_friendly.lower())
  
      def _twitter_search_player(self, html):
          return self._html_search_meta('twitter:player', html,
@@ -899,6 +900,16 @@ class InfoExtractor(object):
                      item='%s video format' % f.get('format_id') if f.get('format_id') else 'video'),
                  formats)
  
+    @staticmethod
+    def _remove_duplicate_formats(formats):
+        format_urls = set()
+        unique_formats = []
+        for f in formats:
+            if f['url'] not in format_urls:
+                format_urls.add(f['url'])
+                unique_formats.append(f)
+        formats[:] = unique_formats
+
      def _is_valid_url(self, url, video_id, item='video'):
          url = self._proto_relative_url(url, scheme='http:')
          # For now assume non HTTP(S) URLs always valid
@@ -1073,19 +1084,29 @@ class InfoExtractor(object):
                      'protocol': entry_protocol,
                      'preference': preference,
                  }
-                codecs = last_info.get('CODECS')
-                if codecs:
-                    # TODO: looks like video codec is not always necessarily goes first
-                    va_codecs = codecs.split(',')
-                    if va_codecs[0]:
-                        f['vcodec'] = va_codecs[0]
-                    if len(va_codecs) > 1 and va_codecs[1]:
-                        f['acodec'] = va_codecs[1]
                  resolution = last_info.get('RESOLUTION')
                  if resolution:
                      width_str, height_str = resolution.split('x')
                      f['width'] = int(width_str)
                      f['height'] = int(height_str)
+                codecs = last_info.get('CODECS')
+                if codecs:
+                    vcodec, acodec = [None] * 2
+                    va_codecs = codecs.split(',')
+                    if len(va_codecs) == 1:
+                        # Audio only entries usually come with single codec and
+                        # no resolution. For more robustness we also check it to
+                        # be mp4 audio.
+                        if not resolution and va_codecs[0].startswith('mp4a'):
+                            vcodec, acodec = 'none', va_codecs[0]
+                        else:
+                            vcodec = va_codecs[0]
+                    else:
+                        vcodec, acodec = va_codecs[:2]
+                    f.update({
+                        'acodec': acodec,
+                        'vcodec': vcodec,
+                    })
                  if last_media is not None:
                      f['m3u8_media'] = last_media
                      last_media = None
@@ -1224,6 +1245,7 @@ class InfoExtractor(object):
                  continue
  
              src_url = src if src.startswith('http') else compat_urlparse.urljoin(base, src)
+            src_url = src_url.strip()
  
              if proto == 'm3u8' or src_ext == 'm3u8':
                  m3u8_formats = self._extract_m3u8_formats(
@@ -1276,16 +1298,7 @@ class InfoExtractor(object):
              if not src or src in urls:
                  continue
              urls.append(src)
-            ext = textstream.get('ext') or determine_ext(src)
-            if not ext:
-                type_ = textstream.get('type')
-                SUBTITLES_TYPES = {
-                    'text/vtt': 'vtt',
-                    'text/srt': 'srt',
-                    'application/smptett+xml': 'tt',
-                }
-                if type_ in SUBTITLES_TYPES:
-                    ext = SUBTITLES_TYPES[type_]
+            ext = textstream.get('ext') or determine_ext(src) or mimetype2ext(textstream.get('type'))
              lang = textstream.get('systemLanguage') or textstream.get('systemLanguageName') or textstream.get('lang') or subtitles_lang
              subtitles.setdefault(lang, []).append({
                  'url': src,
@@ -1434,7 +1447,9 @@ class InfoExtractor(object):
                                  base_url = base_url_e.text + base_url
                                  if re.match(r'^https?://', base_url):
                                      break
-                        if not re.match(r'^https?://', base_url):
+                        if mpd_base_url and not re.match(r'^https?://', base_url):
+                            if not mpd_base_url.endswith('/') and not base_url.startswith('/'):
+                                mpd_base_url += '/'
                              base_url = mpd_base_url + base_url
                          representation_id = representation_attrib.get('id')
                          lang = representation_attrib.get('lang')
@@ -1494,7 +1509,7 @@ class InfoExtractor(object):
      def _live_title(self, name):
          """ Generate the title for a live video """
          now = datetime.datetime.now()
-        now_str = now.strftime("%Y-%m-%d %H:%M")
+        now_str = now.strftime('%Y-%m-%d %H:%M')
          return name + ' ' + now_str
  
      def _int(self, v, name, fatal=False, **kwargs):
@@ -1567,7 +1582,7 @@ class InfoExtractor(object):
          return {}
  
      def _get_subtitles(self, *args, **kwargs):
-        raise NotImplementedError("This method must be implemented by subclasses")
+        raise NotImplementedError('This method must be implemented by subclasses')
  
      @staticmethod
      def _merge_subtitle_items(subtitle_list1, subtitle_list2):
@@ -1593,7 +1608,7 @@ class InfoExtractor(object):
          return {}
  
      def _get_automatic_captions(self, *args, **kwargs):
-        raise NotImplementedError("This method must be implemented by subclasses")
+        raise NotImplementedError('This method must be implemented by subclasses')
  
  
  class SearchInfoExtractor(InfoExtractor):
@@ -1633,7 +1648,7 @@ class SearchInfoExtractor(InfoExtractor):
  
      def _get_n_results(self, query, n):
          """Get a specified number of results for a query"""
-        raise NotImplementedError("This method must be implemented by subclasses")
+        raise NotImplementedError('This method must be implemented by subclasses')
  
      @property
      def SEARCH_KEY(self):