[extractor/common] extract youtube dash formats filesize(fixes #8480)

[youtube-dl] / youtube_dl / extractor / common.py
diff --git a/youtube_dl/extractor/common.py b/youtube_dl/extractor/common.py

index 1758ab29b89e01f2077b617df7f04119a7197072..a9497851cd5ac122c9ba5488e70b512d520d0a3a 100644 (file)
--- a/youtube_dl/extractor/common.py
+++ b/youtube_dl/extractor/common.py
@@ -1186,6 +1186,7 @@ class InfoExtractor(object):
          http_count = 0
          m3u8_count = 0
  
+        src_urls = []
          videos = smil.findall(self._xpath_ns('.//video', namespace))
          for video in videos:
              src = video.get('src')
@@ -1222,6 +1223,9 @@ class InfoExtractor(object):
                  continue
  
              src_url = src if src.startswith('http') else compat_urlparse.urljoin(base, src)
+            if src_url in src_urls:
+                continue
+            src_urls.append(src_url)
  
              if proto == 'm3u8' or src_ext == 'm3u8':
                  m3u8_formats = self._extract_m3u8_formats(
@@ -1267,11 +1271,13 @@ class InfoExtractor(object):
          return formats
  
      def _parse_smil_subtitles(self, smil, namespace=None, subtitles_lang='en'):
+        urls = []
          subtitles = {}
          for num, textstream in enumerate(smil.findall(self._xpath_ns('.//textstream', namespace))):
              src = textstream.get('src')
-            if not src:
+            if not src or src in urls:
                  continue
+            urls.append(src)
              ext = textstream.get('ext') or determine_ext(src)
              if not ext:
                  type_ = textstream.get('type')
@@ -1343,32 +1349,40 @@ class InfoExtractor(object):
          mpd, urlh = res
          mpd_base_url = re.match(r'https?://.+/', urlh.geturl()).group()
  
-        return self._parse_mpd(
+        return self._parse_mpd_formats(
              compat_etree_fromstring(mpd.encode('utf-8')), mpd_id, mpd_base_url, formats_dict=formats_dict)
  
-    def _parse_mpd(self, mpd_doc, mpd_id=None, mpd_base_url='', formats_dict={}):
+    def _parse_mpd_formats(self, mpd_doc, mpd_id=None, mpd_base_url='', formats_dict={}):
          if mpd_doc.get('type') == 'dynamic':
              return []
  
+        namespace = self._search_regex(r'(?i)^{([^}]+)?}MPD$', mpd_doc.tag, 'namespace', default=None)
+
+        def _add_ns(path):
+            return self._xpath_ns(path, namespace)
+
+        def is_drm_protected(element):
+            return element.find(_add_ns('ContentProtection')) is not None
+
          def extract_multisegment_info(element, ms_parent_info):
              ms_info = ms_parent_info.copy()
-            segment_list = element.find(self._xpath_ns('SegmentList', namespace))
+            segment_list = element.find(_add_ns('SegmentList'))
              if segment_list is not None:
-                segment_urls_e = segment_list.findall(self._xpath_ns('SegmentURL', namespace))
+                segment_urls_e = segment_list.findall(_add_ns('SegmentURL'))
                  if segment_urls_e:
                      ms_info['segment_urls'] = [segment.attrib['media'] for segment in segment_urls_e]
-                initialization = segment_list.find(self._xpath_ns('Initialization', namespace))
+                initialization = segment_list.find(_add_ns('Initialization'))
                  if initialization is not None:
                      ms_info['initialization_url'] = initialization.attrib['sourceURL']
              else:
-                segment_template = element.find(self._xpath_ns('SegmentTemplate', namespace))
+                segment_template = element.find(_add_ns('SegmentTemplate'))
                  if segment_template is not None:
                      start_number = segment_template.get('startNumber')
                      if start_number:
                          ms_info['start_number'] = int(start_number)
-                    segment_timeline = segment_template.find(self._xpath_ns('SegmentTimeline', namespace))
+                    segment_timeline = segment_template.find(_add_ns('SegmentTimeline'))
                      if segment_timeline is not None:
-                        s_e = segment_timeline.findall(self._xpath_ns('S', namespace))
+                        s_e = segment_timeline.findall(_add_ns('S'))
                          if s_e:
                              ms_info['total_number'] = 0
                              for s in s_e:
@@ -1387,23 +1401,26 @@ class InfoExtractor(object):
                      if initialization:
                          ms_info['initialization_url'] = initialization
                      else:
-                        initialization = segment_template.find(self._xpath_ns('Initialization', namespace))
+                        initialization = segment_template.find(_add_ns('Initialization'))
                          if initialization is not None:
                              ms_info['initialization_url'] = initialization.attrib['sourceURL']
              return ms_info
  
-        namespace = self._search_regex(r'(?i)^{([^}]+)?}MPD$', mpd_doc.tag, 'namespace')
          mpd_duration = parse_duration(mpd_doc.get('mediaPresentationDuration'))
          formats = []
-        for period in mpd_doc.findall(self._xpath_ns('Period', namespace)):
+        for period in mpd_doc.findall(_add_ns('Period')):
              period_duration = parse_duration(period.get('duration')) or mpd_duration
              period_ms_info = extract_multisegment_info(period, {
                  'start_number': 1,
                  'timescale': 1,
              })
-            for adaptation_set in period.findall(self._xpath_ns('AdaptationSet', namespace)):
+            for adaptation_set in period.findall(_add_ns('AdaptationSet')):
+                if is_drm_protected(adaptation_set):
+                    continue
                  adaption_set_ms_info = extract_multisegment_info(adaptation_set, period_ms_info)
-                for representation in adaptation_set.findall(self._xpath_ns('Representation', namespace)):
+                for representation in adaptation_set.findall(_add_ns('Representation')):
+                    if is_drm_protected(representation):
+                        continue
                      representation_attrib = adaptation_set.attrib.copy()
                      representation_attrib.update(representation.attrib)
                      mime_type = representation_attrib.get('mimeType')
@@ -1414,7 +1431,7 @@ class InfoExtractor(object):
                      elif content_type == 'video' or content_type == 'audio':
                          base_url = ''
                          for element in (representation, adaptation_set, period, mpd_doc):
-                            base_url_e = element.find(self._xpath_ns('BaseURL', namespace))
+                            base_url_e = element.find(_add_ns('BaseURL'))
                              if base_url_e is not None:
                                  base_url = base_url_e.text + base_url
                                  if re.match(r'^https?://', base_url):
@@ -1422,6 +1439,9 @@ class InfoExtractor(object):
                          if not re.match(r'^https?://', base_url):
                              base_url = mpd_base_url + base_url
                          representation_id = representation_attrib.get('id')
+                        lang = representation_attrib.get('lang')
+                        url_el = representation.find(_add_ns('BaseURL'))
+                        filesize = int_or_none(url_el.attrib.get('{http://youtube.com/yt/2012/10/10}contentLength') if url_el is not None else None)
                          f = {
                              'format_id': mpd_id or representation_id,
                              'url': base_url,
@@ -1432,21 +1452,20 @@ class InfoExtractor(object):
                              'fps': int_or_none(representation_attrib.get('frameRate')),
                              'vcodec': 'none' if content_type == 'audio' else representation_attrib.get('codecs'),
                              'acodec': 'none' if content_type == 'video' else representation_attrib.get('codecs'),
-                            'language': representation_attrib.get('lang'),
+                            'language': lang if lang not in ('mul', 'und', 'zxx', 'mis') else None,
                              'format_note': 'DASH %s' % content_type,
+                            'filesize': filesize,
                          }
                          representation_ms_info = extract_multisegment_info(representation, adaption_set_ms_info)
                          if 'segment_urls' not in representation_ms_info and 'media_template' in representation_ms_info:
                              if 'total_number' not in representation_ms_info and 'segment_duration':
-                                segment_duration = representation_ms_info['segment_duration'] / representation_ms_info['timescale']
-                                representation_ms_info['total_number'] = int(math.ceil(period_duration / segment_duration))
+                                segment_duration = float(representation_ms_info['segment_duration']) / float(representation_ms_info['timescale'])
+                                representation_ms_info['total_number'] = int(math.ceil(float(period_duration) / segment_duration))
                              media_template = representation_ms_info['media_template']
                              media_template = media_template.replace('$RepresentationID$', representation_id)
-                            media_template = re.sub(r'\$(Bandwidth)(?:%(0\d+d))?\$', r'%(\1)\2', media_template)
-                            media_template = media_template % {'Bandwidth': representation_attrib.get('bandwidth')}
-                            media_template = re.sub(r'\$(Number)(?:%(0\d+d))?\$', r'%(\1)\2', media_template)
+                            media_template = re.sub(r'\$(Number|Bandwidth)(?:%(0\d+)d)?\$', r'%(\1)\2d', media_template)
                              media_template.replace('$$', '$')
-                            representation_ms_info['segment_urls'] = [media_template % {'Number': segment_number} for segment_number in range(representation_ms_info['start_number'], representation_ms_info['total_number'] + representation_ms_info['start_number'])]
+                            representation_ms_info['segment_urls'] = [media_template % {'Number': segment_number, 'Bandwidth': representation_attrib.get('bandwidth')} for segment_number in range(representation_ms_info['start_number'], representation_ms_info['total_number'] + representation_ms_info['start_number'])]
                          if 'segment_urls' in representation_ms_info:
                              f.update({
                                  'segment_urls': representation_ms_info['segment_urls'],
@@ -1471,6 +1490,7 @@ class InfoExtractor(object):
                              existing_format.update(f)
                      else:
                          self.report_warning('Unknown MIME type %s in DASH manifest' % mime_type)
+        self._sort_formats(formats)
          return formats
  
      def _live_title(self, name):