[YoutubeDL] Fix format selection with filters (Closes #10083)

[youtube-dl] / youtube_dl / YoutubeDL.py
diff --git a/youtube_dl/YoutubeDL.py b/youtube_dl/YoutubeDL.py

index a89a71a250e3c02cb1157bfc0970e308474e4e89..cf9cd82975a6ff5b703cbb9f9d75bdf659b38091 100755 (executable)
--- a/youtube_dl/YoutubeDL.py
+++ b/youtube_dl/YoutubeDL.py
@@ -5,6 +5,7 @@ from __future__ import absolute_import, unicode_literals
  
  import collections
  import contextlib
+import copy
  import datetime
  import errno
  import fileinput
@@ -64,6 +65,7 @@ from .utils import (
      PostProcessingError,
      preferredencoding,
      prepend_extension,
+    register_socks_protocols,
      render_table,
      replace_extension,
      SameFileError,
@@ -195,8 +197,8 @@ class YoutubeDL(object):
      prefer_insecure:   Use HTTP instead of HTTPS to retrieve information.
                         At the moment, this is only supported by YouTube.
      proxy:             URL of the proxy server to use
-    cn_verification_proxy:  URL of the proxy to use for IP address verification
-                       on Chinese sites. (Experimental)
+    geo_verification_proxy:  URL of the proxy to use for IP address verification
+                       on geo-restricted sites. (Experimental)
      socket_timeout:    Time to wait for unresponsive hosts, in seconds
      bidi_workaround:   Work around buggy terminals without bidirectional text
                         support, using fridibi
@@ -260,7 +262,9 @@ class YoutubeDL(object):
      The following options determine which downloader is picked:
      external_downloader: Executable of the external downloader to call.
                         None or unset for standard (built-in) downloader.
-    hls_prefer_native: Use the native HLS downloader instead of ffmpeg/avconv.
+    hls_prefer_native: Use the native HLS downloader instead of ffmpeg/avconv
+                       if True, otherwise use ffmpeg/avconv if False, otherwise
+                       use downloader suggested by extractor if None.
  
      The following parameters are not used by YoutubeDL itself, they are used by
      the downloader (see youtube_dl/downloader/common.py):
@@ -301,6 +305,11 @@ class YoutubeDL(object):
          self.params.update(params)
          self.cache = Cache(self)
  
+        if self.params.get('cn_verification_proxy') is not None:
+            self.report_warning('--cn-verification-proxy is deprecated. Use --geo-verification-proxy instead.')
+            if self.params.get('geo_verification_proxy') is None:
+                self.params['geo_verification_proxy'] = self.params['cn_verification_proxy']
+
          if params.get('bidi_workaround', False):
              try:
                  import pty
@@ -323,7 +332,7 @@ class YoutubeDL(object):
                          ['fribidi', '-c', 'UTF-8'] + width_args, **sp_kwargs)
                  self._output_channel = os.fdopen(master, 'rb')
              except OSError as ose:
-                if ose.errno == 2:
+                if ose.errno == errno.ENOENT:
                      self.report_warning('Could not find fribidi executable, ignoring --bidi-workaround . Make sure that  fribidi  is an executable file in one of the directories in your $PATH.')
                  else:
                      raise
@@ -359,6 +368,8 @@ class YoutubeDL(object):
          for ph in self.params.get('progress_hooks', []):
              self.add_progress_hook(ph)
  
+        register_socks_protocols()
+
      def warn_if_short_id(self, argv):
          # short YouTube ID starting with dash?
          idxs = [
@@ -578,7 +589,7 @@ class YoutubeDL(object):
                  is_id=(k == 'id'))
              template_dict = dict((k, sanitize(k, v))
                                   for k, v in template_dict.items()
-                                 if v is not None)
+                                 if v is not None and not isinstance(v, (list, tuple, dict)))
              template_dict = collections.defaultdict(lambda: 'NA', template_dict)
  
              outtmpl = self.params.get('outtmpl', DEFAULT_OUTTMPL)
@@ -715,6 +726,7 @@ class YoutubeDL(object):
          result_type = ie_result.get('_type', 'video')
  
          if result_type in ('url', 'url_transparent'):
+            ie_result['url'] = sanitize_url(ie_result['url'])
              extract_flat = self.params.get('extract_flat', False)
              if ((extract_flat == 'in_playlist' and 'playlist' in extra_info) or
                      extract_flat is True):
@@ -1040,9 +1052,9 @@ class YoutubeDL(object):
              if isinstance(selector, list):
                  fs = [_build_selector_function(s) for s in selector]
  
-                def selector_function(formats):
+                def selector_function(ctx):
                      for f in fs:
-                        for format in f(formats):
+                        for format in f(ctx):
                              yield format
                  return selector_function
              elif selector.type == GROUP:
@@ -1050,17 +1062,17 @@ class YoutubeDL(object):
              elif selector.type == PICKFIRST:
                  fs = [_build_selector_function(s) for s in selector.selector]
  
-                def selector_function(formats):
+                def selector_function(ctx):
                      for f in fs:
-                        picked_formats = list(f(formats))
+                        picked_formats = list(f(ctx))
                          if picked_formats:
                              return picked_formats
                      return []
              elif selector.type == SINGLE:
                  format_spec = selector.selector
  
-                def selector_function(formats):
-                    formats = list(formats)
+                def selector_function(ctx):
+                    formats = list(ctx['formats'])
                      if not formats:
                          return
                      if format_spec == 'all':
@@ -1073,9 +1085,10 @@ class YoutubeDL(object):
                              if f.get('vcodec') != 'none' and f.get('acodec') != 'none']
                          if audiovideo_formats:
                              yield audiovideo_formats[format_idx]
-                        # for audio only (soundcloud) or video only (imgur) urls, select the best/worst audio format
-                        elif (all(f.get('acodec') != 'none' for f in formats) or
-                              all(f.get('vcodec') != 'none' for f in formats)):
+                        # for extractors with incomplete formats (audio only (soundcloud)
+                        # or video only (imgur)) we will fallback to best/worst
+                        # {video,audio}-only format
+                        elif ctx['incomplete_formats']:
                              yield formats[format_idx]
                      elif format_spec == 'bestaudio':
                          audio_formats = [
@@ -1149,17 +1162,18 @@ class YoutubeDL(object):
                      }
                  video_selector, audio_selector = map(_build_selector_function, selector.selector)
  
-                def selector_function(formats):
-                    formats = list(formats)
-                    for pair in itertools.product(video_selector(formats), audio_selector(formats)):
+                def selector_function(ctx):
+                    for pair in itertools.product(
+                            video_selector(copy.deepcopy(ctx)), audio_selector(copy.deepcopy(ctx))):
                          yield _merge(pair)
  
              filters = [self._build_format_filter(f) for f in selector.filters]
  
-            def final_selector(formats):
+            def final_selector(ctx):
+                ctx_copy = copy.deepcopy(ctx)
                  for _filter in filters:
-                    formats = list(filter(_filter, formats))
-                return selector_function(formats)
+                    ctx_copy['formats'] = list(filter(_filter, ctx_copy['formats']))
+                return selector_function(ctx_copy)
              return final_selector
  
          stream = io.BytesIO(format_spec.encode('utf-8'))
@@ -1217,6 +1231,10 @@ class YoutubeDL(object):
          if 'title' not in info_dict:
              raise ExtractorError('Missing "title" field in extractor result')
  
+        if not isinstance(info_dict['id'], compat_str):
+            self.report_warning('"id" field is not a string - forcing string conversion')
+            info_dict['id'] = compat_str(info_dict['id'])
+
          if 'playlist' not in info_dict:
              # It isn't part of a playlist
              info_dict['playlist'] = None
@@ -1362,7 +1380,35 @@ class YoutubeDL(object):
              req_format_list.append('best')
              req_format = '/'.join(req_format_list)
          format_selector = self.build_format_selector(req_format)
-        formats_to_download = list(format_selector(formats))
+
+        # While in format selection we may need to have an access to the original
+        # format set in order to calculate some metrics or do some processing.
+        # For now we need to be able to guess whether original formats provided
+        # by extractor are incomplete or not (i.e. whether extractor provides only
+        # video-only or audio-only formats) for proper formats selection for
+        # extractors with such incomplete formats (see
+        # https://github.com/rg3/youtube-dl/pull/5556).
+        # Since formats may be filtered during format selection and may not match
+        # the original formats the results may be incorrect. Thus original formats
+        # or pre-calculated metrics should be passed to format selection routines
+        # as well.
+        # We will pass a context object containing all necessary additional data
+        # instead of just formats.
+        # This fixes incorrect format selection issue (see
+        # https://github.com/rg3/youtube-dl/issues/10083).
+        incomplete_formats = all(
+            # All formats are video-only or
+            f.get('vcodec') != 'none' and f.get('acodec') == 'none' or
+            # all formats are audio-only
+            f.get('vcodec') == 'none' and f.get('acodec') != 'none'
+            for f in formats)
+
+        ctx = {
+            'formats': formats,
+            'incomplete_formats': incomplete_formats,
+        }
+
+        formats_to_download = list(format_selector(ctx))
          if not formats_to_download:
              raise ExtractorError('requested format not available',
                                   expected=True)
@@ -1637,7 +1683,7 @@ class YoutubeDL(object):
                      # Just a single file
                      success = dl(filename, info_dict)
              except (compat_urllib_error.URLError, compat_http_client.HTTPException, socket.error) as err:
-                self.report_error('unable to download video data: %s' % str(err))
+                self.report_error('unable to download video data: %s' % error_to_compat_str(err))
                  return
              except (OSError, IOError) as err:
                  raise UnavailableVideoError(err)
@@ -2016,6 +2062,7 @@ class YoutubeDL(object):
          if opts_cookiefile is None:
              self.cookiejar = compat_cookiejar.CookieJar()
          else:
+            opts_cookiefile = compat_expanduser(opts_cookiefile)
              self.cookiejar = compat_cookiejar.MozillaCookieJar(
                  opts_cookiefile)
              if os.access(opts_cookiefile, os.R_OK):