[README.md] Add format_id to the list of string meta fields available for use in...

[youtube-dl] / youtube_dl / utils.py
diff --git a/youtube_dl/utils.py b/youtube_dl/utils.py

index ec186918cd8672ada2da2d5521e0ba8b22eb273d..6d27b80c02912b01b7f64ef85f20097f8216abe4 100644 (file)
--- a/youtube_dl/utils.py
+++ b/youtube_dl/utils.py
@@ -47,9 +47,11 @@ from .compat import (
      compat_str,
      compat_urllib_error,
      compat_urllib_parse,
+    compat_urllib_parse_urlencode,
      compat_urllib_parse_urlparse,
      compat_urllib_request,
      compat_urlparse,
+    compat_xpath,
      shlex_quote,
  )
  
@@ -165,12 +167,7 @@ if sys.version_info >= (2, 7):
          return node.find(expr)
  else:
      def find_xpath_attr(node, xpath, key, val=None):
-        # Here comes the crazy part: In 2.6, if the xpath is a unicode,
-        # .//node does not match if a node is a direct child of . !
-        if isinstance(xpath, compat_str):
-            xpath = xpath.encode('ascii')
-
-        for f in node.findall(xpath):
+        for f in node.findall(compat_xpath(xpath)):
              if key not in f.attrib:
                  continue
              if val is None or f.attrib.get(key) == val:
@@ -195,9 +192,7 @@ def xpath_with_ns(path, ns_map):
  
  def xpath_element(node, xpath, name=None, fatal=False, default=NO_DEFAULT):
      def _find_xpath(xpath):
-        if sys.version_info < (2, 7):  # Crazy 2.6
-            xpath = xpath.encode('ascii')
-        return node.find(xpath)
+        return node.find(compat_xpath(xpath))
  
      if isinstance(xpath, (str, compat_str)):
          n = _find_xpath(xpath)
@@ -273,15 +268,17 @@ def get_element_by_attribute(attribute, value, html):
  
      return unescapeHTML(res)
  
+
  class HTMLAttributeParser(compat_HTMLParser):
      """Trivial HTML parser to gather the attributes for a single element"""
      def __init__(self):
-        self.attrs = { }
+        self.attrs = {}
          compat_HTMLParser.__init__(self)
  
      def handle_starttag(self, tag, attrs):
          self.attrs = dict(attrs)
  
+
  def extract_attributes(html_element):
      """Given a string for an HTML element such as
      <el
@@ -303,6 +300,7 @@ def extract_attributes(html_element):
      parser.close()
      return parser.attrs
  
+
  def clean_html(html):
      """Clean an HTML snippet into a readable string"""
  
@@ -419,9 +417,12 @@ def sanitize_path(s):
  
  # Prepend protocol-less URLs with `http:` scheme in order to mitigate the number of
  # unwanted failures due to missing protocol
+def sanitize_url(url):
+    return 'http:%s' % url if url.startswith('//') else url
+
+
  def sanitized_Request(url, *args, **kwargs):
-    return compat_urllib_request.Request(
-        'http:%s' % url if url.startswith('//') else url, *args, **kwargs)
+    return compat_urllib_request.Request(sanitize_url(url), *args, **kwargs)
  
  
  def orderedSet(iterable):
@@ -1318,7 +1319,7 @@ def shell_quote(args):
  def smuggle_url(url, data):
      """ Pass additional data in a URL for internal use. """
  
-    sdata = compat_urllib_parse.urlencode(
+    sdata = compat_urllib_parse_urlencode(
          {'__youtubedl_smuggle': json.dumps(data)})
      return url + '#' + sdata
  
@@ -1349,7 +1350,7 @@ def format_bytes(bytes):
  def lookup_unit_table(unit_table, s):
      units_re = '|'.join(re.escape(u) for u in unit_table)
      m = re.match(
-        r'(?P<num>[0-9]+(?:[,.][0-9]*)?)\s*(?P<unit>%s)' % units_re, s)
+        r'(?P<num>[0-9]+(?:[,.][0-9]*)?)\s*(?P<unit>%s)\b' % units_re, s)
      if not m:
          return None
      num_str = m.group('num').replace(',', '.')
@@ -1749,6 +1750,7 @@ def escape_url(url):
      """Escape URL as suggested by RFC 3986"""
      url_parsed = compat_urllib_parse_urlparse(url)
      return url_parsed._replace(
+        netloc=url_parsed.netloc.encode('idna').decode('ascii'),
          path=escape_rfc3986(url_parsed.path),
          params=escape_rfc3986(url_parsed.params),
          query=escape_rfc3986(url_parsed.query),
@@ -1758,7 +1760,8 @@ def escape_url(url):
  try:
      struct.pack('!I', 0)
  except TypeError:
-    # In Python 2.6 (and some 2.7 versions), struct requires a bytes argument
+    # In Python 2.6 and 2.7.x < 2.7.7, struct requires a bytes argument
+    # See https://bugs.python.org/issue19099
      def struct_pack(spec, *args):
          if isinstance(spec, compat_str):
              spec = spec.encode('ascii')
@@ -1790,22 +1793,15 @@ def read_batch_urls(batch_fd):
  
  
  def urlencode_postdata(*args, **kargs):
-    return compat_urllib_parse.urlencode(*args, **kargs).encode('ascii')
+    return compat_urllib_parse_urlencode(*args, **kargs).encode('ascii')
  
  
  def update_url_query(url, query):
      parsed_url = compat_urlparse.urlparse(url)
      qs = compat_parse_qs(parsed_url.query)
      qs.update(query)
-    qs = encode_dict(qs)
      return compat_urlparse.urlunparse(parsed_url._replace(
-        query=compat_urllib_parse.urlencode(qs, True)))
-
-
-def encode_dict(d, encoding='utf-8'):
-    def encode(v):
-        return v.encode(encoding) if isinstance(v, compat_basestring) else v
-    return dict((encode(k), encode(v)) for k, v in d.items())
+        query=compat_urllib_parse_urlencode(qs, True)))
  
  
  def dict_get(d, key_or_keys, default=None, skip_false_values=True):