[utils] Handle HTMLParseError in extract_attributes (closes #13349)

[youtube-dl] / youtube_dl / utils.py
diff --git a/youtube_dl/utils.py b/youtube_dl/utils.py

index 41bc205446a7594ece9f14dc43b0617808962053..1973bd4836a407d3e66fcc4c3a54d052e958ae19 100644 (file)
--- a/youtube_dl/utils.py
+++ b/youtube_dl/utils.py
@@ -11,6 +11,7 @@ import contextlib
  import ctypes
  import datetime
  import email.utils
+import email.header
  import errno
  import functools
  import gzip
@@ -35,6 +36,7 @@ import xml.etree.ElementTree
  import zlib
  
  from .compat import (
+    compat_HTMLParseError,
      compat_HTMLParser,
      compat_basestring,
      compat_chr,
@@ -408,8 +410,12 @@ def extract_attributes(html_element):
      but the cases in the unit test will work for all of 2.6, 2.7, 3.2-3.5.
      """
      parser = HTMLAttributeParser()
-    parser.feed(html_element)
-    parser.close()
+    try:
+        parser.feed(html_element)
+        parser.close()
+    # Older Python may throw HTMLParseError in case of malformed HTML
+    except compat_HTMLParseError:
+        pass
      return parser.attrs
  
  
@@ -931,14 +937,6 @@ class YoutubeDLHandler(compat_urllib_request.HTTPHandler):
          except zlib.error:
              return zlib.decompress(data)
  
-    @staticmethod
-    def addinfourl_wrapper(stream, headers, url, code):
-        if hasattr(compat_urllib_request.addinfourl, 'getcode'):
-            return compat_urllib_request.addinfourl(stream, headers, url, code)
-        ret = compat_urllib_request.addinfourl(stream, headers, url)
-        ret.code = code
-        return ret
-
      def http_request(self, req):
          # According to RFC 3986, URLs can not contain non-ASCII characters, however this is not
          # always respected by websites, some tend to give out URLs with non percent-encoded
@@ -990,13 +988,13 @@ class YoutubeDLHandler(compat_urllib_request.HTTPHandler):
                      break
                  else:
                      raise original_ioerror
-            resp = self.addinfourl_wrapper(uncompressed, old_resp.headers, old_resp.url, old_resp.code)
+            resp = compat_urllib_request.addinfourl(uncompressed, old_resp.headers, old_resp.url, old_resp.code)
              resp.msg = old_resp.msg
              del resp.headers['Content-encoding']
          # deflate
          if resp.headers.get('Content-encoding', '') == 'deflate':
              gz = io.BytesIO(self.deflate(resp.read()))
-            resp = self.addinfourl_wrapper(gz, old_resp.headers, old_resp.url, old_resp.code)
+            resp = compat_urllib_request.addinfourl(gz, old_resp.headers, old_resp.url, old_resp.code)
              resp.msg = old_resp.msg
              del resp.headers['Content-encoding']
          # Percent-encode redirect URL of Location HTTP header to satisfy RFC 3986 (see
@@ -1186,7 +1184,7 @@ def unified_timestamp(date_str, day_first=True):
      if date_str is None:
          return None
  
-    date_str = date_str.replace(',', ' ')
+    date_str = re.sub(r'[,|]', '', date_str)
  
      pm_delta = 12 if re.search(r'(?i)PM', date_str) else 0
      timezone, date_str = extract_timezone(date_str)
@@ -1194,6 +1192,11 @@ def unified_timestamp(date_str, day_first=True):
      # Remove AM/PM + timezone
      date_str = re.sub(r'(?i)\s*(?:AM|PM)(?:\s+[A-Z]+)?', '', date_str)
  
+    # Remove unrecognized timezones from ISO 8601 alike timestamps
+    m = re.search(r'\d{1,2}:\d{1,2}(?:\.\d+)?(?P<tz>\s*[A-Z]+)$', date_str)
+    if m:
+        date_str = date_str[:-len(m.group('tz'))]
+
      for expression in date_formats(day_first):
          try:
              dt = datetime.datetime.strptime(date_str, expression) - timezone + datetime.timedelta(hours=pm_delta)
@@ -2092,6 +2095,58 @@ def update_Request(req, url=None, data=None, headers={}, query={}):
      return new_req
  
  
+def _multipart_encode_impl(data, boundary):
+    content_type = 'multipart/form-data; boundary=%s' % boundary
+
+    out = b''
+    for k, v in data.items():
+        out += b'--' + boundary.encode('ascii') + b'\r\n'
+        if isinstance(k, compat_str):
+            k = k.encode('utf-8')
+        if isinstance(v, compat_str):
+            v = v.encode('utf-8')
+        # RFC 2047 requires non-ASCII field names to be encoded, while RFC 7578
+        # suggests sending UTF-8 directly. Firefox sends UTF-8, too
+        content = b'Content-Disposition: form-data; name="' + k + b'"\r\n\r\n' + v + b'\r\n'
+        if boundary.encode('ascii') in content:
+            raise ValueError('Boundary overlaps with data')
+        out += content
+
+    out += b'--' + boundary.encode('ascii') + b'--\r\n'
+
+    return out, content_type
+
+
+def multipart_encode(data, boundary=None):
+    '''
+    Encode a dict to RFC 7578-compliant form-data
+
+    data:
+        A dict where keys and values can be either Unicode or bytes-like
+        objects.
+    boundary:
+        If specified a Unicode object, it's used as the boundary. Otherwise
+        a random boundary is generated.
+
+    Reference: https://tools.ietf.org/html/rfc7578
+    '''
+    has_specified_boundary = boundary is not None
+
+    while True:
+        if boundary is None:
+            boundary = '---------------' + str(random.randrange(0x0fffffff, 0xffffffff))
+
+        try:
+            out, content_type = _multipart_encode_impl(data, boundary)
+            break
+        except ValueError:
+            if has_specified_boundary:
+                raise
+            boundary = None
+
+    return out, content_type
+
+
  def dict_get(d, key_or_keys, default=None, skip_false_values=True):
      if isinstance(key_or_keys, (list, tuple)):
          for key in key_or_keys:
@@ -2153,7 +2208,12 @@ def parse_age_limit(s):
  
  def strip_jsonp(code):
      return re.sub(
-        r'(?s)^[a-zA-Z0-9_.$]+\s*\(\s*(.*)\);?\s*?(?://[^\n]*)*$', r'\1', code)
+        r'''(?sx)^
+            (?:window\.)?(?P<func_name>[a-zA-Z0-9_.$]+)
+            (?:\s*&&\s*(?P=func_name))?
+            \s*\(\s*(?P<callback_data>.*)\);?
+            \s*?(?://[^\n]*)*$''',
+        r'\g<callback_data>', code)
  
  
  def js_to_json(code):
@@ -2273,10 +2333,8 @@ def mimetype2ext(mt):
      return {
          '3gpp': '3gp',
          'smptett+xml': 'tt',
-        'srt': 'srt',
          'ttaf+xml': 'dfxp',
          'ttml+xml': 'ttml',
-        'vtt': 'vtt',
          'x-flv': 'flv',
          'x-mp4-fragmented': 'mp4',
          'x-ms-wmv': 'wmv',
@@ -2284,11 +2342,11 @@ def mimetype2ext(mt):
          'x-mpegurl': 'm3u8',
          'vnd.apple.mpegurl': 'm3u8',
          'dash+xml': 'mpd',
-        'f4m': 'f4m',
          'f4m+xml': 'f4m',
          'hds+xml': 'f4m',
          'vnd.ms-sstr+xml': 'ism',
          'quicktime': 'mov',
+        'mp2t': 'ts',
      }.get(res, res)
  
  
@@ -2304,11 +2362,11 @@ def parse_codecs(codecs_str):
          if codec in ('avc1', 'avc2', 'avc3', 'avc4', 'vp9', 'vp8', 'hev1', 'hev2', 'h263', 'h264', 'mp4v'):
              if not vcodec:
                  vcodec = full_codec
-        elif codec in ('mp4a', 'opus', 'vorbis', 'mp3', 'aac', 'ac-3'):
+        elif codec in ('mp4a', 'opus', 'vorbis', 'mp3', 'aac', 'ac-3', 'ec-3', 'eac3', 'dtsc', 'dtse', 'dtsh', 'dtsl'):
              if not acodec:
                  acodec = full_codec
          else:
-            write_string('WARNING: Unknown codec %s' % full_codec, sys.stderr)
+            write_string('WARNING: Unknown codec %s\n' % full_codec, sys.stderr)
      if not vcodec and not acodec:
          if len(splited_codecs) == 2:
              return {
@@ -3757,3 +3815,11 @@ def write_xattr(path, key, value):
                          "Couldn't find a tool to set the xattrs. "
                          "Install either the python 'xattr' module, "
                          "or the 'xattr' binary.")
+
+
+def random_birthday(year_field, month_field, day_field):
+    return {
+        year_field: str(random.randint(1950, 1995)),
+        month_field: str(random.randint(1, 12)),
+        day_field: str(random.randint(1, 31)),
+    }