Merge pull request #736 from rg3/retry

[youtube-dl] / youtube_dl / utils.py
diff --git a/youtube_dl/utils.py b/youtube_dl/utils.py

index 7d6041929538ff0e77c08d7e73e191b076ee4eb4..017f06c42e9a019e18e25480c5e5d8d3aaaef335 100644 (file)
--- a/youtube_dl/utils.py
+++ b/youtube_dl/utils.py
@@ -8,6 +8,7 @@ import locale
  import os
  import re
  import sys
+import traceback
  import zlib
  import email.utils
  import json
@@ -154,6 +155,7 @@ std_headers = {
      'Accept-Encoding': 'gzip, deflate',
      'Accept-Language': 'en-us,en;q=0.5',
  }
+
  def preferredencoding():
      """Get preferred encoding.
  
@@ -187,7 +189,6 @@ else:
          with open(fn, 'w', encoding='utf-8') as f:
              json.dump(obj, f)
  
-
  def htmlentity_transform(matchobj):
      """Transforms an HTML entity to a character.
  
@@ -279,6 +280,12 @@ class AttrParser(compat_html_parser.HTMLParser):
              lines[-1] = lines[-1][:self.result[2][1]-self.result[1][1]]
          lines[-1] = lines[-1][:self.result[2][1]]
          return '\n'.join(lines).strip()
+# Hack for https://github.com/rg3/youtube-dl/issues/662
+if sys.version_info < (2, 7, 3):
+    AttrParser.parse_endtag = (lambda self, i:
+        i + len("</scr'+'ipt>")
+        if self.rawdata[i:].startswith("</scr'+'ipt>")
+        else compat_html_parser.HTMLParser.parse_endtag(self, i))
  
  def get_element_by_id(id, html):
      """Return the content of the tag with the specified ID in the passed HTML document"""
@@ -304,7 +311,7 @@ def clean_html(html):
      html = re.sub('<.*?>', '', html)
      # Replace html entities
      html = unescapeHTML(html)
-    return html
+    return html.strip()
  
  
  def sanitize_open(filename, open_mode):
@@ -322,7 +329,7 @@ def sanitize_open(filename, open_mode):
              if sys.platform == 'win32':
                  import msvcrt
                  msvcrt.setmode(sys.stdout.fileno(), os.O_BINARY)
-            return (sys.stdout, filename)
+            return (sys.stdout.buffer if hasattr(sys.stdout, 'buffer') else sys.stdout, filename)
          stream = open(encodeFilename(filename), open_mode)
          return (stream, filename)
      except (IOError, OSError) as err:
@@ -408,35 +415,33 @@ def encodeFilename(s):
          # match Windows 9x series as well. Besides, NT 4 is obsolete.)
          return s
      else:
-        return s.encode(sys.getfilesystemencoding(), 'ignore')
-
-def rsa_verify(message, signature, key):
-    from struct import pack
-    from hashlib import sha256
-    from sys import version_info
-    def b(x):
-        if version_info[0] == 2: return x
-        else: return x.encode('latin1')
-    assert(type(message) == type(b('')))
-    block_size = 0
-    n = key[0]
-    while n:
-        block_size += 1
-        n >>= 8
-    signature = pow(int(signature, 16), key[1], key[0])
-    raw_bytes = []
-    while signature:
-        raw_bytes.insert(0, pack("B", signature & 0xFF))
-        signature >>= 8
-    signature = (block_size - len(raw_bytes)) * b('\x00') + b('').join(raw_bytes)
-    if signature[0:2] != b('\x00\x01'): return False
-    signature = signature[2:]
-    if not b('\x00') in signature: return False
-    signature = signature[signature.index(b('\x00'))+1:]
-    if not signature.startswith(b('\x30\x31\x30\x0D\x06\x09\x60\x86\x48\x01\x65\x03\x04\x02\x01\x05\x00\x04\x20')): return False
-    signature = signature[19:]
-    if signature != sha256(message).digest(): return False
-    return True
+        encoding = sys.getfilesystemencoding()
+        if encoding is None:
+            encoding = 'utf-8'
+        return s.encode(encoding, 'ignore')
+
+def decodeOption(optval):
+    if optval is None:
+        return optval
+    if isinstance(optval, bytes):
+        optval = optval.decode(preferredencoding())
+
+    assert isinstance(optval, compat_str)
+    return optval
+
+class ExtractorError(Exception):
+    """Error during info extraction."""
+    def __init__(self, msg, tb=None):
+        """ tb, if given, is the original traceback (so that it can be printed out). """
+        super(ExtractorError, self).__init__(msg)
+        self.traceback = tb
+        self.exc_info = sys.exc_info()  # preserve original exception
+
+    def format_traceback(self):
+        if self.traceback is None:
+            return None
+        return u''.join(traceback.format_tb(self.traceback))
+
  
  class DownloadError(Exception):
      """Download Error exception.
@@ -445,7 +450,10 @@ class DownloadError(Exception):
      configured to continue on errors. They will contain the appropriate
      error message.
      """
-    pass
+    def __init__(self, msg, exc_info=None):
+        """ exc_info, if given, is the original exception that caused the trouble (as returned by sys.exc_info()). """
+        super(DownloadError, self).__init__(msg)
+        self.exc_info = exc_info
  
  
  class SameFileError(Exception):
@@ -463,7 +471,8 @@ class PostProcessingError(Exception):
      This exception may be raised by PostProcessor's .run() method to
      indicate an error in the postprocessing task.
      """
-    pass
+    def __init__(self, msg):
+        self.msg = msg
  
  class MaxDownloadsReached(Exception):
      """ --max-downloads limit has been reached. """
@@ -528,14 +537,19 @@ class YoutubeDLHandler(compat_urllib_request.HTTPHandler):
          return ret
  
      def http_request(self, req):
-        for h in std_headers:
+        for h,v in std_headers.items():
              if h in req.headers:
                  del req.headers[h]
-            req.add_header(h, std_headers[h])
+            req.add_header(h, v)
          if 'Youtubedl-no-compression' in req.headers:
              if 'Accept-encoding' in req.headers:
                  del req.headers['Accept-encoding']
              del req.headers['Youtubedl-no-compression']
+        if 'Youtubedl-user-agent' in req.headers:
+            if 'User-agent' in req.headers:
+                del req.headers['User-agent']
+            req.headers['User-agent'] = req.headers['Youtubedl-user-agent']
+            del req.headers['Youtubedl-user-agent']
          return req
  
      def http_response(self, req, resp):