[utils] Introduce YoutubeDLError base class for all youtube-dl exceptions

[youtube-dl] / youtube_dl / utils.py
diff --git a/youtube_dl/utils.py b/youtube_dl/utils.py

index 04452003766130996feb47ed9826463bb5024cf4..3f9e592e36033b2377e72872e2826cfc1a00764b 100644 (file)
--- a/youtube_dl/utils.py
+++ b/youtube_dl/utils.py
@@ -1,5 +1,5 @@
  #!/usr/bin/env python
-# -*- coding: utf-8 -*-
+# coding: utf-8
  
  from __future__ import unicode_literals
  
@@ -86,6 +86,11 @@ std_headers = {
  }
  
  
+USER_AGENTS = {
+    'Safari': 'Mozilla/5.0 (X11; Linux x86_64; rv:10.0) AppleWebKit/533.20.25 (KHTML, like Gecko) Version/5.0.4 Safari/533.20.27',
+}
+
+
  NO_DEFAULT = object()
  
  ENGLISH_MONTH_NAMES = [
@@ -123,7 +128,13 @@ DATE_FORMATS = (
      '%d %B %Y',
      '%d %b %Y',
      '%B %d %Y',
+    '%B %dst %Y',
+    '%B %dnd %Y',
+    '%B %dth %Y',
      '%b %d %Y',
+    '%b %dst %Y',
+    '%b %dnd %Y',
+    '%b %dth %Y',
      '%b %dst %Y %I:%M',
      '%b %dnd %Y %I:%M',
      '%b %dth %Y %I:%M',
@@ -132,6 +143,7 @@ DATE_FORMATS = (
      '%Y/%m/%d',
      '%Y/%m/%d %H:%M',
      '%Y/%m/%d %H:%M:%S',
+    '%Y-%m-%d %H:%M',
      '%Y-%m-%d %H:%M:%S',
      '%Y-%m-%d %H:%M:%S.%f',
      '%d.%m.%Y %H:%M',
@@ -165,6 +177,8 @@ DATE_FORMATS_MONTH_FIRST.extend([
      '%m/%d/%Y %H:%M:%S',
  ])
  
+PACKED_CODES_RE = r"}\('(.+)',(\d+),(\d+),'([^']+)'\.split\('\|'\)"
+
  
  def preferredencoding():
      """Get preferred encoding.
@@ -323,17 +337,30 @@ def get_element_by_id(id, html):
  
  
  def get_element_by_class(class_name, html):
-    return get_element_by_attribute(
+    """Return the content of the first tag with the specified class in the passed HTML document"""
+    retval = get_elements_by_class(class_name, html)
+    return retval[0] if retval else None
+
+
+def get_element_by_attribute(attribute, value, html, escape_value=True):
+    retval = get_elements_by_attribute(attribute, value, html, escape_value)
+    return retval[0] if retval else None
+
+
+def get_elements_by_class(class_name, html):
+    """Return the content of all tags with the specified class in the passed HTML document as a list"""
+    return get_elements_by_attribute(
          'class', r'[^\'"]*\b%s\b[^\'"]*' % re.escape(class_name),
          html, escape_value=False)
  
  
-def get_element_by_attribute(attribute, value, html, escape_value=True):
+def get_elements_by_attribute(attribute, value, html, escape_value=True):
      """Return the content of the tag with the specified attribute in the passed HTML document"""
  
      value = re.escape(value) if escape_value else value
  
-    m = re.search(r'''(?xs)
+    retlist = []
+    for m in re.finditer(r'''(?xs)
          <([a-zA-Z0-9:._-]+)
           (?:\s+[a-zA-Z0-9:._-]+(?:=[a-zA-Z0-9:._-]*|="[^"]*"|='[^']*'))*?
           \s+%s=['"]?%s['"]?
@@ -341,16 +368,15 @@ def get_element_by_attribute(attribute, value, html, escape_value=True):
          \s*>
          (?P<content>.*?)
          </\1>
-    ''' % (re.escape(attribute), value), html)
+    ''' % (re.escape(attribute), value), html):
+        res = m.group('content')
  
-    if not m:
-        return None
-    res = m.group('content')
+        if res.startswith('"') or res.startswith("'"):
+            res = res[1:-1]
  
-    if res.startswith('"') or res.startswith("'"):
-        res = res[1:-1]
+        retlist.append(unescapeHTML(res))
  
-    return unescapeHTML(res)
+    return retlist
  
  
  class HTMLAttributeParser(compat_HTMLParser):
@@ -494,7 +520,7 @@ def sanitize_path(s):
      if drive_or_unc:
          norm_path.pop(0)
      sanitized_path = [
-        path_part if path_part in ['.', '..'] else re.sub('(?:[/<>:"\\|\\\\?\\*]|[\s.]$)', '#', path_part)
+        path_part if path_part in ['.', '..'] else re.sub(r'(?:[/<>:"\|\\?\*]|[\s.]$)', '#', path_part)
          for path_part in norm_path]
      if drive_or_unc:
          sanitized_path.insert(0, drive_or_unc + os.path.sep)
@@ -675,7 +701,12 @@ def bug_reports_message():
      return msg
  
  
-class ExtractorError(Exception):
+class YoutubeDLError(Exception):
+    """Base exception for YoutubeDL errors."""
+    pass
+
+
+class ExtractorError(YoutubeDLError):
      """Error during info extraction."""
  
      def __init__(self, msg, tb=None, expected=False, cause=None, video_id=None):
@@ -716,7 +747,7 @@ class RegexNotFoundError(ExtractorError):
      pass
  
  
-class DownloadError(Exception):
+class DownloadError(YoutubeDLError):
      """Download Error exception.
  
      This exception may be thrown by FileDownloader objects if they are not
@@ -730,7 +761,7 @@ class DownloadError(Exception):
          self.exc_info = exc_info
  
  
-class SameFileError(Exception):
+class SameFileError(YoutubeDLError):
      """Same File exception.
  
      This exception will be thrown by FileDownloader objects if they detect
@@ -739,7 +770,7 @@ class SameFileError(Exception):
      pass
  
  
-class PostProcessingError(Exception):
+class PostProcessingError(YoutubeDLError):
      """Post Processing exception.
  
      This exception may be raised by PostProcessor's .run() method to
@@ -747,15 +778,16 @@ class PostProcessingError(Exception):
      """
  
      def __init__(self, msg):
+        super(PostProcessingError, self).__init__(msg)
          self.msg = msg
  
  
-class MaxDownloadsReached(Exception):
+class MaxDownloadsReached(YoutubeDLError):
      """ --max-downloads limit has been reached. """
      pass
  
  
-class UnavailableVideoError(Exception):
+class UnavailableVideoError(YoutubeDLError):
      """Unavailable Format exception.
  
      This exception will be thrown when a video is requested
@@ -764,7 +796,7 @@ class UnavailableVideoError(Exception):
      pass
  
  
-class ContentTooShortError(Exception):
+class ContentTooShortError(YoutubeDLError):
      """Content Too Short exception.
  
      This exception may be raised by FileDownloader objects when a file they
@@ -773,12 +805,15 @@ class ContentTooShortError(Exception):
      """
  
      def __init__(self, downloaded, expected):
+        super(ContentTooShortError, self).__init__(
+            'Downloaded {0} bytes, expected {1} bytes'.format(downloaded, expected)
+        )
          # Both in bytes
          self.downloaded = downloaded
          self.expected = expected
  
  
-class XAttrMetadataError(Exception):
+class XAttrMetadataError(YoutubeDLError):
      def __init__(self, code=None, msg='Unknown error'):
          super(XAttrMetadataError, self).__init__(msg)
          self.code = code
@@ -794,7 +829,7 @@ class XAttrMetadataError(Exception):
              self.reason = 'NOT_SUPPORTED'
  
  
-class XAttrUnavailableError(Exception):
+class XAttrUnavailableError(YoutubeDLError):
      pass
  
  
@@ -1176,7 +1211,7 @@ def date_from_str(date_str):
          return today
      if date_str == 'yesterday':
          return today - datetime.timedelta(days=1)
-    match = re.match('(now|today)(?P<sign>[+-])(?P<time>\d+)(?P<unit>day|week|month|year)(s)?', date_str)
+    match = re.match(r'(now|today)(?P<sign>[+-])(?P<time>\d+)(?P<unit>day|week|month|year)(s)?', date_str)
      if match is not None:
          sign = match.group('sign')
          time = int(match.group('time'))
@@ -1658,6 +1693,11 @@ def setproctitle(title):
          libc = ctypes.cdll.LoadLibrary('libc.so.6')
      except OSError:
          return
+    except TypeError:
+        # LoadLibrary in Windows Python 2.7.13 only expects
+        # a bytestring, but since unicode_literals turns
+        # every string into a unicode string, it fails.
+        return
      title_bytes = title.encode('utf-8')
      buf = ctypes.create_string_buffer(len(title_bytes))
      buf.value = title_bytes
@@ -1689,6 +1729,20 @@ def url_basename(url):
      return path.strip('/').split('/')[-1]
  
  
+def base_url(url):
+    return re.match(r'https?://[^?#&]+/', url).group()
+
+
+def urljoin(base, path):
+    if not isinstance(path, compat_str) or not path:
+        return None
+    if re.match(r'^(?:https?:)?//', path):
+        return path
+    if not isinstance(base, compat_str) or not re.match(r'^(?:https?:)?//', base):
+        return None
+    return compat_urlparse.urljoin(base, path)
+
+
  class HEADRequest(compat_urllib_request.Request):
      def get_method(self):
          return 'HEAD'
@@ -1745,7 +1799,7 @@ def parse_duration(s):
      s = s.strip()
  
      days, hours, mins, secs, ms = [None] * 5
-    m = re.match(r'(?:(?:(?:(?P<days>[0-9]+):)?(?P<hours>[0-9]+):)?(?P<mins>[0-9]+):)?(?P<secs>[0-9]+)(?P<ms>\.[0-9]+)?$', s)
+    m = re.match(r'(?:(?:(?:(?P<days>[0-9]+):)?(?P<hours>[0-9]+):)?(?P<mins>[0-9]+):)?(?P<secs>[0-9]+)(?P<ms>\.[0-9]+)?Z?$', s)
      if m:
          days, hours, mins, secs, ms = m.groups()
      else:
@@ -1762,11 +1816,11 @@ def parse_duration(s):
                  )?
                  (?:
                      (?P<secs>[0-9]+)(?P<ms>\.[0-9]+)?\s*s(?:ec(?:ond)?s?)?\s*
-                )?$''', s)
+                )?Z?$''', s)
          if m:
              days, hours, mins, secs, ms = m.groups()
          else:
-            m = re.match(r'(?i)(?:(?P<hours>[0-9.]+)\s*(?:hours?)|(?P<mins>[0-9.]+)\s*(?:mins?\.?|minutes?)\s*)$', s)
+            m = re.match(r'(?i)(?:(?P<hours>[0-9.]+)\s*(?:hours?)|(?P<mins>[0-9.]+)\s*(?:mins?\.?|minutes?)\s*)Z?$', s)
              if m:
                  hours, mins = m.groups()
              else:
@@ -1816,8 +1870,12 @@ def get_exe_version(exe, args=['--version'],
      """ Returns the version of the specified executable,
      or False if the executable is not present """
      try:
+        # STDIN should be redirected too. On UNIX-like systems, ffmpeg triggers
+        # SIGTTOU if youtube-dl is run in the background.
+        # See https://github.com/rg3/youtube-dl/issues/955#issuecomment-209789656
          out, _ = subprocess.Popen(
              [encodeArgument(exe)] + args,
+            stdin=subprocess.PIPE,
              stdout=subprocess.PIPE, stderr=subprocess.STDOUT).communicate()
      except OSError:
          return False
@@ -2071,11 +2129,18 @@ def strip_jsonp(code):
  
  
  def js_to_json(code):
+    COMMENT_RE = r'/\*(?:(?!\*/).)*?\*/|//[^\n]*'
+    SKIP_RE = r'\s*(?:{comment})?\s*'.format(comment=COMMENT_RE)
+    INTEGER_TABLE = (
+        (r'(?s)^(0[xX][0-9a-fA-F]+){skip}:?$'.format(skip=SKIP_RE), 16),
+        (r'(?s)^(0+[0-7]+){skip}:?$'.format(skip=SKIP_RE), 8),
+    )
+
      def fix_kv(m):
          v = m.group(0)
          if v in ('true', 'false', 'null'):
              return v
-        elif v.startswith('/*') or v == ',':
+        elif v.startswith('/*') or v.startswith('//') or v == ',':
              return ""
  
          if v[0] in ("'", '"'):
@@ -2086,11 +2151,6 @@ def js_to_json(code):
                  '\\x': '\\u00',
              }.get(m.group(0), m.group(0)), v[1:-1])
  
-        INTEGER_TABLE = (
-            (r'^(0[xX][0-9a-fA-F]+)\s*:?$', 16),
-            (r'^(0+[0-7]+)\s*:?$', 8),
-        )
-
          for regex, base in INTEGER_TABLE:
              im = re.match(regex, v)
              if im:
@@ -2102,11 +2162,11 @@ def js_to_json(code):
      return re.sub(r'''(?sx)
          "(?:[^"\\]*(?:\\\\|\\['"nurtbfx/\n]))*[^"\\]*"|
          '(?:[^'\\]*(?:\\\\|\\['"nurtbfx/\n]))*[^'\\]*'|
-        /\*.*?\*/|,(?=\s*[\]}])|
+        {comment}|,(?={skip}[\]}}])|
          [a-zA-Z_][.a-zA-Z_0-9]*|
-        \b(?:0[xX][0-9a-fA-F]+|0+[0-7]+)(?:\s*:)?|
-        [0-9]+(?=\s*:)
-        ''', fix_kv, code)
+        \b(?:0[xX][0-9a-fA-F]+|0+[0-7]+)(?:{skip}:)?|
+        [0-9]+(?={skip}:)
+        '''.format(comment=COMMENT_RE, skip=SKIP_RE), fix_kv, code)
  
  
  def qualities(quality_ids):
@@ -2332,6 +2392,7 @@ def _match_one(filter_part, dct):
          \s*(?P<op>%s)(?P<none_inclusive>\s*\?)?\s*
          (?:
              (?P<intval>[0-9.]+(?:[kKmMgGtTpPeEzZyY]i?[Bb]?)?)|
+            (?P<quote>["\'])(?P<quotedstrval>(?:\\.|(?!(?P=quote)|\\).)+?)(?P=quote)|
              (?P<strval>(?![0-9.])[a-z0-9A-Z]*)
          )
          \s*$
@@ -2339,11 +2400,22 @@ def _match_one(filter_part, dct):
      m = operator_rex.search(filter_part)
      if m:
          op = COMPARISON_OPERATORS[m.group('op')]
-        if m.group('strval') is not None:
+        actual_value = dct.get(m.group('key'))
+        if (m.group('quotedstrval') is not None or
+            m.group('strval') is not None or
+            # If the original field is a string and matching comparisonvalue is
+            # a number we should respect the origin of the original field
+            # and process comparison value as a string (see
+            # https://github.com/rg3/youtube-dl/issues/11082).
+            actual_value is not None and m.group('intval') is not None and
+                isinstance(actual_value, compat_str)):
              if m.group('op') not in ('=', '!='):
                  raise ValueError(
                      'Operator %s does not support string values!' % m.group('op'))
-            comparison_value = m.group('strval')
+            comparison_value = m.group('quotedstrval') or m.group('strval') or m.group('intval')
+            quote = m.group('quote')
+            if quote is not None:
+                comparison_value = comparison_value.replace(r'\%s' % quote, quote)
          else:
              try:
                  comparison_value = int(m.group('intval'))
@@ -2355,7 +2427,6 @@ def _match_one(filter_part, dct):
                      raise ValueError(
                          'Invalid integer value %r in filter part %r' % (
                              m.group('intval'), filter_part))
-        actual_value = dct.get(m.group('key'))
          if actual_value is None:
              return m.group('none_inclusive')
          return op(actual_value, comparison_value)
@@ -3017,9 +3088,7 @@ def encode_base_n(num, n, table=None):
  
  
  def decode_packed_codes(code):
-    mobj = re.search(
-        r"}\('(.+)',(\d+),(\d+),'([^']+)'\.split\('\|'\)",
-        code)
+    mobj = re.search(PACKED_CODES_RE, code)
      obfucasted_code, base, count, symbols = mobj.groups()
      base = int(base)
      count = int(count)