Merge https://github.com/rg3/youtube-dl into vimeo
authorRogério Brito <rbrito@ime.usp.br>
Thu, 4 Aug 2011 20:44:49 +0000 (17:44 -0300)
committerRogério Brito <rbrito@ime.usp.br>
Thu, 4 Aug 2011 20:44:49 +0000 (17:44 -0300)
1  2 
youtube-dl

diff --combined youtube-dl
index b734c997c8e3678dae2a48bc12c4bbf33db0becf,e8b19c8d0a6a90265a17000522e5a4e455bb1149..fe23e9f8f036eadb9c8ba711a08144bee93d459f
@@@ -38,7 -38,7 +38,7 @@@ except ImportError
        from cgi import parse_qs
  
  std_headers = {
-       'User-Agent': 'Mozilla/5.0 (X11; Linux x86_64; rv:2.0b11) Gecko/20100101 Firefox/4.0b11',
+       'User-Agent': 'Mozilla/5.0 (X11; Linux x86_64; rv:5.0.1) Gecko/20100101 Firefox/5.0.1',
        'Accept-Charset': 'ISO-8859-1,utf-8;q=0.7,*;q=0.7',
        'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8',
        'Accept-Encoding': 'gzip, deflate',
@@@ -1079,8 -1079,10 +1079,10 @@@ class YoutubeIE(InfoExtractor)
                # Decide which formats to download
                req_format = self._downloader.params.get('format', None)
  
-               if 'fmt_url_map' in video_info and len(video_info['fmt_url_map']) >= 1 and ',' in video_info['fmt_url_map'][0]:
-                       url_map = dict(tuple(pair.split('|')) for pair in video_info['fmt_url_map'][0].split(','))
+               if 'url_encoded_fmt_stream_map' in video_info and len(video_info['url_encoded_fmt_stream_map']) >= 1:
+                       url_data_strs = video_info['url_encoded_fmt_stream_map'][0].split(',')
+                       url_data = [dict(pairStr.split('=') for pairStr in uds.split('&')) for uds in url_data_strs]
+                       url_map = dict((ud['itag'], urllib.unquote(ud['url'])) for ud in url_data)
                        format_limit = self._downloader.params.get('format_limit', None)
                        if format_limit is not None and format_limit in self._available_formats:
                                format_list = self._available_formats[self._available_formats.index(format_limit):]
@@@ -1720,122 -1722,6 +1722,122 @@@ class YahooIE(InfoExtractor)
                        self._downloader.trouble(u'\nERROR: unable to download video')
  
  
 +class VimeoIE(InfoExtractor):
 +      """Information extractor for vimeo.com."""
 +
 +      # _VALID_URL matches Vimeo URLs
 +      _VALID_URL = r'(?:https?://)?(?:(?:www|player).)?vimeo\.com/(?:groups/[^/]+/)?(?:videos?/)?([0-9]+)'
 +
 +      def __init__(self, downloader=None):
 +              InfoExtractor.__init__(self, downloader)
 +
 +      @staticmethod
 +      def suitable(url):
 +              return (re.match(VimeoIE._VALID_URL, url) is not None)
 +
 +      def report_download_webpage(self, video_id):
 +              """Report webpage download."""
 +              self._downloader.to_screen(u'[vimeo] %s: Downloading webpage' % video_id)
 +
 +      def report_extraction(self, video_id):
 +              """Report information extraction."""
 +              self._downloader.to_screen(u'[vimeo] %s: Extracting information' % video_id)
 +
 +      def _real_initialize(self):
 +              return
 +
 +      def _real_extract(self, url, new_video=True):
 +              # Extract ID from URL
 +              mobj = re.match(self._VALID_URL, url)
 +              if mobj is None:
 +                      self._downloader.trouble(u'ERROR: Invalid URL: %s' % url)
 +                      return
 +
 +              # At this point we have a new video
 +              self._downloader.increment_downloads()
 +              video_id = mobj.group(1)
 +
 +              # Retrieve video webpage to extract further information
 +              request = urllib2.Request("http://vimeo.com/moogaloop/load/clip:%s" % video_id, None, std_headers)
 +              try:
 +                      self.report_download_webpage(video_id)
 +                      webpage = urllib2.urlopen(request).read()
 +              except (urllib2.URLError, httplib.HTTPException, socket.error), err:
 +                      self._downloader.trouble(u'ERROR: Unable to retrieve video webpage: %s' % str(err))
 +                      return
 +
 +              # Now we begin extracting as much information as we can from what we
 +              # retrieved. First we extract the information common to all extractors,
 +              # and latter we extract those that are Vimeo specific.
 +              self.report_extraction(video_id)
 +
 +              # Extract title
 +              mobj = re.search(r'<caption>(.*?)</caption>', webpage)
 +              if mobj is None:
 +                      self._downloader.trouble(u'ERROR: unable to extract video title')
 +                      return
 +              video_title = mobj.group(1).decode('utf-8')
 +              simple_title = re.sub(ur'(?u)([^%s]+)' % simple_title_chars, ur'_', video_title)
 +
 +              # Extract uploader
 +              mobj = re.search(r'<uploader_url>http://vimeo.com/(.*?)</uploader_url>', webpage)
 +              if mobj is None:
 +                      self._downloader.trouble(u'ERROR: unable to extract video uploader')
 +                      return
 +              video_uploader = mobj.group(1).decode('utf-8')
 +
 +              # Extract video thumbnail
 +              mobj = re.search(r'<thumbnail>(.*?)</thumbnail>', webpage)
 +              if mobj is None:
 +                      self._downloader.trouble(u'ERROR: unable to extract video thumbnail')
 +                      return
 +              video_thumbnail = mobj.group(1).decode('utf-8')
 +
 +              # # Extract video description
 +              # mobj = re.search(r'<meta property="og:description" content="(.*)" />', webpage)
 +              # if mobj is None:
 +              #       self._downloader.trouble(u'ERROR: unable to extract video description')
 +              #       return
 +              # video_description = mobj.group(1).decode('utf-8')
 +              # if not video_description: video_description = 'No description available.'
 +              video_description = 'Foo.'
 +
 +              # Vimeo specific: extract request signature
 +              mobj = re.search(r'<request_signature>(.*?)</request_signature>', webpage)
 +              if mobj is None:
 +                      self._downloader.trouble(u'ERROR: unable to extract request signature')
 +                      return
 +              sig = mobj.group(1).decode('utf-8')
 +
 +              # Vimeo specific: Extract request signature expiration
 +              mobj = re.search(r'<request_signature_expires>(.*?)</request_signature_expires>', webpage)
 +              if mobj is None:
 +                      self._downloader.trouble(u'ERROR: unable to extract request signature expiration')
 +                      return
 +              sig_exp = mobj.group(1).decode('utf-8')
 +
 +              video_url = "http://vimeo.com/moogaloop/play/clip:%s/%s/%s" % (video_id, sig, sig_exp)
 +
 +              try:
 +                      # Process video information
 +                      self._downloader.process_info({
 +                              'id':           video_id.decode('utf-8'),
 +                              'url':          video_url,
 +                              'uploader':     video_uploader,
 +                              'upload_date':  u'NA',
 +                              'title':        video_title,
 +                              'stitle':       simple_title,
 +                              'ext':          u'mp4',
 +                              'thumbnail':    video_thumbnail.decode('utf-8'),
 +                              'description':  video_description,
 +                              'thumbnail':    video_thumbnail,
 +                              'description':  video_description,
 +                              'player_url':   None,
 +                      })
 +              except UnavailableVideoError:
 +                      self._downloader.trouble(u'ERROR: unable to download video')
 +
 +
  class GenericIE(InfoExtractor):
        """Generic last-resort information extractor."""
  
@@@ -2839,7 -2725,7 +2841,7 @@@ if __name__ == '__main__'
                # Parse command line
                parser = optparse.OptionParser(
                        usage='Usage: %prog [options] url...',
-                       version='2011.03.29',
+                       version='2011.08.04',
                        conflict_handler='resolve',
                )
  
                                parser.error(u'invalid audio format specified')
  
                # Information extractors
 +              vimeo_ie = VimeoIE()
                youtube_ie = YoutubeIE()
                metacafe_ie = MetacafeIE(youtube_ie)
                dailymotion_ie = DailymotionIE()
                        'nopart': opts.nopart,
                        'updatetime': opts.updatetime,
                        })
 +              fd.add_info_extractor(vimeo_ie)
                fd.add_info_extractor(youtube_search_ie)
                fd.add_info_extractor(youtube_pl_ie)
                fd.add_info_extractor(youtube_user_ie)