Merge branch 'jukebox' of https://github.com/remitamine/youtube-dl into remitamine...
[youtube-dl] / youtube_dl / extractor / freesound.py
index 9a2774d3ba6aabb64dd54b7cbee8d114e124a18e..5ff62af2a33d1743709bdb076dc0c80be0e3156b 100644 (file)
@@ -1,36 +1,39 @@
-# -*- coding: utf-8 -*-
+from __future__ import unicode_literals
+
 import re
 
 from .common import InfoExtractor
 
-class FreeSoundIE(InfoExtractor):
-    _VALID_URL = r'(?:http://)?(?:www\.)?freesound\.org/people/([^/]+)/sounds/([^/]+)'
+
+class FreesoundIE(InfoExtractor):
+    _VALID_URL = r'https?://(?:www\.)?freesound\.org/people/([^/]+)/sounds/(?P<id>[^/]+)'
     _TEST = {
-        u'url': u'http://www.freesound.org/people/miklovan/sounds/194503/',
-        u'file': u'194503.mp3',
-        u'md5': u'12280ceb42c81f19a515c745eae07650',
-        u'info_dict': {
-            u"title": u"gulls in the city.wav by miklovan",
-            u"uploader" : u"miklovan"
+        'url': 'http://www.freesound.org/people/miklovan/sounds/194503/',
+        'md5': '12280ceb42c81f19a515c745eae07650',
+        'info_dict': {
+            'id': '194503',
+            'ext': 'mp3',
+            'title': 'gulls in the city.wav',
+            'uploader': 'miklovan',
+            'description': 'the sounds of seagulls in the city',
         }
     }
 
     def _real_extract(self, url):
         mobj = re.match(self._VALID_URL, url)
-        music_id = mobj.group(2)
+        music_id = mobj.group('id')
         webpage = self._download_webpage(url, music_id)
-        title = self._html_search_regex(r'<meta property="og:title" content="([^"]*)"',
-                                webpage, 'music title')
-        music_url = self._html_search_regex(r'<meta property="og:audio" content="([^"]*)"',
-                                webpage, 'music url')       
-        uploader = self._html_search_regex(r'<meta property="og:audio:artist" content="([^"]*)"',
-                                webpage, 'music uploader')                                                                        
-        ext = music_url.split('.')[-1]
+        title = self._html_search_regex(
+            r'<div id="single_sample_header">.*?<a href="#">(.+?)</a>',
+            webpage, 'music title', flags=re.DOTALL)
+        description = self._html_search_regex(
+            r'<div id="sound_description">(.*?)</div>', webpage, 'description',
+            fatal=False, flags=re.DOTALL)
 
-        return [{
-            'id':       music_id,
-            'title':    title,            
-            'url':      music_url,
-            'uploader': uploader,
-            'ext':      ext,
-        }]
\ No newline at end of file
+        return {
+            'id': music_id,
+            'title': title,
+            'url': self._og_search_property('audio', webpage, 'music url'),
+            'uploader': self._og_search_property('audio:artist', webpage, 'music uploader'),
+            'description': description,
+        }