[pornhub] Fix extraction (closes #12007)
[youtube-dl] / youtube_dl / extractor / pornhub.py
index 3eaf56973ec35072d8f0549c5850357ca94ed12b..5e930f45e0ac271a3c37e5bffa4a23f791a2ebf0 100644 (file)
@@ -156,7 +156,25 @@ class PornHubIE(InfoExtractor):
         comment_count = self._extract_count(
             r'All Comments\s*<span>\(([\d,.]+)\)', webpage, 'comment')
 
-        video_urls = list(map(compat_urllib_parse_unquote, re.findall(r"player_quality_[0-9]{3}p\s*=\s*'([^']+)'", webpage)))
+        video_variables = {}
+        for video_variablename, quote, video_variable in re.findall(
+                r'(player_quality_[0-9]{3,4}p[0-9a-z]+?)=\s*(["\'])(.*?)\2;', webpage):
+            video_variables[video_variablename] = video_variable
+
+        encoded_video_urls = []
+        for encoded_video_url in re.findall(
+                r'player_quality_[0-9]{3,4}p\s*=(.*?);', webpage):
+            encoded_video_urls.append(encoded_video_url)
+
+        # Decode the URLs 
+        video_urls = []
+        for url in encoded_video_urls:
+            for varname, varval in video_variables.items():
+                url = url.replace(varname, varval)
+            url = url.replace('+', '')
+            url = url.replace(' ', '')
+            video_urls.append(url)
+
         if webpage.find('"encrypted":true') != -1:
             password = compat_urllib_parse_unquote_plus(
                 self._search_regex(r'"video_title":"([^"]+)', webpage, 'password'))