projects
/
youtube-dl
/ commitdiff
commit
grep
author
committer
pickaxe
?
search:
re
summary
|
shortlog
|
log
|
commit
| commitdiff |
tree
raw
|
patch
|
inline
| side by side (parent:
5f562bd
)
[spankbang] Fix and improve metadata extraction
author
Sergey M․
<dstftw@gmail.com>
Sat, 13 Jul 2019 17:21:39 +0000
(
00:21
+0700)
committer
Sergey M․
<dstftw@gmail.com>
Sat, 13 Jul 2019 17:21:39 +0000
(
00:21
+0700)
youtube_dl/extractor/spankbang.py
patch
|
blob
|
history
diff --git
a/youtube_dl/extractor/spankbang.py
b/youtube_dl/extractor/spankbang.py
index eb0919e3aa08ee98ccd680de8b1a65b280ef472b..e040ada29b24542582f72f08f31b843d928af251 100644
(file)
--- a/
youtube_dl/extractor/spankbang.py
+++ b/
youtube_dl/extractor/spankbang.py
@@
-5,6
+5,7
@@
import re
from .common import InfoExtractor
from ..utils import (
ExtractorError,
from .common import InfoExtractor
from ..utils import (
ExtractorError,
+ merge_dicts,
orderedSet,
parse_duration,
parse_resolution,
orderedSet,
parse_duration,
parse_resolution,
@@
-26,6
+27,8
@@
class SpankBangIE(InfoExtractor):
'description': 'dillion harper masturbates on a bed',
'thumbnail': r're:^https?://.*\.jpg$',
'uploader': 'silly2587',
'description': 'dillion harper masturbates on a bed',
'thumbnail': r're:^https?://.*\.jpg$',
'uploader': 'silly2587',
+ 'timestamp': 1422571989,
+ 'upload_date': '20150129',
'age_limit': 18,
}
}, {
'age_limit': 18,
}
}, {
@@
-113,26
+116,29
@@
class SpankBangIE(InfoExtractor):
self._sort_formats(formats)
self._sort_formats(formats)
+ info = self._search_json_ld(webpage, video_id, default={})
+
title = self._html_search_regex(
title = self._html_search_regex(
- r'(?s)<h1[^>]*>(.+?)</h1>', webpage, 'title')
+ r'(?s)<h1[^>]*>(.+?)</h1>', webpage, 'title'
, default=None
)
description = self._search_regex(
r'<div[^>]+\bclass=["\']bottom[^>]+>\s*<p>[^<]*</p>\s*<p>([^<]+)',
description = self._search_regex(
r'<div[^>]+\bclass=["\']bottom[^>]+>\s*<p>[^<]*</p>\s*<p>([^<]+)',
- webpage, 'description', fatal=False)
- thumbnail = self._og_search_thumbnail(webpage)
- uploader = self._search_regex(
- r'class="user"[^>]*><img[^>]+>([^<]+)',
+ webpage, 'description', default=None)
+ thumbnail = self._og_search_thumbnail(webpage, default=None)
+ uploader = self._html_search_regex(
+ (r'(?s)<li[^>]+class=["\']profile[^>]+>(.+?)</a>',
+ r'class="user"[^>]*><img[^>]+>([^<]+)'),
webpage, 'uploader', default=None)
duration = parse_duration(self._search_regex(
r'<div[^>]+\bclass=["\']right_side[^>]+>\s*<span>([^<]+)',
webpage, 'uploader', default=None)
duration = parse_duration(self._search_regex(
r'<div[^>]+\bclass=["\']right_side[^>]+>\s*<span>([^<]+)',
- webpage, 'duration',
fatal=Fals
e))
+ webpage, 'duration',
default=Non
e))
view_count = str_to_int(self._search_regex(
view_count = str_to_int(self._search_regex(
- r'([\d,.]+)\s+plays', webpage, 'view count',
fatal=Fals
e))
+ r'([\d,.]+)\s+plays', webpage, 'view count',
default=Non
e))
age_limit = self._rta_search(webpage)
age_limit = self._rta_search(webpage)
- return {
+ return
merge_dicts(
{
'id': video_id,
'id': video_id,
- 'title': title,
+ 'title': title
or video_id
,
'description': description,
'thumbnail': thumbnail,
'uploader': uploader,
'description': description,
'thumbnail': thumbnail,
'uploader': uploader,
@@
-140,7
+146,8
@@
class SpankBangIE(InfoExtractor):
'view_count': view_count,
'formats': formats,
'age_limit': age_limit,
'view_count': view_count,
'formats': formats,
'age_limit': age_limit,
- }
+ }, info
+ )
class SpankBangPlaylistIE(InfoExtractor):
class SpankBangPlaylistIE(InfoExtractor):