X-Git-Url: http://git.bitcoin.ninja/index.cgi?a=blobdiff_plain;f=youtube_dl%2Fextractor%2Fnova.py;h=3f9c776ef665ab47624eeab7ba60f5754dbf213e;hb=78be2eca7cb2806c3a51547da14968336febb57c;hp=30c64aaf861d58816d1b05f3d773887a27e70fe3;hpb=fcb04bcaca1b83cd3f13f494d7d775e35e0b6182;p=youtube-dl diff --git a/youtube_dl/extractor/nova.py b/youtube_dl/extractor/nova.py index 30c64aaf8..3f9c776ef 100644 --- a/youtube_dl/extractor/nova.py +++ b/youtube_dl/extractor/nova.py @@ -4,14 +4,17 @@ from __future__ import unicode_literals import re from .common import InfoExtractor -from ..utils import determine_ext +from ..utils import ( + clean_html, + unified_strdate, +) class NovaIE(InfoExtractor): IE_DESC = 'TN.cz, Prásk.tv, Nova.cz, Novaplus.cz, FANDA.tv, Krásná.cz and Doma.cz' - _VALID_URL = 'http://(?:[^.]+\.)?(?Ptv(?:noviny)?|tn|novaplus|vymena|fanda|krasna|doma|prask)\.nova\.cz/(?:[^/]+/)+(?P[^/]+?)(?:\.html|/?)$' + _VALID_URL = 'http://(?:[^.]+\.)?(?Ptv(?:noviny)?|tn|novaplus|vymena|fanda|krasna|doma|prask)\.nova\.cz/(?:[^/]+/)+(?P[^/]+?)(?:\.html|/|$)' _TESTS = [{ - 'url': 'http://tvnoviny.nova.cz/clanek/novinky/co-na-sebe-sportaci-praskli-vime-jestli-pujde-hrdlicka-na-materskou.html', + 'url': 'http://tvnoviny.nova.cz/clanek/novinky/co-na-sebe-sportaci-praskli-vime-jestli-pujde-hrdlicka-na-materskou.html?utm_source=tvnoviny&utm_medium=cpfooter&utm_campaign=novaplus', 'info_dict': { 'id': '1608920', 'display_id': 'co-na-sebe-sportaci-praskli-vime-jestli-pujde-hrdlicka-na-materskou', @@ -25,7 +28,7 @@ class NovaIE(InfoExtractor): 'skip_download': True, } }, { - 'url': 'http://tn.nova.cz/clanek/tajemstvi-ukryte-v-podzemi-specialni-nemocnice-v-prazske-krci.html', + 'url': 'http://tn.nova.cz/clanek/tajemstvi-ukryte-v-podzemi-specialni-nemocnice-v-prazske-krci.html#player_13260', 'md5': '1dd7b9d5ea27bc361f110cd855a19bd3', 'info_dict': { 'id': '1757139', @@ -36,13 +39,13 @@ class NovaIE(InfoExtractor): 'thumbnail': 're:^https?://.*\.(?:jpg)', } }, { - 'url': 'http://novaplus.nova.cz/porad/policie-modrava/video/5591-policie-modrava-15-dil-blondynka-na-hrbitove/', + 'url': 'http://novaplus.nova.cz/porad/policie-modrava/video/5591-policie-modrava-15-dil-blondynka-na-hrbitove', 'info_dict': { 'id': '1756825', 'display_id': '5591-policie-modrava-15-dil-blondynka-na-hrbitove', - 'ext': 'mp4', + 'ext': 'flv', 'title': 'Policie Modrava - 15. díl - Blondýnka na hřbitově', - 'description': 'md5:d804ba6b30bc7da2705b1fea961bddfe', + 'description': 'md5:dc24e50be5908df83348e50d1431295e', # Make sure this description is clean of html tags 'thumbnail': 're:^https?://.*\.(?:jpg)', }, 'params': { @@ -53,7 +56,7 @@ class NovaIE(InfoExtractor): 'url': 'http://novaplus.nova.cz/porad/televizni-noviny/video/5585-televizni-noviny-30-5-2015/', 'info_dict': { 'id': '1756858', - 'ext': 'mp4', + 'ext': 'flv', 'title': 'Televizní noviny - 30. 5. 2015', 'thumbnail': 're:^https?://.*\.(?:jpg)', 'upload_date': '20150530', @@ -132,25 +135,36 @@ class NovaIE(InfoExtractor): config = self._download_json( config_url, display_id, 'Downloading config JSON', - transform_source=lambda s: re.sub(r'var\s+[\da-zA-Z_]+\s*=\s*({.+?});', r'\1', s)) + transform_source=lambda s: s[s.index('{'):s.rindex('}') + 1]) mediafile = config['mediafile'] video_url = mediafile['src'] - ext = determine_ext(video_url) - video_url = video_url.replace('&{}:'.format(ext), '') + + m = re.search(r'^(?Prtmpe?://[^/]+/(?P[^/]+?))/&*(?P.+)$', video_url) + if m: + formats = [{ + 'url': m.group('url'), + 'app': m.group('app'), + 'play_path': m.group('playpath'), + 'player_path': 'http://tvnoviny.nova.cz/static/shared/app/videojs/video-js.swf', + 'ext': 'flv', + }] + else: + formats = [{ + 'url': video_url, + }] + self._sort_formats(formats) title = mediafile.get('meta', {}).get('title') or self._og_search_title(webpage) - description = self._og_search_description(webpage) + description = clean_html(self._og_search_description(webpage, default=None)) thumbnail = config.get('poster') - mobj = None if site == 'novaplus': - mobj = re.search(r'(?P\d{1,2})-(?P\d{1,2})-(?P\d{4})$', display_id) - if site == 'fanda': - mobj = re.search( - r'(?P\d{1,2})\.(?P\d{1,2})\.(?P\d{4})\b', webpage) - if mobj: - upload_date = '{}{:02d}{:02d}'.format(mobj.group('year'), int(mobj.group('month')), int(mobj.group('day'))) + upload_date = unified_strdate(self._search_regex( + r'(\d{1,2}-\d{1,2}-\d{4})$', display_id, 'upload date', default=None)) + elif site == 'fanda': + upload_date = unified_strdate(self._search_regex( + r'(\d{1,2}\.\d{1,2}\.\d{4})', webpage, 'upload date', default=None)) else: upload_date = None @@ -161,6 +175,5 @@ class NovaIE(InfoExtractor): 'description': description, 'upload_date': upload_date, 'thumbnail': thumbnail, - 'url': video_url, - 'ext': ext, + 'formats': formats, }