Merge remote-tracking branch 'Boris-de/wdrmaus_fix#8562'

[youtube-dl] / youtube_dl / extractor / tvp.py
diff --git a/youtube_dl/extractor/tvp.py b/youtube_dl/extractor/tvp.py

index 2248a9fdff9bee514ac4cb9d5272567e388d8ea1..a4997cb8965dd9c31be4f5fc4a679fb0aa147b82 100644 (file)
--- a/youtube_dl/extractor/tvp.py
+++ b/youtube_dl/extractor/tvp.py
@@ -1,4 +1,4 @@
-# -*- coding: utf-8 -*-
+# coding: utf-8
  from __future__ import unicode_literals
  
  import re
@@ -6,91 +6,68 @@ import re
  from .common import InfoExtractor
  
  
-class TvpIE(InfoExtractor):
-    IE_NAME = 'tvp.pl'
-    _VALID_URL = r'https?://(?P<type>vod|www)\.tvp\.pl/.*/(?P<id>\d+)$'
-
-    _TESTS = [
-        {
-            'url': 'http://www.tvp.pl/warszawa/magazyny/campusnews/wideo/31102013/12878238',
-            'info_dict': {
-                'id': '12878238',
-                'ext': 'wmv',
-                'title': 'CAMPUSnews, 31.10.2013 - Odcinek 2',
-                'description': '',
-            },
-            'skip': 'Download has to use same server IP as extraction. Therefore, a good (load-balancing) DNS resolver will make the download fail.',
-        }, {
-            'url': 'http://vod.tvp.pl/filmy-fabularne/filmy-za-darmo/ogniem-i-mieczem/wideo/odc-2/4278035',
-            'info_dict': {
-                'id': '4278035',
-                'ext': 'wmv',
-                'title': 'Ogniem i mieczem, odc. 2',
-                'description': 'Bohun dowiaduje się o złamaniu przez kniahinię danego mu słowa i wyrusza do Rozłogów. Helenie w ostatniej chwili udaje się uciec dzięki pomocy Zagłoby.',
-            },
-            'skip': 'As above',
-        }, {
-            'url': 'http://vod.tvp.pl/seriale/obyczajowe/czas-honoru/sezon-1-1-13/i-seria-odc-13/194536',
-            'info_dict': {
-                'id': '194536',
-                'ext': 'mp4',
-                'title': 'Czas honoru, I seria – odc. 13',
-                'description': 'WŁADEK\nCzesław prosi Marię o dostarczenie Władkowi zarazki tyfusu. Jeśli zachoruje zostanie przewieziony do szpitala skąd łatwiej będzie go odbić. Czy matka zdecyduje się zarazić syna? Karol odwiedza Wandę przyznaje się, że ją oszukiwał, ale ostrzega też, że grozi jej aresztowanie i nalega, żeby wyjechała z Warszawy. Czy dziewczyna zdecyduje się znów oddalić od ukochanego? Rozpoczyna się akcja odbicia Władka.',
-            },
-        }, {
-            'url': 'http://www.tvp.pl/there-can-be-anything-so-i-shortened-it/17916176',
-            'info_dict': {
-                'id': '17916176',
-                'ext': 'mp4',
-                'title': 'rozmaitosci, TVP Gorzów pokaże filmy studentów z podroży dookoła świata',
-                'description': '',
-            },
-            'params': {
-                # m3u8 download
-                'skip_download': 'true',
-            },
-        }, {
-            'url': 'http://vod.tvp.pl/seriale/obyczajowe/na-sygnale/sezon-2-27-/odc-39/17834272',
-            'info_dict': {
-                'id': '17834272',
-                'ext': 'mp4',
-                'title': 'Na sygnale, odc. 39',
-                'description': 'Ekipa Wiktora ratuje młodą matkę, która spadła ze schodów trzymając na rękach noworodka. Okazuje się, że dziewczyna jest surogatką, a biologiczni rodzice dziecka próbują zmusić ją do oddania synka…',
-            },
-            'params': {
-                # m3u8 download
-                'skip_download': 'true',
-            },
+class TVPIE(InfoExtractor):
+    IE_NAME = 'tvp'
+    IE_DESC = 'Telewizja Polska'
+    _VALID_URL = r'https?://[^/]+\.tvp\.(?:pl|info)/(?:(?!\d+/)[^/]+/)*(?P<id>\d+)'
+
+    _TESTS = [{
+        'url': 'http://vod.tvp.pl/194536/i-seria-odc-13',
+        'md5': '8aa518c15e5cc32dfe8db400dc921fbb',
+        'info_dict': {
+            'id': '194536',
+            'ext': 'mp4',
+            'title': 'Czas honoru, I seria – odc. 13',
+        },
+    }, {
+        'url': 'http://www.tvp.pl/there-can-be-anything-so-i-shortened-it/17916176',
+        'md5': 'c3b15ed1af288131115ff17a17c19dda',
+        'info_dict': {
+            'id': '17916176',
+            'ext': 'mp4',
+            'title': 'TVP Gorzów pokaże filmy studentów z podroży dookoła świata',
          },
-    ]
+    }, {
+        'url': 'http://vod.tvp.pl/seriale/obyczajowe/na-sygnale/sezon-2-27-/odc-39/17834272',
+        'only_matching': True,
+    }, {
+        'url': 'http://wiadomosci.tvp.pl/25169746/24052016-1200',
+        'only_matching': True,
+    }, {
+        'url': 'http://krakow.tvp.pl/25511623/25lecie-mck-wyjatkowe-miejsce-na-mapie-krakowa',
+        'only_matching': True,
+    }, {
+        'url': 'http://teleexpress.tvp.pl/25522307/wierni-wzieli-udzial-w-procesjach',
+        'only_matching': True,
+    }, {
+        'url': 'http://sport.tvp.pl/25522165/krychowiak-uspokaja-w-sprawie-kontuzji-dwa-tygodnie-to-maksimum',
+        'only_matching': True,
+    }, {
+        'url': 'http://www.tvp.info/25511919/trwa-rewolucja-wladza-zdecydowala-sie-na-pogwalcenie-konstytucji',
+        'only_matching': True,
+    }]
  
      def _real_extract(self, url):
-        mobj = re.match(self._VALID_URL, url)
-        video_id = mobj.group('id')
+        video_id = self._match_id(url)
+
          webpage = self._download_webpage(
              'http://www.tvp.pl/sess/tvplayer.php?object_id=%s' % video_id, video_id)
-        title = self._og_search_title(webpage)
-        series = self._search_regex(
-            r'{name:\s*([\'"])SeriesTitle\1,\s*value:\s*\1(?P<series>.*?)\1},',
+
+        title = self._search_regex(
+            r'name\s*:\s*([\'"])Title\1\s*,\s*value\s*:\s*\1(?P<title>.+?)\1',
+            webpage, 'title', group='title')
+        series_title = self._search_regex(
+            r'name\s*:\s*([\'"])SeriesTitle\1\s*,\s*value\s*:\s*\1(?P<series>.+?)\1',
              webpage, 'series', group='series', default=None)
-        if series is not None and series not in title:
-            title = '%s, %s' % (series, title)
-        info_dict = {
-            'id': video_id,
-            'title': title,
-            'thumbnail': self._og_search_thumbnail(webpage),
-            'description': self._og_search_description(webpage, default=''),
-        }
-        if mobj.group('type') == 'vod' and info_dict['description'] == '':
-            info_dict.update({
-                'description': self._html_search_regex(
-                    r'(?s)<div\s+class=[\'"]opis.*?</div>',
-                    self._download_webpage(url, video_id), 'description', group=0),
-            })
+        if series_title:
+            title = '%s, %s' % (series_title, title)
+
+        thumbnail = self._search_regex(
+            r"poster\s*:\s*'([^']+)'", webpage, 'thumbnail', default=None)
  
          video_url = self._search_regex(
              r'0:{src:([\'"])(?P<url>.*?)\1', webpage, 'formats', group='url', default=None)
-        if video_url is None:
+        if not video_url:
              video_url = self._download_json(
                  'http://www.tvp.pl/pub/stat/videofileinfo?video_id=%s' % video_id,
                  video_id)['video_url']
@@ -99,66 +76,63 @@ class TvpIE(InfoExtractor):
          if ext != 'ism/manifest':
              if '/' in ext:
                  ext = 'mp4'
-            info_dict.update({
-                'ext': ext,
+            formats = [{
+                'format_id': 'direct',
                  'url': video_url,
-            })
+                'ext': ext,
+            }]
          else:
              m3u8_url = re.sub('([^/]*)\.ism/manifest', r'\1.ism/\1.m3u8', video_url)
              formats = self._extract_m3u8_formats(m3u8_url, video_id, 'mp4')
-            info_dict.update({
-                'formats': formats,
-            })
-        return info_dict
  
+        self._sort_formats(formats)
  
-class TvpSeriesIE(InfoExtractor):
-    IE_NAME = 'tvp.pl:Series'
+        return {
+            'id': video_id,
+            'title': title,
+            'thumbnail': thumbnail,
+            'formats': formats,
+        }
+
+
+class TVPSeriesIE(InfoExtractor):
+    IE_NAME = 'tvp:series'
      _VALID_URL = r'https?://vod\.tvp\.pl/(?:[^/]+/){2}(?P<id>[^/]+)/?$'
  
-    _TESTS = [
-        {
-            'url': 'http://vod.tvp.pl/filmy-fabularne/filmy-za-darmo/ogniem-i-mieczem',
-            'info_dict': {
-                'title': 'Ogniem i mieczem',
-                'id': '4278026',
-            },
-            'playlist_count': 4,
-        }, {
-            'url': 'http://vod.tvp.pl/audycje/podroze/boso-przez-swiat',
-            'info_dict': {
-                'title': 'Boso przez świat',
-                'id': '9329207',
-            },
-            'playlist_count': 86,
-        }
-    ]
-
-    def _force_download_webpage(self, url, v_id, tries=0):
-        if tries >= 5:
-            raise ExtractorError(
-                '%s: Cannot download webpage, try again later' % v_id)
-        # Sometimes happen, but in my tests second try always succeeded
-        try:
-            return self._download_webpage(url, v_id)
-        except IncompleteRead as e:
-            return self._force_download_webpage(url, v_id, tries+1)
-    
+    _TESTS = [{
+        'url': 'http://vod.tvp.pl/filmy-fabularne/filmy-za-darmo/ogniem-i-mieczem',
+        'info_dict': {
+            'title': 'Ogniem i mieczem',
+            'id': '4278026',
+        },
+        'playlist_count': 4,
+    }, {
+        'url': 'http://vod.tvp.pl/audycje/podroze/boso-przez-swiat',
+        'info_dict': {
+            'title': 'Boso przez świat',
+            'id': '9329207',
+        },
+        'playlist_count': 86,
+    }]
+
      def _real_extract(self, url):
          display_id = self._match_id(url)
-        webpage = self._force_download_webpage(url, display_id)
+        webpage = self._download_webpage(url, display_id, tries=5)
+
          title = self._html_search_regex(
-            r'(?s) id=[\'"]path[\'"]>(.*?)</span>', webpage, 'series')
-        title = title.split(' / ', 2)[-1]
+            r'(?s) id=[\'"]path[\'"]>(?:.*? / ){2}(.*?)</span>', webpage, 'series')
          playlist_id = self._search_regex(r'nodeId:\s*(\d+)', webpage, 'playlist id')
-        playlist = self._force_download_webpage(
+        playlist = self._download_webpage(
              'http://vod.tvp.pl/vod/seriesAjax?type=series&nodeId=%s&recommend'
-            'edId=0&sort=&page=0&pageSize=1000000' % playlist_id, display_id)
+            'edId=0&sort=&page=0&pageSize=10000' % playlist_id, display_id, tries=5,
+            note='Downloading playlist')
+
          videos_paths = re.findall(
              '(?s)class="shortTitle">.*?href="(/[^"]+)', playlist)
          entries = [
-            self.url_result('http://vod.tvp.pl%s' % v_path, ie=TvpIE.ie_key())
+            self.url_result('http://vod.tvp.pl%s' % v_path, ie=TVPIE.ie_key())
              for v_path in videos_paths]
+
          return {
              '_type': 'playlist',
              'id': playlist_id,