[ceskateleveize:porady] Add extractor (closes #7411, closes #12645)

author Sergey M․ <dstftw@gmail.com>

Sat, 8 Apr 2017 12:42:09 +0000 (19:42 +0700)

committer Sergey M․ <dstftw@gmail.com>

Sat, 8 Apr 2017 12:46:42 +0000 (19:46 +0700)
author Sergey M․ <dstftw@gmail.com>
Sat, 8 Apr 2017 12:42:09 +0000 (19:42 +0700)
committer Sergey M․ <dstftw@gmail.com>
Sat, 8 Apr 2017 12:46:42 +0000 (19:46 +0700)
diff --git a/youtube_dl/extractor/ceskatelevize.py b/youtube_dl/extractor/ceskatelevize.py

index 0daee313fb12434db7bc903235185cefea4cc57f..e250de18ceb555e4750df54fcb7de1f1b92d6d49 100644 (file)
--- a/youtube_dl/extractor/ceskatelevize.py
+++ b/youtube_dl/extractor/ceskatelevize.py
@@ -12,6 +12,7 @@ from ..utils import (
      ExtractorError,
      float_or_none,
      sanitized_Request,
+    unescapeHTML,
      urlencode_postdata,
      USER_AGENTS,
  )
@@ -232,3 +233,47 @@ class CeskaTelevizeIE(InfoExtractor):
                      yield line
  
          return '\r\n'.join(_fix_subtitle(subtitles))
+
+
+class CeskaTelevizePoradyIE(InfoExtractor):
+    _VALID_URL = r'https?://(?:www\.)?ceskatelevize\.cz/porady/(?:[^/?#&]+/)*(?P<id>[^/#?]+)'
+    _TESTS = [{
+        # video with 18+ caution trailer
+        'url': 'http://www.ceskatelevize.cz/porady/10520528904-queer/215562210900007-bogotart/',
+        'info_dict': {
+            'id': '215562210900007-bogotart',
+            'title': 'Queer: Bogotart',
+            'description': 'Alternativní průvodce současným queer světem',
+        },
+        'playlist': [{
+            'info_dict': {
+                'id': '61924494876844842',
+                'ext': 'mp4',
+                'title': 'Queer: Bogotart (Varování 18+)',
+                'duration': 10.2,
+            },
+        }, {
+            'info_dict': {
+                'id': '61924494877068022',
+                'ext': 'mp4',
+                'title': 'Queer: Bogotart (Queer)',
+                'thumbnail': r're:^https?://.*\.jpg',
+                'duration': 1558.3,
+            },
+        }],
+        'params': {
+            # m3u8 download
+            'skip_download': True,
+        },
+    }]
+
+    def _real_extract(self, url):
+        video_id = self._match_id(url)
+
+        webpage = self._download_webpage(url, video_id)
+
+        data_url = unescapeHTML(self._search_regex(
+            r'<span[^>]*\bdata-url=(["\'])(?P<url>(?:(?!\1).)+)\1',
+            webpage, 'iframe player url', group='url'))
+
+        return self.url_result(data_url, ie=CeskaTelevizeIE.ie_key())
diff --git a/youtube_dl/extractor/extractors.py b/youtube_dl/extractor/extractors.py

index 2904dd4d1311e837818a1e1c1322d0f8687e9a27..72728d919304cd83a42270d5b98ecfe79c6fd519 100644 (file)
--- a/youtube_dl/extractor/extractors.py
+++ b/youtube_dl/extractor/extractors.py
@@ -165,7 +165,10 @@ from .ccc import CCCIE
  from .ccma import CCMAIE
  from .cctv import CCTVIE
  from .cda import CDAIE
-from .ceskatelevize import CeskaTelevizeIE
+from .ceskatelevize import (
+    CeskaTelevizeIE,
+    CeskaTelevizePoradyIE,
+)
  from .channel9 import Channel9IE
  from .charlierose import CharlieRoseIE
  from .chaturbate import ChaturbateIE
author	Sergey M․ <dstftw@gmail.com>
	Sat, 8 Apr 2017 12:42:09 +0000 (19:42 +0700)
committer	Sergey M․ <dstftw@gmail.com>
	Sat, 8 Apr 2017 12:46:42 +0000 (19:46 +0700)
youtube_dl/extractor/ceskatelevize.py		patch \| blob \| history
youtube_dl/extractor/extractors.py		patch \| blob \| history