[wdr] Support overviews (Fixes #4651)

author Philipp Hagemeister <phihag@phihag.de>

Fri, 9 Jan 2015 20:33:07 +0000 (21:33 +0100)

committer Philipp Hagemeister <phihag@phihag.de>

Fri, 9 Jan 2015 20:33:07 +0000 (21:33 +0100)
author Philipp Hagemeister <phihag@phihag.de>
Fri, 9 Jan 2015 20:33:07 +0000 (21:33 +0100)
committer Philipp Hagemeister <phihag@phihag.de>
Fri, 9 Jan 2015 20:33:07 +0000 (21:33 +0100)
diff --git a/youtube_dl/extractor/wdr.py b/youtube_dl/extractor/wdr.py

index d1c46ccb347125fd048be7b3b8a2b5860db35e32..45466e31b7445f8dd8da742308dcc69f2ff1152f 100644 (file)
--- a/youtube_dl/extractor/wdr.py
+++ b/youtube_dl/extractor/wdr.py
@@ -1,6 +1,7 @@
  # -*- coding: utf-8 -*-
  from __future__ import unicode_literals
  
+import itertools
  import re
  
  from .common import InfoExtractor
@@ -67,6 +68,10 @@ class WDRIE(InfoExtractor):
                  'upload_date': '20140717',
              },
          },
+        {
+            'url': 'http://www1.wdr.de/mediathek/video/sendungen/quarks_und_co/filterseite-quarks-und-co100.html',
+            'playlist_mincount': 146,
+        }
      ]
  
      def _real_extract(self, url):
@@ -81,6 +86,27 @@ class WDRIE(InfoExtractor):
                  self.url_result(page_url + href, 'WDR')
                  for href in re.findall(r'<a href="/?(.+?%s\.html)" rel="nofollow"' % self._PLAYER_REGEX, webpage)
              ]
+
+            if entries:  # Playlist page
+                return self.playlist_result(entries, page_id)
+
+            # Overview page
+            entries = []
+            for page_num in itertools.count(2):
+                hrefs = re.findall(
+                    r'<li class="mediathekvideo"\s*>\s*<img[^>]*>\s*<a href="(/mediathek/video/[^"]+)"',
+                    webpage)
+                entries.extend(
+                    self.url_result(page_url + href, 'WDR')
+                    for href in hrefs)
+                next_url_m = re.search(
+                    r'<li class="nextToLast">\s*<a href="([^"]+)"', webpage)
+                if not next_url_m:
+                    break
+                next_url = page_url + next_url_m.group(1)
+                webpage = self._download_webpage(
+                    next_url, page_id,
+                    note='Downloading playlist page %d' % page_num)
              return self.playlist_result(entries, page_id)
  
          flashvars = compat_parse_qs(
author	Philipp Hagemeister <phihag@phihag.de>
	Fri, 9 Jan 2015 20:33:07 +0000 (21:33 +0100)
committer	Philipp Hagemeister <phihag@phihag.de>
	Fri, 9 Jan 2015 20:33:07 +0000 (21:33 +0100)