[npo] Update test

[youtube-dl.git] / youtube_dl / extractor / npo.py
diff --git a/youtube_dl/extractor/npo.py b/youtube_dl/extractor/npo.py

index c075618e84cb8181e27c2a9dc3cc033a16d5dea4..cf6a388e56485f34cf7cebe89e9d97d9c58fd7f9 100644 (file)
--- a/youtube_dl/extractor/npo.py
+++ b/youtube_dl/extractor/npo.py
@@ -1,7 +1,12 @@
  from __future__ import unicode_literals
  
-from .subtitles import SubtitlesInfoExtractor
+import re
+
  from .common import InfoExtractor
+from ..compat import (
+    compat_urllib_request,
+    compat_urllib_parse,
+)
  from ..utils import (
      fix_xml_ampersands,
      parse_duration,
@@ -12,18 +17,44 @@ from ..utils import (
  )
  
  
-class NPOBaseIE(SubtitlesInfoExtractor):
+class NPOBaseIE(InfoExtractor):
      def _get_token(self, video_id):
          token_page = self._download_webpage(
              'http://ida.omroep.nl/npoplayer/i.js',
              video_id, note='Downloading token')
-        return self._search_regex(
+        token = self._search_regex(
              r'npoplayer\.token = "(.+?)"', token_page, 'token')
+        # Decryption algorithm extracted from http://npoplayer.omroep.nl/csjs/npoplayer-min.js
+        token_l = list(token)
+        first = second = None
+        for i in range(5, len(token_l) - 4):
+            if token_l[i].isdigit():
+                if first is None:
+                    first = i
+                elif second is None:
+                    second = i
+        if first is None or second is None:
+            first = 12
+            second = 13
+
+        token_l[first], token_l[second] = token_l[second], token_l[first]
+
+        return ''.join(token_l)
  
  
  class NPOIE(NPOBaseIE):
-    IE_NAME = 'npo.nl'
-    _VALID_URL = r'https?://(?:www\.)?npo\.nl/(?!live|radio)[^/]+/[^/]+/(?P<id>[^/?]+)'
+    IE_NAME = 'npo'
+    IE_DESC = 'npo.nl and ntr.nl'
+    _VALID_URL = r'''(?x)
+                    https?://
+                        (?:www\.)?
+                        (?:
+                            npo\.nl/(?!live|radio)(?:[^/]+/){2}|
+                            ntr\.nl/(?:[^/]+/){2,}|
+                            omroepwnl\.nl/video/fragment/[^/]+__
+                        )
+                        (?P<id>[^/?#]+)
+                '''
  
      _TESTS = [
          {
@@ -56,7 +87,7 @@ class NPOIE(NPOBaseIE):
                  'id': 'VPWON_1169289',
                  'ext': 'm4v',
                  'title': 'Tegenlicht',
-                'description': 'md5:d6476bceb17a8c103c76c3b708f05dd1',
+                'description': 'md5:52cf4eefbc96fffcbdc06d024147abea',
                  'upload_date': '20130225',
                  'duration': 3000,
              },
@@ -85,6 +116,30 @@ class NPOIE(NPOBaseIE):
                  'title': 'Hoe gaat Europa verder na Parijs?',
              },
          },
+        {
+            'url': 'http://www.ntr.nl/Aap-Poot-Pies/27/detail/Aap-poot-pies/VPWON_1233944#content',
+            'md5': '01c6a2841675995da1f0cf776f03a9c3',
+            'info_dict': {
+                'id': 'VPWON_1233944',
+                'ext': 'm4v',
+                'title': 'Aap, poot, pies',
+                'description': 'md5:c9c8005d1869ae65b858e82c01a91fde',
+                'upload_date': '20150508',
+                'duration': 599,
+            },
+        },
+        {
+            'url': 'http://www.omroepwnl.nl/video/fragment/vandaag-de-dag-verkiezingen__POMS_WNL_853698',
+            'md5': 'd30cd8417b8b9bca1fdff27428860d08',
+            'info_dict': {
+                'id': 'POW_00996502',
+                'ext': 'm4v',
+                'title': '''"Dit is wel een 'landslide'..."''',
+                'description': 'md5:f8d66d537dfb641380226e31ca57b8e8',
+                'upload_date': '20150508',
+                'duration': 462,
+            },
+        }
      ]
  
      def _real_extract(self, url):
@@ -93,12 +148,17 @@ class NPOIE(NPOBaseIE):
  
      def _get_info(self, video_id):
          metadata = self._download_json(
-            'http://e.omroep.nl/metadata/aflevering/%s' % video_id,
+            'http://e.omroep.nl/metadata/%s' % video_id,
              video_id,
              # We have to remove the javascript callback
              transform_source=strip_jsonp,
          )
  
+        # For some videos actual video id (prid) is different (e.g. for
+        # http://www.omroepwnl.nl/video/fragment/vandaag-de-dag-verkiezingen__POMS_WNL_853698
+        # video id is POMS_WNL_853698 but prid is POW_00996502)
+        video_id = metadata.get('prid') or video_id
+
          token = self._get_token(video_id)
  
          formats = []
@@ -164,13 +224,10 @@ class NPOIE(NPOBaseIE):
  
          subtitles = {}
          if metadata.get('tt888') == 'ja':
-            subtitles['nl'] = 'http://e.omroep.nl/tt888/%s' % video_id
-
-        if self._downloader.params.get('listsubtitles', False):
-            self._list_available_subtitles(video_id, subtitles)
-            return
-
-        subtitles = self.extract_subtitles(video_id, subtitles)
+            subtitles['nl'] = [{
+                'ext': 'vtt',
+                'url': 'http://e.omroep.nl/tt888/%s' % video_id,
+            }]
  
          return {
              'id': video_id,
@@ -223,7 +280,8 @@ class NPOLiveIE(NPOBaseIE):
          if streams:
              for stream in streams:
                  stream_type = stream.get('type').lower()
-                if stream_type == 'ss':
+                # smooth streaming is not supported
+                if stream_type in ['ss', 'ms']:
                      continue
                  stream_info = self._download_json(
                      'http://ida.omroep.nl/aapi/?stream=%s&token=%s&type=jsonp'
@@ -234,7 +292,10 @@ class NPOLiveIE(NPOBaseIE):
                  stream_url = self._download_json(
                      stream_info['stream'], display_id,
                      'Downloading %s URL' % stream_type,
-                    transform_source=strip_jsonp)
+                    'Unable to download %s URL' % stream_type,
+                    transform_source=strip_jsonp, fatal=False)
+                if not stream_url:
+                    continue
                  if stream_type == 'hds':
                      f4m_formats = self._extract_f4m_formats(stream_url, display_id)
                      # f4m downloader downloads only piece of live stream
@@ -246,6 +307,7 @@ class NPOLiveIE(NPOBaseIE):
                  else:
                      formats.append({
                          'url': stream_url,
+                        'preference': -10,
                      })
  
          self._sort_formats(formats)