debian/README.source: Change 'markup'.

[youtubedl] / youtube_dl / extractor / go.py
diff --git a/youtube_dl/extractor/go.py b/youtube_dl/extractor/go.py

index e781405f2f7d55aa51f8cc50cc99df22ee5dff40..03e48f4ea4b93153ef445f0d9a66779144821361 100644 (file)
--- a/youtube_dl/extractor/go.py
+++ b/youtube_dl/extractor/go.py
@@ -25,18 +25,23 @@ class GoIE(AdobePassIE):
          },
          'watchdisneychannel': {
              'brand': '004',
          },
          'watchdisneychannel': {
              'brand': '004',
-            'requestor_id': 'Disney',
+            'resource_id': 'Disney',
          },
          'watchdisneyjunior': {
              'brand': '008',
          },
          'watchdisneyjunior': {
              'brand': '008',
-            'requestor_id': 'DisneyJunior',
+            'resource_id': 'DisneyJunior',
          },
          'watchdisneyxd': {
              'brand': '009',
          },
          'watchdisneyxd': {
              'brand': '009',
-            'requestor_id': 'DisneyXD',
+            'resource_id': 'DisneyXD',
+        },
+        'disneynow': {
+            'brand': '011',
+            'resource_id': 'Disney',
          }
      }
          }
      }
-    _VALID_URL = r'https?://(?:(?P<sub_domain>%s)\.)?go\.com/(?:(?:[^/]+/)*(?P<id>vdka\w+)|(?:[^/]+/)*(?P<display_id>[^/?#]+))' % '|'.join(_SITE_INFO.keys())
+    _VALID_URL = r'https?://(?:(?:(?P<sub_domain>%s)\.)?go|(?P<sub_domain_2>disneynow))\.com/(?:(?:[^/]+/)*(?P<id>vdka\w+)|(?:[^/]+/)*(?P<display_id>[^/?#]+))'\
+                 % '|'.join(list(_SITE_INFO.keys()) + ['disneynow'])
      _TESTS = [{
          'url': 'http://abc.go.com/shows/designated-survivor/video/most-recent/VDKA3807643',
          'info_dict': {
      _TESTS = [{
          'url': 'http://abc.go.com/shows/designated-survivor/video/most-recent/VDKA3807643',
          'info_dict': {
@@ -62,6 +67,17 @@ class GoIE(AdobePassIE):
      }, {
          'url': 'http://abc.go.com/shows/world-news-tonight/episode-guide/2017-02/17-021717-intense-stand-off-between-man-with-rifle-and-police-in-oakland',
          'only_matching': True,
      }, {
          'url': 'http://abc.go.com/shows/world-news-tonight/episode-guide/2017-02/17-021717-intense-stand-off-between-man-with-rifle-and-police-in-oakland',
          'only_matching': True,
+    }, {
+        # brand 004
+        'url': 'http://disneynow.go.com/shows/big-hero-6-the-series/season-01/episode-10-mr-sparkles-loses-his-sparkle/vdka4637915',
+        'only_matching': True,
+    }, {
+        # brand 008
+        'url': 'http://disneynow.go.com/shows/minnies-bow-toons/video/happy-campers/vdka4872013',
+        'only_matching': True,
+    }, {
+        'url': 'https://disneynow.com/shows/minnies-bow-toons/video/happy-campers/vdka4872013',
+        'only_matching': True,
      }]
  
      def _extract_videos(self, brand, video_id='-1', show_id='-1'):
      }]
  
      def _extract_videos(self, brand, video_id='-1', show_id='-1'):
@@ -71,15 +87,26 @@ class GoIE(AdobePassIE):
              display_id)['video']
  
      def _real_extract(self, url):
              display_id)['video']
  
      def _real_extract(self, url):
-        sub_domain, video_id, display_id = re.match(self._VALID_URL, url).groups()
-        site_info = self._SITE_INFO[sub_domain]
-        brand = site_info['brand']
-        if not video_id:
-            webpage = self._download_webpage(url, display_id)
+        mobj = re.match(self._VALID_URL, url)
+        sub_domain = mobj.group('sub_domain') or mobj.group('sub_domain_2')
+        video_id, display_id = mobj.group('id', 'display_id')
+        site_info = self._SITE_INFO.get(sub_domain, {})
+        brand = site_info.get('brand')
+        if not video_id or not site_info:
+            webpage = self._download_webpage(url, display_id or video_id)
              video_id = self._search_regex(
                  # There may be inner quotes, e.g. data-video-id="'VDKA3609139'"
                  # from http://freeform.go.com/shows/shadowhunters/episodes/season-2/1-this-guilty-blood
              video_id = self._search_regex(
                  # There may be inner quotes, e.g. data-video-id="'VDKA3609139'"
                  # from http://freeform.go.com/shows/shadowhunters/episodes/season-2/1-this-guilty-blood
-                r'data-video-id=["\']*(VDKA\w+)', webpage, 'video id', default=None)
+                r'data-video-id=["\']*(VDKA\w+)', webpage, 'video id',
+                default=video_id)
+            if not site_info:
+                brand = self._search_regex(
+                    (r'data-brand=\s*["\']\s*(\d+)',
+                     r'data-page-brand=\s*["\']\s*(\d+)'), webpage, 'brand',
+                    default='004')
+                site_info = next(
+                    si for _, si in self._SITE_INFO.items()
+                    if si.get('brand') == brand)
              if not video_id:
                  # show extraction works for Disney, DisneyJunior and DisneyXD
                  # ABC and Freeform has different layout
              if not video_id:
                  # show extraction works for Disney, DisneyJunior and DisneyXD
                  # ABC and Freeform has different layout
@@ -112,8 +139,8 @@ class GoIE(AdobePassIE):
                      'device': '001',
                  }
                  if video_data.get('accesslevel') == '1':
                      'device': '001',
                  }
                  if video_data.get('accesslevel') == '1':
-                    requestor_id = site_info['requestor_id']
-                    resource = self._get_mvpd_resource(
+                    requestor_id = site_info.get('requestor_id', 'DisneyChannels')
+                    resource = site_info.get('resource_id') or self._get_mvpd_resource(
                          requestor_id, title, video_id, None)
                      auth = self._extract_mvpd_auth(
                          url, video_id, requestor_id, resource)
                          requestor_id, title, video_id, None)
                      auth = self._extract_mvpd_auth(
                          url, video_id, requestor_id, resource)