New upstream version 2019.01.16

[youtubedl] / youtube_dl / extractor / iqiyi.py
diff --git a/youtube_dl/extractor/iqiyi.py b/youtube_dl/extractor/iqiyi.py

index 01c7b30428f8a750c9932b0ea734f795c09866d6..4b081bd469ca084f5ecf47ac38cc97326b011b31 100644 (file)
--- a/youtube_dl/extractor/iqiyi.py
+++ b/youtube_dl/extractor/iqiyi.py
@@ -173,11 +173,12 @@ class IqiyiIE(InfoExtractor):
          }
      }, {
          'url': 'http://www.iqiyi.com/v_19rrhnnclk.html',
          }
      }, {
          'url': 'http://www.iqiyi.com/v_19rrhnnclk.html',
-        'md5': '667171934041350c5de3f5015f7f1152',
+        'md5': 'b7dc800a4004b1b57749d9abae0472da',
          'info_dict': {
              'id': 'e3f585b550a280af23c98b6cb2be19fb',
              'ext': 'mp4',
          'info_dict': {
              'id': 'e3f585b550a280af23c98b6cb2be19fb',
              'ext': 'mp4',
-            'title': '名侦探柯南 国语版：第752集 迫近灰原秘密的黑影 下篇',
+            # This can be either Simplified Chinese or Traditional Chinese
+            'title': r're:^(?:名侦探柯南 国语版：第752集 迫近灰原秘密的黑影 下篇|名偵探柯南 國語版：第752集 迫近灰原秘密的黑影 下篇)$',
          },
          'skip': 'Geo-restricted to China',
      }, {
          },
          'skip': 'Geo-restricted to China',
      }, {
@@ -188,7 +189,11 @@ class IqiyiIE(InfoExtractor):
          'only_matching': True,
      }, {
          'url': 'http://yule.iqiyi.com/pcb.html',
          'only_matching': True,
      }, {
          'url': 'http://yule.iqiyi.com/pcb.html',
-        'only_matching': True,
+        'info_dict': {
+            'id': '4a0af228fddb55ec96398a364248ed7f',
+            'ext': 'mp4',
+            'title': '第2017-04-21期 女艺人频遭极端粉丝骚扰',
+        },
      }, {
          # VIP-only video. The first 2 parts (6 minutes) are available without login
          # MD5 sums omitted as values are different on Travis CI and my machine
      }, {
          # VIP-only video. The first 2 parts (6 minutes) are available without login
          # MD5 sums omitted as values are different on Travis CI and my machine
@@ -234,7 +239,7 @@ class IqiyiIE(InfoExtractor):
          return ohdave_rsa_encrypt(data, e, N)
  
      def _login(self):
          return ohdave_rsa_encrypt(data, e, N)
  
      def _login(self):
-        (username, password) = self._get_login_info()
+        username, password = self._get_login_info()
  
          # No authentication to be performed
          if not username:
  
          # No authentication to be performed
          if not username:
@@ -336,15 +341,18 @@ class IqiyiIE(InfoExtractor):
              url, 'temp_id', note='download video page')
  
          # There's no simple way to determine whether an URL is a playlist or not
              url, 'temp_id', note='download video page')
  
          # There's no simple way to determine whether an URL is a playlist or not
-        # So detect it
-        playlist_result = self._extract_playlist(webpage)
-        if playlist_result:
-            return playlist_result
-
+        # Sometimes there are playlist links in individual videos, so treat it
+        # as a single video first
          tvid = self._search_regex(
          tvid = self._search_regex(
-            r'data-player-tvid\s*=\s*[\'"](\d+)', webpage, 'tvid')
+            r'data-(?:player|shareplattrigger)-tvid\s*=\s*[\'"](\d+)', webpage, 'tvid', default=None)
+        if tvid is None:
+            playlist_result = self._extract_playlist(webpage)
+            if playlist_result:
+                return playlist_result
+            raise ExtractorError('Can\'t find any video')
+
          video_id = self._search_regex(
          video_id = self._search_regex(
-            r'data-player-videoid\s*=\s*[\'"]([a-f\d]+)', webpage, 'video_id')
+            r'data-(?:player|shareplattrigger)-videoid\s*=\s*[\'"]([a-f\d]+)', webpage, 'video_id')
  
          formats = []
          for _ in range(5):
  
          formats = []
          for _ in range(5):
@@ -376,7 +384,8 @@ class IqiyiIE(InfoExtractor):
  
          self._sort_formats(formats)
          title = (get_element_by_id('widget-videotitle', webpage) or
  
          self._sort_formats(formats)
          title = (get_element_by_id('widget-videotitle', webpage) or
-                 clean_html(get_element_by_attribute('class', 'mod-play-tit', webpage)))
+                 clean_html(get_element_by_attribute('class', 'mod-play-tit', webpage)) or
+                 self._html_search_regex(r'<span[^>]+data-videochanged-title="word"[^>]*>([^<]+)</span>', webpage, 'title'))
  
          return {
              'id': video_id,
  
          return {
              'id': video_id,