Merge tag 'upstream/2014.10.30'

[youtubedl] / youtube_dl / YoutubeDL.py
diff --git a/youtube_dl/YoutubeDL.py b/youtube_dl/YoutubeDL.py

index 14a1d06ab1ed3350547822cac71501745a14842a..73a372df4724e05607acd0fb202ad7e774f680cc 100755 (executable)
--- a/youtube_dl/YoutubeDL.py
+++ b/youtube_dl/YoutubeDL.py
@@ -24,10 +24,12 @@ if os.name == 'nt':
  
  from .utils import (
      compat_cookiejar,
  
  from .utils import (
      compat_cookiejar,
+    compat_expanduser,
      compat_http_client,
      compat_str,
      compat_urllib_error,
      compat_urllib_request,
      compat_http_client,
      compat_str,
      compat_urllib_error,
      compat_urllib_request,
+    escape_url,
      ContentTooShortError,
      date_from_str,
      DateRange,
      ContentTooShortError,
      date_from_str,
      DateRange,
@@ -57,9 +59,10 @@ from .utils import (
      YoutubeDLHandler,
      prepend_extension,
  )
      YoutubeDLHandler,
      prepend_extension,
  )
+from .cache import Cache
  from .extractor import get_info_extractor, gen_extractors
  from .downloader import get_suitable_downloader
  from .extractor import get_info_extractor, gen_extractors
  from .downloader import get_suitable_downloader
-from .postprocessor import FFmpegMergerPP
+from .postprocessor import FFmpegMergerPP, FFmpegPostProcessor
  from .version import __version__
  
  
  from .version import __version__
  
  
@@ -105,6 +108,8 @@ class YoutubeDL(object):
      forcefilename:     Force printing final filename.
      forceduration:     Force printing duration.
      forcejson:         Force printing info_dict as JSON.
      forcefilename:     Force printing final filename.
      forceduration:     Force printing duration.
      forcejson:         Force printing info_dict as JSON.
+    dump_single_json:  Force printing the info_dict of the whole playlist
+                       (or video) as a single JSON line.
      simulate:          Do not download the video files.
      format:            Video format code.
      format_limit:      Highest quality format to try.
      simulate:          Do not download the video files.
      format:            Video format code.
      format_limit:      Highest quality format to try.
@@ -133,7 +138,7 @@ class YoutubeDL(object):
      daterange:         A DateRange object, download only if the upload_date is in the range.
      skip_download:     Skip the actual download of the video file
      cachedir:          Location of the cache files in the filesystem.
      daterange:         A DateRange object, download only if the upload_date is in the range.
      skip_download:     Skip the actual download of the video file
      cachedir:          Location of the cache files in the filesystem.
-                       None to disable filesystem cache.
+                       False to disable filesystem cache.
      noplaylist:        Download single video instead of a playlist if in doubt.
      age_limit:         An integer representing the user's age in years.
                         Unsuitable videos for the given age are skipped.
      noplaylist:        Download single video instead of a playlist if in doubt.
      age_limit:         An integer representing the user's age in years.
                         Unsuitable videos for the given age are skipped.
@@ -162,6 +167,9 @@ class YoutubeDL(object):
      default_search:    Prepend this string if an input url is not valid.
                         'auto' for elaborate guessing
      encoding:          Use this encoding instead of the system-specified.
      default_search:    Prepend this string if an input url is not valid.
                         'auto' for elaborate guessing
      encoding:          Use this encoding instead of the system-specified.
+    extract_flat:      Do not resolve URLs, return the immediate result.
+                       Pass in 'in_playlist' to only show this behavior for
+                       playlist items.
  
      The following parameters are not used by YoutubeDL itself, they are used by
      the FileDownloader:
  
      The following parameters are not used by YoutubeDL itself, they are used by
      the FileDownloader:
@@ -171,6 +179,7 @@ class YoutubeDL(object):
      The following options are used by the post processors:
      prefer_ffmpeg:     If True, use ffmpeg instead of avconv if both are available,
                         otherwise prefer avconv.
      The following options are used by the post processors:
      prefer_ffmpeg:     If True, use ffmpeg instead of avconv if both are available,
                         otherwise prefer avconv.
+    exec_cmd:          Arbitrary command to run after downloading
      """
  
      params = None
      """
  
      params = None
@@ -180,7 +189,7 @@ class YoutubeDL(object):
      _num_downloads = None
      _screen_file = None
  
      _num_downloads = None
      _screen_file = None
  
-    def __init__(self, params=None):
+    def __init__(self, params=None, auto_init=True):
          """Create a FileDownloader object with the given options."""
          if params is None:
              params = {}
          """Create a FileDownloader object with the given options."""
          if params is None:
              params = {}
@@ -193,6 +202,7 @@ class YoutubeDL(object):
          self._screen_file = [sys.stdout, sys.stderr][params.get('logtostderr', False)]
          self._err_file = sys.stderr
          self.params = params
          self._screen_file = [sys.stdout, sys.stderr][params.get('logtostderr', False)]
          self._err_file = sys.stderr
          self.params = params
+        self.cache = Cache(self)
  
          if params.get('bidi_workaround', False):
              try:
  
          if params.get('bidi_workaround', False):
              try:
@@ -223,11 +233,11 @@ class YoutubeDL(object):
  
          if (sys.version_info >= (3,) and sys.platform != 'win32' and
                  sys.getfilesystemencoding() in ['ascii', 'ANSI_X3.4-1968']
  
          if (sys.version_info >= (3,) and sys.platform != 'win32' and
                  sys.getfilesystemencoding() in ['ascii', 'ANSI_X3.4-1968']
-                and not params['restrictfilenames']):
+                and not params.get('restrictfilenames', False)):
              # On Python 3, the Unicode filesystem API will throw errors (#1474)
              self.report_warning(
                  'Assuming --restrict-filenames since file system encoding '
              # On Python 3, the Unicode filesystem API will throw errors (#1474)
              self.report_warning(
                  'Assuming --restrict-filenames since file system encoding '
-                'cannot encode all charactes. '
+                'cannot encode all characters. '
                  'Set the LC_ALL environment variable to fix this.')
              self.params['restrictfilenames'] = True
  
                  'Set the LC_ALL environment variable to fix this.')
              self.params['restrictfilenames'] = True
  
@@ -236,6 +246,10 @@ class YoutubeDL(object):
  
          self._setup_opener()
  
  
          self._setup_opener()
  
+        if auto_init:
+            self.print_debug_header()
+            self.add_default_info_extractors()
+
      def add_info_extractor(self, ie):
          """Add an InfoExtractor object to the end of the list."""
          self._ies.append(ie)
      def add_info_extractor(self, ie):
          """Add an InfoExtractor object to the end of the list."""
          self._ies.append(ie)
@@ -423,7 +437,7 @@ class YoutubeDL(object):
              autonumber_templ = '%0' + str(autonumber_size) + 'd'
              template_dict['autonumber'] = autonumber_templ % self._num_downloads
              if template_dict.get('playlist_index') is not None:
              autonumber_templ = '%0' + str(autonumber_size) + 'd'
              template_dict['autonumber'] = autonumber_templ % self._num_downloads
              if template_dict.get('playlist_index') is not None:
-                template_dict['playlist_index'] = '%05d' % template_dict['playlist_index']
+                template_dict['playlist_index'] = '%0*d' % (len(str(template_dict['n_entries'])), template_dict['playlist_index'])
              if template_dict.get('resolution') is None:
                  if template_dict.get('width') and template_dict.get('height'):
                      template_dict['resolution'] = '%dx%d' % (template_dict['width'], template_dict['height'])
              if template_dict.get('resolution') is None:
                  if template_dict.get('width') and template_dict.get('height'):
                      template_dict['resolution'] = '%dx%d' % (template_dict['width'], template_dict['height'])
@@ -442,7 +456,7 @@ class YoutubeDL(object):
              template_dict = collections.defaultdict(lambda: 'NA', template_dict)
  
              outtmpl = self.params.get('outtmpl', DEFAULT_OUTTMPL)
              template_dict = collections.defaultdict(lambda: 'NA', template_dict)
  
              outtmpl = self.params.get('outtmpl', DEFAULT_OUTTMPL)
-            tmpl = os.path.expanduser(outtmpl)
+            tmpl = compat_expanduser(outtmpl)
              filename = tmpl % template_dict
              return filename
          except ValueError as err:
              filename = tmpl % template_dict
              return filename
          except ValueError as err:
@@ -479,7 +493,10 @@ class YoutubeDL(object):
                  return 'Skipping %s, because it has exceeded the maximum view count (%d/%d)' % (video_title, view_count, max_views)
          age_limit = self.params.get('age_limit')
          if age_limit is not None:
                  return 'Skipping %s, because it has exceeded the maximum view count (%d/%d)' % (video_title, view_count, max_views)
          age_limit = self.params.get('age_limit')
          if age_limit is not None:
-            if age_limit < info_dict.get('age_limit', 0):
+            actual_age_limit = info_dict.get('age_limit')
+            if actual_age_limit is None:
+                actual_age_limit = 0
+            if age_limit < actual_age_limit:
                  return 'Skipping "' + title + '" because it is age restricted'
          if self.in_download_archive(info_dict):
              return '%s has already been recorded in archive' % video_title
                  return 'Skipping "' + title + '" because it is age restricted'
          if self.in_download_archive(info_dict):
              return '%s has already been recorded in archive' % video_title
@@ -558,7 +575,16 @@ class YoutubeDL(object):
          Returns the resolved ie_result.
          """
  
          Returns the resolved ie_result.
          """
  
-        result_type = ie_result.get('_type', 'video') # If not given we suppose it's a video, support the default old system
+        result_type = ie_result.get('_type', 'video')
+
+        if result_type in ('url', 'url_transparent'):
+            extract_flat = self.params.get('extract_flat', False)
+            if ((extract_flat == 'in_playlist' and 'playlist' in extra_info) or
+                    extract_flat is True):
+                if self.params.get('forcejson', False):
+                    self.to_stdout(json.dumps(ie_result))
+                return ie_result
+
          if result_type == 'video':
              self.add_extra_info(ie_result, extra_info)
              return self.process_video_result(ie_result, download=download)
          if result_type == 'video':
              self.add_extra_info(ie_result, extra_info)
              return self.process_video_result(ie_result, download=download)
@@ -627,6 +653,7 @@ class YoutubeDL(object):
              for i, entry in enumerate(entries, 1):
                  self.to_screen('[download] Downloading video #%s of %s' % (i, n_entries))
                  extra = {
              for i, entry in enumerate(entries, 1):
                  self.to_screen('[download] Downloading video #%s of %s' % (i, n_entries))
                  extra = {
+                    'n_entries': n_entries,
                      'playlist': playlist,
                      'playlist_index': i + playliststart,
                      'extractor': ie_result['extractor'],
                      'playlist': playlist,
                      'playlist_index': i + playliststart,
                      'extractor': ie_result['extractor'],
@@ -694,7 +721,7 @@ class YoutubeDL(object):
              if video_formats:
                  return video_formats[0]
          else:
              if video_formats:
                  return video_formats[0]
          else:
-            extensions = ['mp4', 'flv', 'webm', '3gp']
+            extensions = ['mp4', 'flv', 'webm', '3gp', 'm4a']
              if format_spec in extensions:
                  filter_f = lambda f: f['ext'] == format_spec
              else:
              if format_spec in extensions:
                  filter_f = lambda f: f['ext'] == format_spec
              else:
@@ -795,28 +822,29 @@ class YoutubeDL(object):
          if req_format in ('-1', 'all'):
              formats_to_download = formats
          else:
          if req_format in ('-1', 'all'):
              formats_to_download = formats
          else:
-            # We can accept formats requested in the format: 34/5/best, we pick
-            # the first that is available, starting from left
-            req_formats = req_format.split('/')
-            for rf in req_formats:
-                if re.match(r'.+?\+.+?', rf) is not None:
-                    # Two formats have been requested like '137+139'
-                    format_1, format_2 = rf.split('+')
-                    formats_info = (self.select_format(format_1, formats),
-                        self.select_format(format_2, formats))
-                    if all(formats_info):
-                        selected_format = {
-                            'requested_formats': formats_info,
-                            'format': rf,
-                            'ext': formats_info[0]['ext'],
-                        }
+            for rfstr in req_format.split(','):
+                # We can accept formats requested in the format: 34/5/best, we pick
+                # the first that is available, starting from left
+                req_formats = rfstr.split('/')
+                for rf in req_formats:
+                    if re.match(r'.+?\+.+?', rf) is not None:
+                        # Two formats have been requested like '137+139'
+                        format_1, format_2 = rf.split('+')
+                        formats_info = (self.select_format(format_1, formats),
+                            self.select_format(format_2, formats))
+                        if all(formats_info):
+                            selected_format = {
+                                'requested_formats': formats_info,
+                                'format': rf,
+                                'ext': formats_info[0]['ext'],
+                            }
+                        else:
+                            selected_format = None
                      else:
                      else:
-                        selected_format = None
-                else:
-                    selected_format = self.select_format(rf, formats)
-                if selected_format is not None:
-                    formats_to_download = [selected_format]
-                    break
+                        selected_format = self.select_format(rf, formats)
+                    if selected_format is not None:
+                        formats_to_download.append(selected_format)
+                        break
          if not formats_to_download:
              raise ExtractorError('requested format not available',
                                   expected=True)
          if not formats_to_download:
              raise ExtractorError('requested format not available',
                                   expected=True)
@@ -882,6 +910,8 @@ class YoutubeDL(object):
          if self.params.get('forcejson', False):
              info_dict['_filename'] = filename
              self.to_stdout(json.dumps(info_dict))
          if self.params.get('forcejson', False):
              info_dict['_filename'] = filename
              self.to_stdout(json.dumps(info_dict))
+        if self.params.get('dump_single_json', False):
+            info_dict['_filename'] = filename
  
          # Do nothing else if in simulate mode
          if self.params.get('simulate', False):
  
          # Do nothing else if in simulate mode
          if self.params.get('simulate', False):
@@ -1000,7 +1030,7 @@ class YoutubeDL(object):
                          downloaded = []
                          success = True
                          merger = FFmpegMergerPP(self, not self.params.get('keepvideo'))
                          downloaded = []
                          success = True
                          merger = FFmpegMergerPP(self, not self.params.get('keepvideo'))
-                        if not merger._get_executable():
+                        if not merger._executable:
                              postprocessors = []
                              self.report_warning('You have requested multiple '
                                  'formats but ffmpeg or avconv are not installed.'
                              postprocessors = []
                              self.report_warning('You have requested multiple '
                                  'formats but ffmpeg or avconv are not installed.'
@@ -1049,12 +1079,15 @@ class YoutubeDL(object):
          for url in url_list:
              try:
                  #It also downloads the videos
          for url in url_list:
              try:
                  #It also downloads the videos
-                self.extract_info(url)
+                res = self.extract_info(url)
              except UnavailableVideoError:
                  self.report_error('unable to download video')
              except MaxDownloadsReached:
                  self.to_screen('[info] Maximum number of downloaded files reached.')
                  raise
              except UnavailableVideoError:
                  self.report_error('unable to download video')
              except MaxDownloadsReached:
                  self.to_screen('[info] Maximum number of downloaded files reached.')
                  raise
+            else:
+                if self.params.get('dump_single_json', False):
+                    self.to_stdout(json.dumps(res))
  
          return self._download_retcode
  
  
          return self._download_retcode
  
@@ -1178,6 +1211,8 @@ class YoutubeDL(object):
              res += 'video@'
          if fdict.get('vbr') is not None:
              res += '%4dk' % fdict['vbr']
              res += 'video@'
          if fdict.get('vbr') is not None:
              res += '%4dk' % fdict['vbr']
+        if fdict.get('fps') is not None:
+            res += ', %sfps' % fdict['fps']
          if fdict.get('acodec') is not None:
              if res:
                  res += ', '
          if fdict.get('acodec') is not None:
              if res:
                  res += ', '
@@ -1228,6 +1263,26 @@ class YoutubeDL(object):
  
      def urlopen(self, req):
          """ Start an HTTP download """
  
      def urlopen(self, req):
          """ Start an HTTP download """
+
+        # According to RFC 3986, URLs can not contain non-ASCII characters, however this is not
+        # always respected by websites, some tend to give out URLs with non percent-encoded
+        # non-ASCII characters (see telemb.py, ard.py [#3412])
+        # urllib chokes on URLs with non-ASCII characters (see http://bugs.python.org/issue3991)
+        # To work around aforementioned issue we will replace request's original URL with
+        # percent-encoded one
+        req_is_string = isinstance(req, basestring if sys.version_info < (3, 0) else compat_str)
+        url = req if req_is_string else req.get_full_url()
+        url_escaped = escape_url(url)
+
+        # Substitute URL if any change after escaping
+        if url != url_escaped:
+            if req_is_string:
+                req = url_escaped
+            else:
+                req = compat_urllib_request.Request(
+                    url_escaped, data=req.data, headers=req.headers,
+                    origin_req_host=req.origin_req_host, unverifiable=req.unverifiable)
+
          return self._opener.open(req, timeout=self._socket_timeout)
  
      def print_debug_header(self):
          return self._opener.open(req, timeout=self._socket_timeout)
  
      def print_debug_header(self):
@@ -1262,8 +1317,18 @@ class YoutubeDL(object):
                  sys.exc_clear()
              except:
                  pass
                  sys.exc_clear()
              except:
                  pass
-        self._write_string('[debug] Python version %s - %s' %
-                     (platform.python_version(), platform_name()) + '\n')
+        self._write_string('[debug] Python version %s - %s\n' % (
+            platform.python_version(), platform_name()))
+
+        exe_versions = FFmpegPostProcessor.get_versions()
+        exe_str = ', '.join(
+            '%s %s' % (exe, v)
+            for exe, v in sorted(exe_versions.items())
+            if v
+        )
+        if not exe_str:
+            exe_str = 'none'
+        self._write_string('[debug] exe versions: %s\n' % exe_str)
  
          proxy_map = {}
          for handler in self._opener.handlers:
  
          proxy_map = {}
          for handler in self._opener.handlers: