Refactor subtitle options from srt to the more generic 'sub'.

[youtube-dl.git] / youtube_dl / FileDownloader.py
diff --git a/youtube_dl/FileDownloader.py b/youtube_dl/FileDownloader.py

index 04ecd1ac5447cd306cc7d45bd67449e6f9bb66a4..4549dd46486ba205ba9bf853987ac4fde9ab437a 100644 (file)
--- a/youtube_dl/FileDownloader.py
+++ b/youtube_dl/FileDownloader.py
@@ -78,10 +78,14 @@ class FileDownloader(object):
      updatetime:        Use the Last-modified header to set output file timestamps.
      writedescription:  Write the video description to a .description file
      writeinfojson:     Write the video description to a .info.json file
-    writesubtitles:    Write the video subtitles to a .srt file
+    writesubtitles:    Write the video subtitles to a file (default=srt)
+    onlysubtitles:     Downloads only the subtitles of the video
+    allsubtitles:      Downloads all the subtitles of the video
      subtitleslang:     Language of the subtitles to download
      test:              Download only first bytes to test the downloader.
      keepvideo:         Keep the video file after post-processing
+    min_filesize:      Skip files smaller than this size
+    max_filesize:      Skip files larger than this size
      """
  
      params = None
@@ -95,6 +99,7 @@ class FileDownloader(object):
          """Create a FileDownloader object with the given options."""
          self._ies = []
          self._pps = []
+        self._progress_hooks = []
          self._download_retcode = 0
          self._num_downloads = 0
          self._screen_file = [sys.stdout, sys.stderr][params.get('logtostderr', False)]
@@ -205,7 +210,7 @@ class FileDownloader(object):
              # already of type unicode()
              ctypes.windll.kernel32.SetConsoleTitleW(ctypes.c_wchar_p(message))
          elif 'TERM' in os.environ:
-            sys.stderr.write('\033]0;%s\007' % message.encode(preferredencoding()))
+            self.to_screen('\033]0;%s\007' % message, skip_eol=True)
  
      def fixed_template(self):
          """Checks if the output template is fixed."""
@@ -286,9 +291,9 @@ class FileDownloader(object):
          """ Report that the description file is being written """
          self.to_screen(u'[info] Writing video description to: ' + descfn)
  
-    def report_writesubtitles(self, srtfn):
+    def report_writesubtitles(self, sub_filename):
          """ Report that the subtitles file is being written """
-        self.to_screen(u'[info] Writing video subtitles to: ' + srtfn)
+        self.to_screen(u'[info] Writing video subtitles to: ' + sub_filename)
  
      def report_writeinfojson(self, infofn):
          """ Report that the metadata file has been written """
@@ -302,7 +307,11 @@ class FileDownloader(object):
          """Report download progress."""
          if self.params.get('noprogress', False):
              return
-        self.to_screen(u'\r[download] %s of %s at %s ETA %s' %
+        if self.params.get('progress_with_newline', False):
+            self.to_screen(u'[download] %s of %s at %s ETA %s' %
+                (percent_str, data_len_str, speed_str, eta_str))
+        else:
+            self.to_screen(u'\r[download] %s of %s at %s ETA %s' %
                  (percent_str, data_len_str, speed_str, eta_str), skip_eol=True)
          self.to_cons_title(u'youtube-dl - %s of %s at %s ETA %s' %
                  (percent_str.strip(), data_len_str.strip(), speed_str.strip(), eta_str.strip()))
@@ -363,12 +372,10 @@ class FileDownloader(object):
          title = info_dict['title']
          matchtitle = self.params.get('matchtitle', False)
          if matchtitle:
-            matchtitle = matchtitle.decode('utf8')
              if not re.search(matchtitle, title, re.IGNORECASE):
                  return u'[download] "' + title + '" title did not match pattern "' + matchtitle + '"'
          rejecttitle = self.params.get('rejecttitle', False)
          if rejecttitle:
-            rejecttitle = rejecttitle.decode('utf8')
              if re.search(rejecttitle, title, re.IGNORECASE):
                  return u'"' + title + '" title matched reject pattern "' + rejecttitle + '"'
          return None
@@ -436,14 +443,33 @@ class FileDownloader(object):
          if self.params.get('writesubtitles', False) and 'subtitles' in info_dict and info_dict['subtitles']:
              # subtitles download errors are already managed as troubles in relevant IE
              # that way it will silently go on when used with unsupporting IE
+            subtitle = info_dict['subtitles'][0]
+            (sub_error, sub_lang, sub) = subtitle
              try:
-                srtfn = filename.rsplit('.', 1)[0] + u'.srt'
-                self.report_writesubtitles(srtfn)
-                with io.open(encodeFilename(srtfn), 'w', encoding='utf-8') as srtfile:
-                    srtfile.write(info_dict['subtitles'])
+                sub_filename = filename.rsplit('.', 1)[0] + u'.' + sub_lang + u'.srt'
+                self.report_writesubtitles(sub_filename)
+                with io.open(encodeFilename(sub_filename), 'w', encoding='utf-8') as subfile:
+                    subfile.write(sub)
              except (OSError, IOError):
                  self.trouble(u'ERROR: Cannot write subtitles file ' + descfn)
                  return
+            if self.params.get('onlysubtitles', False):
+                return 
+
+        if self.params.get('allsubtitles', False) and 'subtitles' in info_dict and info_dict['subtitles']:
+            subtitles = info_dict['subtitles']
+            for subtitle in subtitles:
+                (sub_error, sub_lang, sub) = subtitle
+                try:
+                    sub_filename = filename.rsplit('.', 1)[0] + u'.' + sub_lang + u'.srt'
+                    self.report_writesubtitles(sub_filename)
+                    with io.open(encodeFilename(sub_filename), 'w', encoding='utf-8') as subfile:
+                            subfile.write(sub)
+                except (OSError, IOError):
+                    self.trouble(u'ERROR: Cannot write subtitles file ' + descfn)
+                    return
+            if self.params.get('onlysubtitles', False):
+                return 
  
          if self.params.get('writeinfojson', False):
              infofn = filename + u'.info.json'
@@ -594,8 +620,15 @@ class FileDownloader(object):
                  retval = 0
                  break
          if retval == 0:
-            self.to_screen(u'\r[rtmpdump] %s bytes' % os.path.getsize(encodeFilename(tmpfilename)))
+            fsize = os.path.getsize(encodeFilename(tmpfilename))
+            self.to_screen(u'\r[rtmpdump] %s bytes' % fsize)
              self.try_rename(tmpfilename, filename)
+            self._hook_progress({
+                'downloaded_bytes': fsize,
+                'total_bytes': fsize,
+                'filename': filename,
+                'status': 'finished',
+            })
              return True
          else:
              self.trouble(u'\nERROR: rtmpdump exited with code %d' % retval)
@@ -607,6 +640,10 @@ class FileDownloader(object):
          # Check file already present
          if self.params.get('continuedl', False) and os.path.isfile(encodeFilename(filename)) and not self.params.get('nopart', False):
              self.report_file_already_downloaded(filename)
+            self._hook_progress({
+                'filename': filename,
+                'status': 'finished',
+            })
              return True
  
          # Attempt to download using rtmpdump
@@ -678,6 +715,10 @@ class FileDownloader(object):
                              # the one in the hard drive.
                              self.report_file_already_downloaded(filename)
                              self.try_rename(tmpfilename, filename)
+                            self._hook_progress({
+                                'filename': filename,
+                                'status': 'finished',
+                            })
                              return True
                          else:
                              # The length does not match, we start the download over
@@ -696,6 +737,15 @@ class FileDownloader(object):
          data_len = data.info().get('Content-length', None)
          if data_len is not None:
              data_len = int(data_len) + resume_len
+            min_data_len = self.params.get("min_filesize", None)
+            max_data_len =  self.params.get("max_filesize", None)
+            if min_data_len is not None and data_len < min_data_len:
+                self.to_screen(u'\r[download] File is smaller than min-filesize (%s bytes < %s bytes). Aborting.' % (data_len, min_data_len))
+                return False
+            if max_data_len is not None and data_len > max_data_len:
+                self.to_screen(u'\r[download] File is larger than max-filesize (%s bytes > %s bytes). Aborting.' % (data_len, max_data_len))
+                return False
+
          data_len_str = self.format_bytes(data_len)
          byte_counter = 0 + resume_len
          block_size = self.params.get('buffersize', 1024)
@@ -736,6 +786,14 @@ class FileDownloader(object):
                  eta_str = self.calc_eta(start, time.time(), data_len - resume_len, byte_counter - resume_len)
                  self.report_progress(percent_str, data_len_str, speed_str, eta_str)
  
+            self._hook_progress({
+                'downloaded_bytes': byte_counter,
+                'total_bytes': data_len,
+                'tmpfilename': tmpfilename,
+                'filename': filename,
+                'status': 'downloading',
+            })
+
              # Apply rate limit
              self.slow_down(start, byte_counter - resume_len)
  
@@ -752,4 +810,31 @@ class FileDownloader(object):
          if self.params.get('updatetime', True):
              info_dict['filetime'] = self.try_utime(filename, data.info().get('last-modified', None))
  
+        self._hook_progress({
+            'downloaded_bytes': byte_counter,
+            'total_bytes': byte_counter,
+            'filename': filename,
+            'status': 'finished',
+        })
+
          return True
+
+    def _hook_progress(self, status):
+        for ph in self._progress_hooks:
+            ph(status)
+
+    def add_progress_hook(self, ph):
+        """ ph gets called on download progress, with a dictionary with the entries
+        * filename: The final filename
+        * status: One of "downloading" and "finished"
+
+        It can also have some of the following entries:
+
+        * downloaded_bytes: Bytes on disks
+        * total_bytes: Total bytes, None if unknown
+        * tmpfilename: The filename we're currently writing to
+
+        Hooks are guaranteed to be called at least once (with status "finished")
+        if the download is successful.
+        """
+        self._progress_hooks.append(ph)