Merge branch 'master' into vimeo

author Rogério Brito <rbrito@ime.usp.br>

Fri, 25 Feb 2011 23:08:12 +0000 (20:08 -0300)

committer Rogério Brito <rbrito@ime.usp.br>

Fri, 25 Feb 2011 23:08:12 +0000 (20:08 -0300)
author Rogério Brito <rbrito@ime.usp.br>
Fri, 25 Feb 2011 23:08:12 +0000 (20:08 -0300)
committer Rogério Brito <rbrito@ime.usp.br>
Fri, 25 Feb 2011 23:08:12 +0000 (20:08 -0300)
diff --combined youtube-dl

index 7823726,072a919..9ce89aa
--- 1/youtube-dl
--- 2/youtube-dl
+++ b/youtube-dl
@@@ -38,7 -38,7 +38,7 @@@ except ImportError
         from cgi import parse_qs
   
   std_headers = {
-       'User-Agent': 'Mozilla/5.0 (X11; Linux x86_64; rv:2.0b10) Gecko/20100101 Firefox/4.0b10',
+       'User-Agent': 'Mozilla/5.0 (X11; Linux x86_64; rv:2.0b11) Gecko/20100101 Firefox/4.0b11',
         'Accept-Charset': 'ISO-8859-1,utf-8;q=0.7,*;q=0.7',
         'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8',
         'Accept-Encoding': 'gzip, deflate',
@@@ -1059,7 -1059,7 +1059,7 @@@ class YoutubeIE(InfoExtractor)
                 mobj = re.search(r'id="eow-date".*?>(.*?)</span>', video_webpage, re.DOTALL)
                 if mobj is not None:
                         upload_date = ' '.join(re.sub(r'[/,-]', r' ', mobj.group(1)).split())
-                       format_expressions = ['%d %B %Y', '%B %d %Y']
+                       format_expressions = ['%d %B %Y', '%B %d %Y', '%b %d %Y']
                         for expression in format_expressions:
                                 try:
                                         upload_date = datetime.datetime.strptime(upload_date, expression).strftime('%Y%m%d')
@@@ -1720,123 -1720,6 +1720,123 @@@ class YahooIE(InfoExtractor)
                         self._downloader.trouble(u'\nERROR: unable to download video')
   
   
+ +class VimeoIE(InfoExtractor):
+ +      """Information extractor for vimeo.com."""
+ +
+ +      # _VALID_URL matches Vimeo URLs
+ +      _VALID_URL = r'(?:http://)?(?:(?:www|player).)?vimeo\.com/(?:video/)?([0-9]+)'
+ +
+ +      def __init__(self, downloader=None):
+ +              InfoExtractor.__init__(self, downloader)
+ +
+ +      @staticmethod
+ +      def suitable(url):
+ +              return (re.match(VimeoIE._VALID_URL, url) is not None)
+ +
+ +      def report_download_webpage(self, video_id):
+ +              """Report webpage download."""
+ +              self._downloader.to_screen(u'[video.vimeo] %s: Downloading webpage' % video_id)
+ +
+ +      def report_extraction(self, video_id):
+ +              """Report information extraction."""
+ +              self._downloader.to_screen(u'[video.vimeo] %s: Extracting information' % video_id)
+ +
+ +      def _real_initialize(self):
+ +              return
+ +
+ +      def _real_extract(self, url, new_video=True):
+ +              # Extract ID from URL
+ +              mobj = re.match(self._VALID_URL, url)
+ +              if mobj is None:
+ +                      self._downloader.trouble(u'ERROR: Invalid URL: %s' % url)
+ +                      return
+ +
+ +              # At this point we have a new video
+ +              self._downloader.increment_downloads()
+ +              video_id = mobj.group(1)
+ +              video_extension = 'flv' # FIXME
+ +
+ +              # Retrieve video webpage to extract further information
+ +              request = urllib2.Request("http://vimeo.com/moogaloop/load/clip:%s" % video_id, None, std_headers)
+ +              try:
+ +                      self.report_download_webpage(video_id)
+ +                      webpage = urllib2.urlopen(request).read()
+ +              except (urllib2.URLError, httplib.HTTPException, socket.error), err:
+ +                      self._downloader.trouble(u'ERROR: Unable to retrieve video webpage: %s' % str(err))
+ +                      return
+ +
+ +              # Now we begin extracting as much information as we can from what we
+ +              # retrieved. First we extract the information common to all extractors,
+ +              # and latter we extract those that are Vimeo specific.
+ +              self.report_extraction(video_id)
+ +
+ +              # Extract title
+ +              mobj = re.search(r'<caption>(.*?)</caption>', webpage)
+ +              if mobj is None:
+ +                      self._downloader.trouble(u'ERROR: unable to extract video title')
+ +                      return
+ +              video_title = mobj.group(1).decode('utf-8')
+ +              simple_title = re.sub(ur'(?u)([^%s]+)' % simple_title_chars, ur'_', video_title)
+ +
+ +              # Extract uploader
+ +              mobj = re.search(r'<uploader_url>http://vimeo.com/(.*?)</uploader_url>', webpage)
+ +              if mobj is None:
+ +                      self._downloader.trouble(u'ERROR: unable to extract video uploader')
+ +                      return
+ +              video_uploader = mobj.group(1).decode('utf-8')
+ +
+ +              # Extract video thumbnail
+ +              mobj = re.search(r'<thumbnail>(.*?)</thumbnail>', webpage)
+ +              if mobj is None:
+ +                      self._downloader.trouble(u'ERROR: unable to extract video thumbnail')
+ +                      return
+ +              video_thumbnail = mobj.group(1).decode('utf-8')
+ +
+ +              # # Extract video description
+ +              # mobj = re.search(r'<meta property="og:description" content="(.*)" />', webpage)
+ +              # if mobj is None:
+ +              #       self._downloader.trouble(u'ERROR: unable to extract video description')
+ +              #       return
+ +              # video_description = mobj.group(1).decode('utf-8')
+ +              # if not video_description: video_description = 'No description available.'
+ +              video_description = 'Foo.'
+ +
+ +              # Vimeo specific: extract request signature
+ +              mobj = re.search(r'<request_signature>(.*?)</request_signature>', webpage)
+ +              if mobj is None:
+ +                      self._downloader.trouble(u'ERROR: unable to extract request signature')
+ +                      return
+ +              sig = mobj.group(1).decode('utf-8')
+ +
+ +              # Vimeo specific: Extract request signature expiration
+ +              mobj = re.search(r'<request_signature_expires>(.*?)</request_signature_expires>', webpage)
+ +              if mobj is None:
+ +                      self._downloader.trouble(u'ERROR: unable to extract request signature expiration')
+ +                      return
+ +              sig_exp = mobj.group(1).decode('utf-8')
+ +
+ +              video_url = "http://vimeo.com/moogaloop/play/clip:%s/%s/%s" % (video_id, sig, sig_exp)
+ +
+ +              try:
+ +                      # Process video information
+ +                      self._downloader.process_info({
+ +                              'id':           video_id.decode('utf-8'),
+ +                              'url':          video_url,
+ +                              'uploader':     video_uploader,
+ +                              'upload_date':  u'NA',
+ +                              'title':        video_title,
+ +                              'stitle':       simple_title,
+ +                              'ext':          video_extension.decode('utf-8'),
+ +                              'thumbnail':    video_thumbnail.decode('utf-8'),
+ +                              'description':  video_description,
+ +                              'thumbnail':    video_thumbnail,
+ +                              'description':  video_description,
+ +                              'player_url':   None,
+ +                      })
+ +              except UnavailableVideoError:
+ +                      self._downloader.trouble(u'ERROR: unable to download video')
+ +
+ +
   class GenericIE(InfoExtractor):
         """Generic last-resort information extractor."""
   
@@@ -2726,6 -2609,88 +2726,88 @@@ class PostProcessor(object)
                 """
                 return information # by default, do nothing
   
+ class FFmpegExtractAudioPP(PostProcessor):
+ 
+       def __init__(self, downloader=None, preferredcodec=None):
+               PostProcessor.__init__(self, downloader)
+               if preferredcodec is None:
+                       preferredcodec = 'best'
+               self._preferredcodec = preferredcodec
+ 
+       @staticmethod
+       def get_audio_codec(path):
+               try:
+                       handle = subprocess.Popen(['ffprobe', '-show_streams', path],
+                                       stderr=file(os.path.devnull, 'w'), stdout=subprocess.PIPE)
+                       output = handle.communicate()[0]
+                       if handle.wait() != 0:
+                               return None
+               except (IOError, OSError):
+                       return None
+               audio_codec = None
+               for line in output.split('\n'):
+                       if line.startswith('codec_name='):
+                               audio_codec = line.split('=')[1].strip()
+                       elif line.strip() == 'codec_type=audio' and audio_codec is not None:
+                               return audio_codec
+               return None
+ 
+       @staticmethod
+       def run_ffmpeg(path, out_path, codec, more_opts):
+               try:
+                       ret = subprocess.call(['ffmpeg', '-y', '-i', path, '-vn', '-acodec', codec] + more_opts + [out_path],
+                                       stdout=file(os.path.devnull, 'w'), stderr=subprocess.STDOUT)
+                       return (ret == 0)
+               except (IOError, OSError):
+                       return False
+ 
+       def run(self, information):
+               path = information['filepath']
+ 
+               filecodec = self.get_audio_codec(path)
+               if filecodec is None:
+                       self._downloader.to_stderr(u'WARNING: unable to obtain file audio codec with ffprobe')
+                       return None
+ 
+               more_opts = []
+               if self._preferredcodec == 'best' or self._preferredcodec == filecodec:
+                       if filecodec == 'aac' or filecodec == 'mp3':
+                               # Lossless if possible
+                               acodec = 'copy'
+                               extension = filecodec
+                               if filecodec == 'aac':
+                                       more_opts = ['-f', 'adts']
+                       else:
+                               # MP3 otherwise.
+                               acodec = 'libmp3lame'
+                               extension = 'mp3'
+                               more_opts = ['-ab', '128k']
+               else:
+                       # We convert the audio (lossy)
+                       acodec = {'mp3': 'libmp3lame', 'aac': 'aac'}[self._preferredcodec]
+                       extension = self._preferredcodec
+                       more_opts = ['-ab', '128k']
+                       if self._preferredcodec == 'aac':
+                               more_opts += ['-f', 'adts']
+ 
+               (prefix, ext) = os.path.splitext(path)
+               new_path = prefix + '.' + extension
+               self._downloader.to_screen(u'[ffmpeg] Destination: %s' % new_path)
+               status = self.run_ffmpeg(path, new_path, acodec, more_opts)
+ 
+               if not status:
+                       self._downloader.to_stderr(u'WARNING: error running ffmpeg')
+                       return None
+ 
+               try:
+                       os.remove(path)
+               except (IOError, OSError):
+                       self._downloader.to_stderr(u'WARNING: Unable to remove downloaded video file')
+                       return None
+ 
+               information['filepath'] = new_path
+               return information
+ 
   ### MAIN PROGRAM ###
   if __name__ == '__main__':
         try:
@@@ -2758,7 -2723,7 +2840,7 @@@
                 # Parse command line
                 parser = optparse.OptionParser(
                         usage='Usage: %prog [options] url...',
-                       version='2011.01.30',
+                       version='2011.02.25b',
                         conflict_handler='resolve',
                 )
   
@@@ -2850,6 -2815,13 +2932,13 @@@
                                 help='do not use the Last-modified header to set the file modification time', default=True)
                 parser.add_option_group(filesystem)
   
+               postproc = optparse.OptionGroup(parser, 'Post-processing Options')
+               postproc.add_option('--extract-audio', action='store_true', dest='extractaudio', default=False,
+                               help='convert video files to audio-only files (requires ffmpeg and ffprobe)')
+               postproc.add_option('--audio-format', metavar='FORMAT', dest='audioformat', default='best',
+                               help='"best", "aac" or "mp3"; best by default')
+               parser.add_option_group(postproc)
+ 
                 (opts, args) = parser.parse_args()
   
                 # Open appropriate CookieJar
@@@ -2921,9 -2893,11 +3010,12 @@@
                                 raise ValueError
                 except (TypeError, ValueError), err:
                         parser.error(u'invalid playlist end number specified')
+               if opts.extractaudio:
+                       if opts.audioformat not in ['best', 'aac', 'mp3']:
+                               parser.error(u'invalid audio format specified')
   
                 # Information extractors
+ +              vimeo_ie = VimeoIE()
                 youtube_ie = YoutubeIE()
                 metacafe_ie = MetacafeIE(youtube_ie)
                 dailymotion_ie = DailymotionIE()
@@@ -2976,7 -2950,6 +3068,7 @@@
                         'nopart': opts.nopart,
                         'updatetime': opts.updatetime,
                         })
+ +              fd.add_info_extractor(vimeo_ie)
                 fd.add_info_extractor(youtube_search_ie)
                 fd.add_info_extractor(youtube_pl_ie)
                 fd.add_info_extractor(youtube_user_ie)
@@@ -2995,6 -2968,10 +3087,10 @@@
                 # fallback if none of the others work
                 fd.add_info_extractor(generic_ie)
   
+               # PostProcessors
+               if opts.extractaudio:
+                       fd.add_post_processor(FFmpegExtractAudioPP(preferredcodec=opts.audioformat))
+ 
                 # Update version
                 if opts.update_self:
                         update_self(fd, sys.argv[0])
author	Rogério Brito <rbrito@ime.usp.br>
	Fri, 25 Feb 2011 23:08:12 +0000 (20:08 -0300)
committer	Rogério Brito <rbrito@ime.usp.br>
	Fri, 25 Feb 2011 23:08:12 +0000 (20:08 -0300)