release 2014.01.06

Merge remote-tracking branch 'origin/master'
[vimeo] Add support for review pages
2025-08-02 18:39:51 -05:00 · 2014-01-06 17:37:24 +01:00 · 2014-01-06 17:37:20 +01:00 · 2014-01-06 17:34:23 +01:00 · 2014-01-06 17:15:27 +01:00 · 2014-01-06 12:54:01 +01:00
18 changed files with 347 additions and 184 deletions
--- a/README.md
+++ b/README.md
@@ -40,6 +40,7 @@ which means you can modify it, redistribute it or use it however you like.
                               files (for videos with obfuscated signatures) are
                               cached, but that may change.
    --no-cache-dir             Disable filesystem caching
+    --socket-timeout None      Time to wait before giving up, in seconds
    --bidi-workaround          Work around terminals that lack bidirectional
                               text support. Requires bidiv or fribidi
                               executable in PATH
--- a/devscripts/bash-completion.in
+++ b/devscripts/bash-completion.in
@@ -6,7 +6,7 @@ __youtube_dl()
    prev="${COMP_WORDS[COMP_CWORD-1]}"
    opts="{{flags}}"
    keywords=":ytfavorites :ytrecommended :ytsubscriptions :ytwatchlater :ythistory"
-    fileopts="-a|--batch-file|--download-archive|--cookies"
+    fileopts="-a|--batch-file|--download-archive|--cookies|--load-info"
    diropts="--cache-dir"

    if [[ ${prev} =~ ${fileopts} ]]; then
--- a/devscripts/gh-pages/update-feed.py
+++ b/devscripts/gh-pages/update-feed.py
@@ -1,56 +1,76 @@
 #!/usr/bin/env python3

 import datetime
-
+import io
+import json
 import textwrap

-import json

-atom_template=textwrap.dedent("""\
-								<?xml version='1.0' encoding='utf-8'?>
-								<atom:feed xmlns:atom="http://www.w3.org/2005/Atom">
-									<atom:title>youtube-dl releases</atom:title>
-									<atom:id>youtube-dl-updates-feed</atom:id>
-									<atom:updated>@TIMESTAMP@</atom:updated>
-									@ENTRIES@
-								</atom:feed>""")
+atom_template = textwrap.dedent("""\
+    <?xml version="1.0" encoding="utf-8"?>
+    <feed xmlns="http://www.w3.org/2005/Atom">
+        <link rel="self" href="http://rg3.github.io/youtube-dl/update/releases.atom" />
+        <title>youtube-dl releases</title>
+        <id>https://yt-dl.org/feed/youtube-dl-updates-feed</id>
+        <updated>@TIMESTAMP@</updated>
+        @ENTRIES@
+    </feed>""")

-entry_template=textwrap.dedent("""
-								<atom:entry>
-									<atom:id>youtube-dl-@VERSION@</atom:id>
-									<atom:title>New version @VERSION@</atom:title>
-									<atom:link href="http://rg3.github.io/youtube-dl" />
-									<atom:content type="xhtml">
-										<div xmlns="http://www.w3.org/1999/xhtml">
-											Downloads available at <a href="https://yt-dl.org/downloads/@VERSION@/">https://yt-dl.org/downloads/@VERSION@/</a>
-										</div>
-									</atom:content>
-									<atom:author>
-										<atom:name>The youtube-dl maintainers</atom:name>
-									</atom:author>
-									<atom:updated>@TIMESTAMP@</atom:updated>
-								</atom:entry>
-								""")
+entry_template = textwrap.dedent("""
+    <entry>
+        <id>https://yt-dl.org/feed/youtube-dl-updates-feed/youtube-dl-@VERSION@</id>
+        <title>New version @VERSION@</title>
+        <link href="http://rg3.github.io/youtube-dl" />
+        <content type="xhtml">
+            <div xmlns="http://www.w3.org/1999/xhtml">
+                Downloads available at <a href="https://yt-dl.org/downloads/@VERSION@/">https://yt-dl.org/downloads/@VERSION@/</a>
+            </div>
+        </content>
+        <author>
+            <name>The youtube-dl maintainers</name>
+        </author>
+        <updated>@TIMESTAMP@</updated>
+    </entry>
+    """)

 now = datetime.datetime.now()
-now_iso = now.isoformat()
+now_iso = now.isoformat() + 'Z'

-atom_template = atom_template.replace('@TIMESTAMP@',now_iso)
-
-entries=[]
+atom_template = atom_template.replace('@TIMESTAMP@', now_iso)

 versions_info = json.load(open('update/versions.json'))
 versions = list(versions_info['versions'].keys())
 versions.sort()

+entries = []
 for v in versions:
-	entry = entry_template.replace('@TIMESTAMP@',v.replace('.','-'))
-	entry = entry.replace('@VERSION@',v)
-	entries.append(entry)
+    fields = v.split('.')
+    year, month, day = map(int, fields[:3])
+    faked = 0
+    patchlevel = 0
+    while True:
+        try:
+            datetime.date(year, month, day)
+        except ValueError:
+            day -= 1
+            faked += 1
+            assert day > 0
+            continue
+        break
+    if len(fields) >= 4:
+        try:
+            patchlevel = int(fields[3])
+        except ValueError:
+            patchlevel = 1
+    timestamp = '%04d-%02d-%02dT00:%02d:%02dZ' % (year, month, day, faked, patchlevel)
+
+    entry = entry_template.replace('@TIMESTAMP@', timestamp)
+    entry = entry.replace('@VERSION@', v)
+    entries.append(entry)

 entries_str = textwrap.indent(''.join(entries), '\t')
 atom_template = atom_template.replace('@ENTRIES@', entries_str)

-with open('update/releases.atom','w',encoding='utf-8') as atom_file:
-	atom_file.write(atom_template)
+with io.open('update/releases.atom', 'w', encoding='utf-8') as atom_file:
+    atom_file.write(atom_template)

--- a/devscripts/make_readme.py
+++ b/devscripts/make_readme.py
@@ -1,20 +1,24 @@
+import io
 import sys
 import re

 README_FILE = 'README.md'
 helptext = sys.stdin.read()

-with open(README_FILE) as f:
+if isinstance(helptext, bytes):
+    helptext = helptext.decode('utf-8')
+
+with io.open(README_FILE, encoding='utf-8') as f:
    oldreadme = f.read()

 header = oldreadme[:oldreadme.index('# OPTIONS')]
 footer = oldreadme[oldreadme.index('# CONFIGURATION'):]

-options = helptext[helptext.index('  General Options:')+19:]
+options = helptext[helptext.index('  General Options:') + 19:]
 options = re.sub(r'^  (\w.+)$', r'## \1', options, flags=re.M)
 options = '# OPTIONS\n' + options + '\n'

-with open(README_FILE, 'w') as f:
+with io.open(README_FILE, 'w', encoding='utf-8') as f:
    f.write(header)
    f.write(options)
    f.write(footer)
--- a/setup.py
+++ b/setup.py
@@ -1,7 +1,7 @@
 #!/usr/bin/env python
 # -*- coding: utf-8 -*-

-from __future__ import print_function, unicode_literals
+from __future__ import print_function

 import pkg_resources
 import sys
--- a/test/test_unicode_literals.py
+++ b/test/test_unicode_literals.py
@@ -7,6 +7,10 @@ import unittest

 rootDir = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))

+IGNORED_FILES = [
+    'setup.py',  # http://bugs.python.org/issue13943
+]
+

 class TestUnicodeLiterals(unittest.TestCase):
    def test_all_files(self):
@@ -17,6 +21,9 @@ class TestUnicodeLiterals(unittest.TestCase):
            for basename in filenames:
                if not basename.endswith('.py'):
                    continue
+                if basename in IGNORED_FILES:
+                    continue
+
                fn = os.path.join(dirpath, basename)
                with io.open(fn, encoding='utf-8') as inf:
                    code = inf.read()
--- a/youtube_dl/init.py
+++ b/youtube_dl/init.py
@@ -192,7 +192,7 @@ def parseOpts(overrideArguments=None):
        help='Disable filesystem caching')
    general.add_option(
        '--socket-timeout', dest='socket_timeout',
-        type=float, default=None, help=optparse.SUPPRESS_HELP)
+        type=float, default=None, help=u'Time to wait before giving up, in seconds')
    general.add_option(
        '--bidi-workaround', dest='bidi_workaround', action='store_true',
        help=u'Work around terminals that lack bidirectional text support. Requires bidiv or fribidi executable in PATH')
@@ -522,6 +522,8 @@ def _real_main(argv=None):
            sys.exit(u'ERROR: batch file could not be read')
    all_urls = batchurls + args
    all_urls = [url.strip() for url in all_urls]
+    _enc = preferredencoding()
+    all_urls = [url.decode(_enc, 'ignore') if isinstance(url, bytes) else url for url in all_urls]

    extractors = gen_extractors()

--- a/youtube_dl/extractor/init.py
+++ b/youtube_dl/extractor/init.py
@@ -199,6 +199,7 @@ from .vimeo import (
    VimeoUserIE,
    VimeoAlbumIE,
    VimeoGroupsIE,
+    VimeoReviewIE,
 )
 from .vine import VineIE
 from .viki import VikiIE
--- a/youtube_dl/extractor/bliptv.py
+++ b/youtube_dl/extractor/bliptv.py
@@ -8,10 +8,8 @@ import socket
 from .common import InfoExtractor
 from ..utils import (
    compat_http_client,
-    compat_parse_qs,
    compat_str,
    compat_urllib_error,
-    compat_urllib_parse_urlparse,
    compat_urllib_request,

    ExtractorError,
--- a/youtube_dl/extractor/common.py
+++ b/youtube_dl/extractor/common.py
@@ -73,6 +73,10 @@ class InfoExtractor(object):
                                 by this field.
                                 -1 for default (order by other properties),
                                 -2 or smaller for less than default.
+                    * quality    Order number of the video quality of this
+                                 format, irrespective of the file format.
+                                 -1 for default (order by other properties),
+                                 -2 or smaller for less than default.
    url:            Final video URL.
    ext:            Video filename extension.
    format:         The video format, defaults to ext (used for --get-format)
@@ -483,6 +487,7 @@ class InfoExtractor(object):

            return (
                preference,
+                f.get('quality') if f.get('quality') is not None else -1,
                f.get('height') if f.get('height') is not None else -1,
                f.get('width') if f.get('width') is not None else -1,
                ext_preference,
--- a/youtube_dl/extractor/generic.py
+++ b/youtube_dl/extractor/generic.py
@@ -1,9 +1,12 @@
 # encoding: utf-8

+from __future__ import unicode_literals
+
 import os
 import re

 from .common import InfoExtractor
+from .youtube import YoutubeIE
 from ..utils import (
    compat_urllib_error,
    compat_urllib_parse,
@@ -22,78 +25,78 @@ from .ooyala import OoyalaIE


 class GenericIE(InfoExtractor):
-    IE_DESC = u'Generic downloader that works on some sites'
+    IE_DESC = 'Generic downloader that works on some sites'
    _VALID_URL = r'.*'
-    IE_NAME = u'generic'
+    IE_NAME = 'generic'
    _TESTS = [
        {
-            u'url': u'http://www.hodiho.fr/2013/02/regis-plante-sa-jeep.html',
-            u'file': u'13601338388002.mp4',
-            u'md5': u'6e15c93721d7ec9e9ca3fdbf07982cfd',
-            u'info_dict': {
-                u"uploader": u"www.hodiho.fr",
-                u"title": u"R\u00e9gis plante sa Jeep"
+            'url': 'http://www.hodiho.fr/2013/02/regis-plante-sa-jeep.html',
+            'file': '13601338388002.mp4',
+            'md5': '6e15c93721d7ec9e9ca3fdbf07982cfd',
+            'info_dict': {
+                'uploader': 'www.hodiho.fr',
+                'title': 'R\u00e9gis plante sa Jeep',
            }
        },
        # embedded vimeo video
        {
-            u'add_ie': ['Vimeo'],
-            u'url': u'http://skillsmatter.com/podcast/home/move-semanticsperfect-forwarding-and-rvalue-references',
-            u'file': u'22444065.mp4',
-            u'md5': u'2903896e23df39722c33f015af0666e2',
-            u'info_dict': {
-                u'title': u'ACCU 2011: Move Semantics,Perfect Forwarding, and Rvalue references- Scott Meyers- 13/04/2011',
-                u"uploader_id": u"skillsmatter",
-                u"uploader": u"Skills Matter",
+            'add_ie': ['Vimeo'],
+            'url': 'http://skillsmatter.com/podcast/home/move-semanticsperfect-forwarding-and-rvalue-references',
+            'file': '22444065.mp4',
+            'md5': '2903896e23df39722c33f015af0666e2',
+            'info_dict': {
+                'title': 'ACCU 2011: Move Semantics,Perfect Forwarding, and Rvalue references- Scott Meyers- 13/04/2011',
+                'uploader_id': 'skillsmatter',
+                'uploader': 'Skills Matter',
            }
        },
        # bandcamp page with custom domain
        {
-            u'add_ie': ['Bandcamp'],
-            u'url': u'http://bronyrock.com/track/the-pony-mash',
-            u'file': u'3235767654.mp3',
-            u'info_dict': {
-                u'title': u'The Pony Mash',
-                u'uploader': u'M_Pallante',
+            'add_ie': ['Bandcamp'],
+            'url': 'http://bronyrock.com/track/the-pony-mash',
+            'file': '3235767654.mp3',
+            'info_dict': {
+                'title': 'The Pony Mash',
+                'uploader': 'M_Pallante',
            },
-            u'skip': u'There is a limit of 200 free downloads / month for the test song',
+            'skip': 'There is a limit of 200 free downloads / month for the test song',
        },
        # embedded brightcove video
        # it also tests brightcove videos that need to set the 'Referer' in the
        # http requests
        {
-            u'add_ie': ['Brightcove'],
-            u'url': u'http://www.bfmtv.com/video/bfmbusiness/cours-bourse/cours-bourse-l-analyse-technique-154522/',
-            u'info_dict': {
-                u'id': u'2765128793001',
-                u'ext': u'mp4',
-                u'title': u'Le cours de bourse : l’analyse technique',
-                u'description': u'md5:7e9ad046e968cb2d1114004aba466fd9',
-                u'uploader': u'BFM BUSINESS',
+            'add_ie': ['Brightcove'],
+            'url': 'http://www.bfmtv.com/video/bfmbusiness/cours-bourse/cours-bourse-l-analyse-technique-154522/',
+            'info_dict': {
+                'id': '2765128793001',
+                'ext': 'mp4',
+                'title': 'Le cours de bourse : l’analyse technique',
+                'description': 'md5:7e9ad046e968cb2d1114004aba466fd9',
+                'uploader': 'BFM BUSINESS',
            },
-            u'params': {
-                u'skip_download': True,
+            'params': {
+                'skip_download': True,
            },
        },
        # Direct link to a video
        {
-            u'url': u'http://media.w3.org/2010/05/sintel/trailer.mp4',
-            u'file': u'trailer.mp4',
-            u'md5': u'67d406c2bcb6af27fa886f31aa934bbe',
-            u'info_dict': {
-                u'id': u'trailer',
-                u'title': u'trailer',
-                u'upload_date': u'20100513',
+            'url': 'http://media.w3.org/2010/05/sintel/trailer.mp4',
+            'file': 'trailer.mp4',
+            'md5': '67d406c2bcb6af27fa886f31aa934bbe',
+            'info_dict': {
+                'id': 'trailer',
+                'title': 'trailer',
+                'upload_date': '20100513',
            }
        },
        # ooyala video
        {
-            u'url': u'http://www.rollingstone.com/music/videos/norwegian-dj-cashmere-cat-goes-spartan-on-with-me-premiere-20131219',
-            u'md5': u'5644c6ca5d5782c1d0d350dad9bd840c',
-            u'info_dict': {
-                u'id': u'BwY2RxaTrTkslxOfcan0UCf0YqyvWysJ',
-                u'ext': u'mp4',
-                u'title': u'2cc213299525360.mov', #that's what we get
+            'url': 'http://www.rollingstone.com/music/videos/norwegian-dj-cashmere-cat-goes-spartan-on-with-me-premiere-20131219',
+            'md5': '5644c6ca5d5782c1d0d350dad9bd840c',
+            'info_dict': {
+                'id': 'BwY2RxaTrTkslxOfcan0UCf0YqyvWysJ',
+                'ext': 'mp4',
+                'title': '2cc213299525360.mov', #that's what we get
            },
        },
    ]
@@ -101,12 +104,12 @@ class GenericIE(InfoExtractor):
    def report_download_webpage(self, video_id):
        """Report webpage download."""
        if not self._downloader.params.get('test', False):
-            self._downloader.report_warning(u'Falling back on generic information extractor.')
+            self._downloader.report_warning('Falling back on generic information extractor.')
        super(GenericIE, self).report_download_webpage(video_id)

    def report_following_redirect(self, new_url):
        """Report information extraction."""
-        self._downloader.to_screen(u'[redirect] Following redirect to %s' % new_url)
+        self._downloader.to_screen('[redirect] Following redirect to %s' % new_url)

    def _send_head(self, url):
        """Check if it is a redirect, like url shorteners, in case return the new url."""
@@ -152,7 +155,7 @@ class GenericIE(InfoExtractor):

        response = opener.open(HEADRequest(url))
        if response is None:
-            raise ExtractorError(u'Invalid URL protocol')
+            raise ExtractorError('Invalid URL protocol')
        return response

    def _real_extract(self, url):
@@ -162,7 +165,7 @@ class GenericIE(InfoExtractor):
            return self.url_result('http://' + url)
        video_id = os.path.splitext(url.split('/')[-1])[0]

-        self.to_screen(u'%s: Requesting header' % video_id)
+        self.to_screen('%s: Requesting header' % video_id)

        try:
            response = self._send_head(url)
@@ -186,7 +189,7 @@ class GenericIE(InfoExtractor):
                    'formats': [{
                        'format_id': m.group('format_id'),
                        'url': url,
-                        'vcodec': u'none' if m.group('type') == 'audio' else None
+                        'vcodec': 'none' if m.group('type') == 'audio' else None
                    }],
                    'upload_date': upload_date,
                }
@@ -200,7 +203,7 @@ class GenericIE(InfoExtractor):
        except ValueError:
            # since this is the last-resort InfoExtractor, if
            # this error is thrown, it'll be thrown here
-            raise ExtractorError(u'Failed to download URL: %s' % url)
+            raise ExtractorError('Failed to download URL: %s' % url)

        self.report_extraction(video_id)

@@ -211,17 +214,17 @@ class GenericIE(InfoExtractor):
        #   Video Title - Tagline | Site Name
        # and so on and so forth; it's just not practical
        video_title = self._html_search_regex(
-            r'(?s)<title>(.*?)</title>', webpage, u'video title',
-            default=u'video')
+            r'(?s)<title>(.*?)</title>', webpage, 'video title',
+            default='video')

        # video uploader is domain name
        video_uploader = self._search_regex(
-            r'^(?:https?://)?([^/]*)/.*', url, u'video uploader')
+            r'^(?:https?://)?([^/]*)/.*', url, 'video uploader')

        # Look for BrightCove:
        bc_url = BrightcoveIE._extract_brightcove_url(webpage)
        if bc_url is not None:
-            self.to_screen(u'Brightcove video detected.')
+            self.to_screen('Brightcove video detected.')
            return self.url_result(bc_url, 'Brightcove')

        # Look for embedded (iframe) Vimeo player
@@ -305,6 +308,9 @@ class GenericIE(InfoExtractor):

        # Start with something easy: JW Player in SWFObject
        mobj = re.search(r'flashvars: [\'"](?:.*&)?file=(http[^\'"&]*)', webpage)
+        if mobj is None:
+            # Look for gorilla-vid style embedding
+            mobj = re.search(r'(?s)jw_plugins.*?file:\s*["\'](.*?)["\']', webpage)
        if mobj is None:
            # Broaden the search a little bit
            mobj = re.search(r'[^A-Za-z0-9]?(?:file|source)=(http[^\'"&]*)', webpage)
@@ -325,23 +331,27 @@ class GenericIE(InfoExtractor):
            # HTML5 video
            mobj = re.search(r'<video[^<]*(?:>.*?<source.*?)? src="([^"]+)"', webpage, flags=re.DOTALL)
        if mobj is None:
-            raise ExtractorError(u'Unsupported URL: %s' % url)
+            raise ExtractorError('Unsupported URL: %s' % url)

        # It's possible that one of the regexes
        # matched, but returned an empty group:
        if mobj.group(1) is None:
-            raise ExtractorError(u'Did not find a valid video URL at %s' % url)
+            raise ExtractorError('Did not find a valid video URL at %s' % url)

        video_url = mobj.group(1)
        video_url = compat_urlparse.urljoin(url, video_url)
        video_id = compat_urllib_parse.unquote(os.path.basename(video_url))

+        # Sometimes, jwplayer extraction will result in a YouTube URL
+        if YoutubeIE.suitable(video_url):
+            return self.url_result(video_url, 'Youtube')
+
        # here's a fun little line of code for you:
        video_id = os.path.splitext(video_id)[0]

        return {
-            'id':       video_id,
-            'url':      video_url,
+            'id': video_id,
+            'url': video_url,
            'uploader': video_uploader,
-            'title':    video_title,
+            'title': video_title,
        }
--- a/youtube_dl/extractor/lynda.py
+++ b/youtube_dl/extractor/lynda.py
@@ -3,11 +3,12 @@ from __future__ import unicode_literals
 import re
 import json

+from .subtitles import SubtitlesInfoExtractor
 from .common import InfoExtractor
 from ..utils import ExtractorError


-class LyndaIE(InfoExtractor):
+class LyndaIE(SubtitlesInfoExtractor):
    IE_NAME = 'lynda'
    IE_DESC = 'lynda.com videos'
    _VALID_URL = r'https?://www\.lynda\.com/[^/]+/[^/]+/\d+/(\d+)-\d\.html'
@@ -21,7 +22,7 @@ class LyndaIE(InfoExtractor):
            'duration': 68
        }
    }
-
+    
    def _real_extract(self, url):
        mobj = re.match(self._VALID_URL, url)
        video_id = mobj.group(1)
@@ -45,17 +46,56 @@ class LyndaIE(InfoExtractor):
                    'width': fmt['Width'],
                    'height': fmt['Height'],
                    'filesize': fmt['FileSize'],
-                    'format_id': fmt['Resolution']
+                    'format_id': str(fmt['Resolution'])
                    } for fmt in video_json['Formats']]

        self._sort_formats(formats)
-
+        
+        if self._downloader.params.get('listsubtitles', False):
+            self._list_available_subtitles(video_id, page)
+            return
+        
+        subtitles = self._fix_subtitles(self.extract_subtitles(video_id, page))
+        
        return {
            'id': video_id,
            'title': title,
            'duration': duration,
+            'subtitles': subtitles,
            'formats': formats
        }
+        
+    _TIMECODE_REGEX = r'\[(?P<timecode>\d+:\d+:\d+[\.,]\d+)\]'    
+    
+    def _fix_subtitles(self, subtitles):
+        fixed_subtitles = {}
+        for k, v in subtitles.items():
+            subs = json.loads(v)
+            if len(subs) == 0:
+                continue
+            srt = ''
+            for pos in range(0, len(subs) - 1):
+                seq_current = subs[pos]                
+                m_current = re.match(self._TIMECODE_REGEX, seq_current['Timecode'])
+                if m_current is None:
+                    continue                
+                seq_next = subs[pos+1]
+                m_next = re.match(self._TIMECODE_REGEX, seq_next['Timecode'])
+                if m_next is None:
+                    continue                
+                appear_time = m_current.group('timecode')
+                disappear_time = m_next.group('timecode')
+                text = seq_current['Caption']
+                srt += '%s\r\n%s --> %s\r\n%s' % (str(pos), appear_time, disappear_time, text)
+            if srt:
+                fixed_subtitles[k] = srt
+        return fixed_subtitles
+        
+    def _get_available_subtitles(self, video_id, webpage):
+        url = 'http://www.lynda.com/ajax/player?videoId=%s&type=transcript' % video_id
+        sub = self._download_webpage(url, None, note=False)
+        sub_json = json.loads(sub)
+        return {'en': url} if len(sub_json) > 0 else {}


 class LyndaCourseIE(InfoExtractor):
--- a/youtube_dl/extractor/orf.py
+++ b/youtube_dl/extractor/orf.py
@@ -1,54 +1,98 @@
 # coding: utf-8
+from __future__ import unicode_literals

-import re
-import xml.etree.ElementTree
 import json
+import re

 from .common import InfoExtractor
 from ..utils import (
-    compat_urlparse,
-    ExtractorError,
-    find_xpath_attr,
+    HEADRequest,
+    unified_strdate,
 )

+
 class ORFIE(InfoExtractor):
-    _VALID_URL = r'https?://tvthek\.orf\.at/(programs/.+?/episodes|topics/.+?)/(?P<id>\d+)'
+    _VALID_URL = r'https?://tvthek\.orf\.at/(?:programs/.+?/episodes|topics/.+?|program/[^/]+)/(?P<id>\d+)'
+
+    _TEST = {
+        'url': 'http://tvthek.orf.at/program/matinee-Was-Sie-schon-immer-ueber-Klassik-wissen-wollten/7317210/Was-Sie-schon-immer-ueber-Klassik-wissen-wollten/7319746/Was-Sie-schon-immer-ueber-Klassik-wissen-wollten/7319747',
+        'file': '7319747.mp4',
+        'md5': 'bd803c5d8c32d3c64a0ea4b4eeddf375',
+        'info_dict': {
+            'title': 'Was Sie schon immer über Klassik wissen wollten',
+            'description': 'md5:0ddf0d5f0060bd53f744edaa5c2e04a4',
+            'duration': 3508,
+            'upload_date': '20140105',
+        },
+        'skip': 'Blocked outside of Austria',
+    }

    def _real_extract(self, url):
        mobj = re.match(self._VALID_URL, url)
        playlist_id = mobj.group('id')
        webpage = self._download_webpage(url, playlist_id)

-        flash_xml = self._search_regex('ORF.flashXML = \'(.+?)\'', webpage, u'flash xml')
-        flash_xml = compat_urlparse.parse_qs('xml='+flash_xml)['xml'][0]
-        flash_config = xml.etree.ElementTree.fromstring(flash_xml.encode('utf-8'))
-        playlist_json = self._search_regex(r'playlist\': \'(\[.*?\])\'', webpage, u'playlist').replace(r'\"','"')
-        playlist = json.loads(playlist_json)
+        data_json = self._search_regex(
+            r'initializeAdworx\((.+?)\);\n', webpage, 'video info')
+        all_data = json.loads(data_json)
+        sdata = all_data[0]['values']['segments']

-        videos = []
-        ns = '{http://tempuri.org/XMLSchema.xsd}'
-        xpath = '%(ns)sPlaylist/%(ns)sItems/%(ns)sItem' % {'ns': ns}
-        webpage_description = self._og_search_description(webpage)
-        for (i, (item, info)) in enumerate(zip(flash_config.findall(xpath), playlist), 1):
-            # Get best quality url
-            rtmp_url = None
-            for q in ['Q6A', 'Q4A', 'Q1A']:
-                video_url = find_xpath_attr(item, '%sVideoUrl' % ns, 'quality', q)
-                if video_url is not None:
-                    rtmp_url = video_url.text
-                    break
-            if rtmp_url is None:
-                raise ExtractorError(u'Couldn\'t get video url: %s' % info['id'])
-            description = self._html_search_regex(
-                r'id="playlist_entry_%s".*?<p>(.*?)</p>' % i, webpage,
-                u'description', default=webpage_description, flags=re.DOTALL)
-            videos.append({
+        def quality_to_int(s):
+            m = re.search('([0-9]+)', s)
+            if m is None:
+                return -1
+            return int(m.group(1))
+
+        entries = []
+        for sd in sdata:
+            video_id = sd['id']
+            formats = [{
+                'preference': -10 if fd['delivery'] == 'hls' else None,
+                'format_id': '%s-%s-%s' % (
+                    fd['delivery'], fd['quality'], fd['quality_string']),
+                'url': fd['src'],
+                'protocol': fd['protocol'],
+                'quality': quality_to_int(fd['quality']),
+            } for fd in sd['playlist_item_array']['sources']]
+
+            # Check for geoblocking.
+            # There is a property is_geoprotection, but that's always false
+            geo_str = sd.get('geoprotection_string')
+            if geo_str:
+                try:
+                    http_url = next(
+                        f['url']
+                        for f in formats
+                        if re.match(r'^https?://.*\.mp4$', f['url']))
+                except StopIteration:
+                    pass
+                else:
+                    req = HEADRequest(http_url)
+                    response = self._request_webpage(
+                        req, video_id,
+                        note='Testing for geoblocking',
+                        errnote=((
+                            'This video seems to be blocked outside of %s. '
+                            'You may want to try the streaming-* formats.')
+                            % geo_str),
+                        fatal=False)
+
+            self._sort_formats(formats)
+
+            upload_date = unified_strdate(sd['created_date'])
+            entries.append({
                '_type': 'video',
-                'id': info['id'],
-                'title': info['title'],
-                'url': rtmp_url,
-                'ext': 'flv',
-                'description': description,
-                })
+                'id': video_id,
+                'title': sd['header'],
+                'formats': formats,
+                'description': sd.get('description'),
+                'duration': int(sd['duration_in_seconds']),
+                'upload_date': upload_date,
+                'thumbnail': sd.get('image_full_url'),
+            })

-        return videos
+        return {
+            '_type': 'playlist',
+            'entries': entries,
+            'id': playlist_id,
+        }
--- a/youtube_dl/extractor/veehd.py
+++ b/youtube_dl/extractor/veehd.py
@@ -1,3 +1,5 @@
+from __future__ import unicode_literals
+
 import re
 import json

@@ -8,16 +10,17 @@ from ..utils import (
    clean_html,
 )

+
 class VeeHDIE(InfoExtractor):
    _VALID_URL = r'https?://veehd\.com/video/(?P<id>\d+)'

    _TEST = {
-        u'url': u'http://veehd.com/video/4686958',
-        u'file': u'4686958.mp4',
-        u'info_dict': {
-            u'title': u'Time Lapse View from Space ( ISS)',
-            u'uploader_id': u'spotted',
-            u'description': u'md5:f0094c4cf3a72e22bc4e4239ef767ad7',
+        'url': 'http://veehd.com/video/4686958',
+        'file': '4686958.mp4',
+        'info_dict': {
+            'title': 'Time Lapse View from Space ( ISS)',
+            'uploader_id': 'spotted',
+            'description': 'md5:f0094c4cf3a72e22bc4e4239ef767ad7',
        },
    }

@@ -25,24 +28,30 @@ class VeeHDIE(InfoExtractor):
        mobj = re.match(self._VALID_URL, url)
        video_id = mobj.group('id')

+        # VeeHD seems to send garbage on the first request.
+        # See https://github.com/rg3/youtube-dl/issues/2102
+        self._download_webpage(url, video_id, 'Requesting webpage')
        webpage = self._download_webpage(url, video_id)
-        player_path = self._search_regex(r'\$\("#playeriframe"\).attr\({src : "(.+?)"',
-            webpage, u'player path')
+        player_path = self._search_regex(
+            r'\$\("#playeriframe"\).attr\({src : "(.+?)"',
+            webpage, 'player path')
        player_url = compat_urlparse.urljoin(url, player_path)
-        player_page = self._download_webpage(player_url, video_id,
-            u'Downloading player page')
-        config_json = self._search_regex(r'value=\'config=({.+?})\'',
-            player_page, u'config json')
+
+        self._download_webpage(player_url, video_id, 'Requesting player page')
+        player_page = self._download_webpage(
+            player_url, video_id, 'Downloading player page')
+        config_json = self._search_regex(
+            r'value=\'config=({.+?})\'', player_page, 'config json')
        config = json.loads(config_json)

        video_url = compat_urlparse.unquote(config['clip']['url'])
        title = clean_html(get_element_by_id('videoName', webpage).rpartition('|')[0])
        uploader_id = self._html_search_regex(r'<a href="/profile/\d+">(.+?)</a>',
-            webpage, u'uploader')
+            webpage, 'uploader')
        thumbnail = self._search_regex(r'<img id="veehdpreview" src="(.+?)"',
-            webpage, u'thumbnail')
+            webpage, 'thumbnail')
        description = self._html_search_regex(r'<td class="infodropdown".*?<div>(.*?)<ul',
-            webpage, u'description', flags=re.DOTALL)
+            webpage, 'description', flags=re.DOTALL)

        return {
            '_type': 'video',
--- a/youtube_dl/extractor/veoh.py
+++ b/youtube_dl/extractor/veoh.py
@@ -1,22 +1,22 @@
+from __future__ import unicode_literals
+
 import re
 import json

 from .common import InfoExtractor
-from ..utils import (
-    determine_ext,
-)
+

 class VeohIE(InfoExtractor):
-    _VALID_URL = r'http://www\.veoh\.com/watch/v(?P<id>\d*)'
+    _VALID_URL = r'http://(?:www\.)?veoh\.com/(?:watch|iphone/#_Watch)/v(?P<id>\d*)'

    _TEST = {
-        u'url': u'http://www.veoh.com/watch/v56314296nk7Zdmz3',
-        u'file': u'56314296.mp4',
-        u'md5': u'620e68e6a3cff80086df3348426c9ca3',
-        u'info_dict': {
-            u'title': u'Straight Backs Are Stronger',
-            u'uploader': u'LUMOback',
-            u'description': u'At LUMOback, we believe straight backs are stronger.  The LUMOback Posture & Movement Sensor:  It gently vibrates when you slouch, inspiring improved posture and mobility.  Use the app to track your data and improve your posture over time. ',
+        'url': 'http://www.veoh.com/watch/v56314296nk7Zdmz3',
+        'file': '56314296.mp4',
+        'md5': '620e68e6a3cff80086df3348426c9ca3',
+        'info_dict': {
+            'title': 'Straight Backs Are Stronger',
+            'uploader': 'LUMOback',
+            'description': 'At LUMOback, we believe straight backs are stronger.  The LUMOback Posture & Movement Sensor:  It gently vibrates when you slouch, inspiring improved posture and mobility.  Use the app to track your data and improve your posture over time. ',
        }
    }

@@ -28,20 +28,20 @@ class VeohIE(InfoExtractor):
        m_youtube = re.search(r'http://www\.youtube\.com/v/(.*?)(\&|")', webpage)
        if m_youtube is not None:
            youtube_id = m_youtube.group(1)
-            self.to_screen(u'%s: detected Youtube video.' % video_id)
+            self.to_screen('%s: detected Youtube video.' % video_id)
            return self.url_result(youtube_id, 'Youtube')

        self.report_extraction(video_id)
        info = self._search_regex(r'videoDetailsJSON = \'({.*?})\';', webpage, 'info')
        info = json.loads(info)
-        video_url =  info.get('fullPreviewHashHighPath') or info.get('fullPreviewHashLowPath')
+        video_url = info.get('fullPreviewHashHighPath') or info.get('fullPreviewHashLowPath')

-        return {'id': info['videoId'], 
-                'title': info['title'],
-                'ext': determine_ext(video_url),
-                'url': video_url,
-                'uploader': info['username'],
-                'thumbnail': info.get('highResImage') or info.get('medResImage'),
-                'description': info['description'],
-                'view_count': info['views'],
-                }
+        return {
+            'id': info['videoId'],
+            'title': info['title'],
+            'url': video_url,
+            'uploader': info['username'],
+            'thumbnail': info.get('highResImage') or info.get('medResImage'),
+            'description': info['description'],
+            'view_count': info['views'],
+        }
--- a/youtube_dl/extractor/vimeo.py
+++ b/youtube_dl/extractor/vimeo.py
@@ -311,7 +311,7 @@ class VimeoChannelIE(InfoExtractor):

 class VimeoUserIE(VimeoChannelIE):
    IE_NAME = u'vimeo:user'
-    _VALID_URL = r'(?:https?://)?vimeo.\com/(?P<name>[^/]+)'
+    _VALID_URL = r'(?:https?://)?vimeo.\com/(?P<name>[^/]+)(?:[#?]|$)'
    _TITLE_RE = r'<a[^>]+?class="user">([^<>]+?)</a>'

    @classmethod
@@ -336,7 +336,7 @@ class VimeoAlbumIE(VimeoChannelIE):

    def _real_extract(self, url):
        mobj = re.match(self._VALID_URL, url)
-        album_id =  mobj.group('id')
+        album_id = mobj.group('id')
        return self._extract_videos(album_id, 'http://vimeo.com/album/%s' % album_id)


@@ -351,3 +351,24 @@ class VimeoGroupsIE(VimeoAlbumIE):
        mobj = re.match(self._VALID_URL, url)
        name = mobj.group('name')
        return self._extract_videos(name, 'http://vimeo.com/groups/%s' % name)
+
+
+class VimeoReviewIE(InfoExtractor):
+    IE_NAME = u'vimeo:review'
+    IE_DESC = u'Review pages on vimeo'
+    _VALID_URL = r'(?:https?://)?vimeo.\com/[^/]+/review/(?P<id>[^/]+)'
+    _TEST = {
+        'url': 'https://vimeo.com/user21297594/review/75524534/3c257a1b5d',
+        'file': '75524534.mp4',
+        'md5': 'c507a72f780cacc12b2248bb4006d253',
+        'info_dict': {
+            'title': "DICK HARDWICK 'Comedian'",
+            'uploader': 'Richard Hardwick',
+        }
+    }
+
+    def _real_extract(self, url):
+        mobj = re.match(self._VALID_URL, url)
+        video_id = mobj.group('id')
+        player_url = 'https://player.vimeo.com/player/' + video_id
+        return self.url_result(player_url, 'Vimeo', video_id)
--- a/youtube_dl/utils.py
+++ b/youtube_dl/utils.py
@@ -764,6 +764,7 @@ def unified_strdate(date_str):
        '%Y-%m-%d',
        '%d/%m/%Y',
        '%Y/%m/%d %H:%M:%S',
+        '%Y-%m-%d %H:%M:%S',
        '%d.%m.%Y %H:%M',
        '%Y-%m-%dT%H:%M:%SZ',
        '%Y-%m-%dT%H:%M:%S.%fZ',
--- a/youtube_dl/version.py
+++ b/youtube_dl/version.py
@@ -1,2 +1,2 @@

-__version__ = '2014.01.05'
+__version__ = '2014.01.06'
Author	SHA1	Message	Date
Philipp Hagemeister	e6162a90e6	release 2014.01.06	2014-01-06 17:37:24 +01:00
Philipp Hagemeister	9a6422a81e	Merge remote-tracking branch 'origin/master'	2014-01-06 17:37:20 +01:00
Philipp Hagemeister	fcea44c6d5	[vimeo] Add support for review pages Since the regexp is already overboarding and review pages have a distinct URL format (with non-trivial stuff after the ID), use a dedicated IE. Fixes #2106	2014-01-06 17:34:23 +01:00
Philipp Hagemeister	5d73273f6f	[orf] Use new extraction method (Fixes #2057 )	2014-01-06 17:15:27 +01:00
Philipp Hagemeister	c11a0611d9	[veehd] Send requests twice (Fixes #2102 )	2014-01-06 12:54:01 +01:00
Philipp Hagemeister	796495886e	[generic] Use unicode_literals instead of duplicating the u'	2014-01-06 01:47:52 +01:00
Philipp Hagemeister	fa27f667c8	Merge pull request #2104 from dstftw/lynda [lynda] Add subtitles extraction	2014-01-05 16:44:21 -08:00
Philipp Hagemeister	fc9713a1d2	[youtube] Support jwplayer with YouTube URLs (Closes #2075 )	2014-01-06 01:42:58 +01:00
dst	62bcfa8c57	[lynda] Add subtitles extraction	2014-01-05 23:59:33 +07:00
Philipp Hagemeister	7f9886379c	release 2014.01.05.6	2014-01-05 11:44:20 +01:00
Philipp Hagemeister	c6e4b225b1	Restore binary files for backwards compatibility Fixes `9656ee5d1d` New year's resolution: Check which systems of Ubuntu / RHEL still serve the ancient versions. If it's only RHEL, consider removing these binary files in 2015 or so.	2014-01-05 11:41:44 +01:00
Jaime Marquínez Ferrándiz	1c0f31f9f7	[bash-completion] Complete filename if `—load-info` is given	2014-01-05 11:28:01 +01:00
Jaime Marquínez Ferrándiz	41292a3827	Fix list comprehension for decoding the URLs (fixes #2100 ) It wasn’t a comprehension, it was just using the last url from the previous comprehension. That didn’t raise an error in python 2, but in python 3 the variable was not defined.	2014-01-05 10:58:36 +01:00
Philipp Hagemeister	20f1be02df	release 2014.01.05.5	2014-01-05 05:48:39 +01:00
Philipp Hagemeister	a339e5cfb5	Remove unused imports	2014-01-05 05:48:30 +01:00
Philipp Hagemeister	f46f4a995b	[veoh] Simplify	2014-01-05 05:48:12 +01:00
Philipp Hagemeister	4ddba33f78	[veoh] Add support for mobile URLs Fixes #2052	2014-01-05 05:47:50 +01:00
Philipp Hagemeister	e3b7aa8428	release 2014.01.05.4	2014-01-05 05:41:30 +01:00
Philipp Hagemeister	d981cef6b9	[generic] Support gorillavid.in Previously, we were a little bit over-eager and got a random swf file. Fixes #2084.	2014-01-05 05:34:08 +01:00
Philipp Hagemeister	6fa81ee96e	release 2014.01.05.3	2014-01-05 05:26:43 +01:00
Philipp Hagemeister	a1a337ade9	release 2014.01.05.02	2014-01-05 05:25:07 +01:00
Philipp Hagemeister	c774b3c696	Make sure URLs are always character strings (Fixes #2051 )	2014-01-05 05:24:50 +01:00
Philipp Hagemeister	3e34db3170	More Atom feed improvements (#2081 )	2014-01-05 05:16:16 +01:00
Philipp Hagemeister	317d4edfa8	Improve Atom feed creation (Fixes #2081 )	2014-01-05 05:04:46 +01:00
Philipp Hagemeister	9b12003c35	atom feed generator: Make IDs proper URLs (#2081 )	2014-01-05 04:49:43 +01:00
Philipp Hagemeister	4ea170b8a0	release 2014.01.05.1	2014-01-05 04:44:34 +01:00
Philipp Hagemeister	49f2bf76a8	Fix make_readme on Python 2	2014-01-05 04:44:29 +01:00
Philipp Hagemeister	01c62591d1	[setup.py] Do not use unicode literals See http://bugs.python.org/issue13943 for context	2014-01-05 04:41:50 +01:00
Philipp Hagemeister	1e91866f77	Make make_readme run in a locale-less environment Mentioned in #267	2014-01-05 04:39:27 +01:00
Philipp Hagemeister	9656ee5d1d	Document --socket-timeout	2014-01-05 04:36:46 +01:00