release 2016.06.04

2016-06-04 22:42:10 +07:00
5 changed files with 68 additions and 144 deletions
--- a/.github/ISSUE_TEMPLATE.md
+++ b/.github/ISSUE_TEMPLATE.md
@@ -6,8 +6,8 @@

 ---

-### Make sure you are using the *latest* version: run `youtube-dl --version` and ensure your version is *2016.06.03*. If it's not read [this FAQ entry](https://github.com/rg3/youtube-dl/blob/master/README.md#how-do-i-update-youtube-dl) and update. Issues with outdated version will be rejected.
- [ ] I've **verified** and **I assure** that I'm running youtube-dl **2016.06.03**
+### Make sure you are using the *latest* version: run `youtube-dl --version` and ensure your version is *2016.06.04*. If it's not read [this FAQ entry](https://github.com/rg3/youtube-dl/blob/master/README.md#how-do-i-update-youtube-dl) and update. Issues with outdated version will be rejected.
+- [ ] I've **verified** and **I assure** that I'm running youtube-dl **2016.06.04**

 ### Before submitting an *issue* make sure you have:
 - [ ] At least skimmed through [README](https://github.com/rg3/youtube-dl/blob/master/README.md) and **most notably** [FAQ](https://github.com/rg3/youtube-dl#faq) and [BUGS](https://github.com/rg3/youtube-dl#bugs) sections
@@ -35,7 +35,7 @@ $ youtube-dl -v <your command line>
 [debug] User config: []
 [debug] Command-line args: [u'-v', u'http://www.youtube.com/watch?v=BaW_jenozKcj']
 [debug] Encodings: locale cp1251, fs mbcs, out cp866, pref cp1251
-[debug] youtube-dl version 2016.06.03
+[debug] youtube-dl version 2016.06.04
 [debug] Python version 2.7.11 - Windows-2003Server-5.2.3790-SP2
 [debug] exe versions: ffmpeg N-75573-g1d0487f, ffprobe N-75573-g1d0487f, rtmpdump 2.4
 [debug] Proxy map: {}
--- a/docs/supportedsites.md
+++ b/docs/supportedsites.md
@@ -43,8 +43,8 @@
 - **appletrailers:section**
 - **archive.org**: archive.org videos
 - **ARD**
- - **ARD:mediathek**: Saarländischer Rundfunk
 - **ARD:mediathek**
+ - **ARD:mediathek**: Saarländischer Rundfunk
 - **arte.tv**
 - **arte.tv:+7**
 - **arte.tv:cinema**
--- a/youtube_dl/extractor/channel9.py
+++ b/youtube_dl/extractor/channel9.py
@@ -20,64 +20,54 @@ class Channel9IE(InfoExtractor):
    '''
    IE_DESC = 'Channel 9'
    IE_NAME = 'channel9'
-    _VALID_URL = r'https?://(?:www\.)?channel9\.msdn\.com/(?P<contentpath>.+?)(?P<rss>/RSS)?/?(?:[?#&]|$)'
+    _VALID_URL = r'https?://(?:www\.)?channel9\.msdn\.com/(?P<contentpath>.+)/?'

-    _TESTS = [{
-        'url': 'http://channel9.msdn.com/Events/TechEd/Australia/2013/KOS002',
-        'md5': 'bbd75296ba47916b754e73c3a4bbdf10',
-        'info_dict': {
-            'id': 'Events/TechEd/Australia/2013/KOS002',
-            'ext': 'mp4',
-            'title': 'Developer Kick-Off Session: Stuff We Love',
-            'description': 'md5:c08d72240b7c87fcecafe2692f80e35f',
-            'duration': 4576,
-            'thumbnail': 're:http://.*\.jpg',
-            'session_code': 'KOS002',
-            'session_day': 'Day 1',
-            'session_room': 'Arena 1A',
-            'session_speakers': ['Ed Blankenship', 'Andrew Coates', 'Brady Gaster', 'Patrick Klug',
-                                 'Mads Kristensen'],
+    _TESTS = [
+        {
+            'url': 'http://channel9.msdn.com/Events/TechEd/Australia/2013/KOS002',
+            'md5': 'bbd75296ba47916b754e73c3a4bbdf10',
+            'info_dict': {
+                'id': 'Events/TechEd/Australia/2013/KOS002',
+                'ext': 'mp4',
+                'title': 'Developer Kick-Off Session: Stuff We Love',
+                'description': 'md5:c08d72240b7c87fcecafe2692f80e35f',
+                'duration': 4576,
+                'thumbnail': 're:http://.*\.jpg',
+                'session_code': 'KOS002',
+                'session_day': 'Day 1',
+                'session_room': 'Arena 1A',
+                'session_speakers': ['Ed Blankenship', 'Andrew Coates', 'Brady Gaster', 'Patrick Klug', 'Mads Kristensen'],
+            },
        },
-    }, {
-        'url': 'http://channel9.msdn.com/posts/Self-service-BI-with-Power-BI-nuclear-testing',
-        'md5': 'b43ee4529d111bc37ba7ee4f34813e68',
-        'info_dict': {
-            'id': 'posts/Self-service-BI-with-Power-BI-nuclear-testing',
-            'ext': 'mp4',
-            'title': 'Self-service BI with Power BI - nuclear testing',
-            'description': 'md5:d1e6ecaafa7fb52a2cacdf9599829f5b',
-            'duration': 1540,
-            'thumbnail': 're:http://.*\.jpg',
-            'authors': ['Mike Wilmot'],
+        {
+            'url': 'http://channel9.msdn.com/posts/Self-service-BI-with-Power-BI-nuclear-testing',
+            'md5': 'b43ee4529d111bc37ba7ee4f34813e68',
+            'info_dict': {
+                'id': 'posts/Self-service-BI-with-Power-BI-nuclear-testing',
+                'ext': 'mp4',
+                'title': 'Self-service BI with Power BI - nuclear testing',
+                'description': 'md5:d1e6ecaafa7fb52a2cacdf9599829f5b',
+                'duration': 1540,
+                'thumbnail': 're:http://.*\.jpg',
+                'authors': ['Mike Wilmot'],
+            },
        },
-    }, {
-        # low quality mp4 is best
-        'url': 'https://channel9.msdn.com/Events/CPP/CppCon-2015/Ranges-for-the-Standard-Library',
-        'info_dict': {
-            'id': 'Events/CPP/CppCon-2015/Ranges-for-the-Standard-Library',
-            'ext': 'mp4',
-            'title': 'Ranges for the Standard Library',
-            'description': 'md5:2e6b4917677af3728c5f6d63784c4c5d',
-            'duration': 5646,
-            'thumbnail': 're:http://.*\.jpg',
-        },
-        'params': {
-            'skip_download': True,
-        },
-    }, {
-        'url': 'https://channel9.msdn.com/Niners/Splendid22/Queue/76acff796e8f411184b008028e0d492b/RSS',
-        'info_dict': {
-            'id': 'Niners/Splendid22/Queue/76acff796e8f411184b008028e0d492b',
-            'title': 'Channel 9',
-        },
-        'playlist_count': 2,
-    }, {
-        'url': 'https://channel9.msdn.com/Events/DEVintersection/DEVintersection-2016/RSS',
-        'only_matching': True,
-    }, {
-        'url': 'https://channel9.msdn.com/Events/Speakers/scott-hanselman/RSS?UrlSafeName=scott-hanselman',
-        'only_matching': True,
-    }]
+        {
+            # low quality mp4 is best
+            'url': 'https://channel9.msdn.com/Events/CPP/CppCon-2015/Ranges-for-the-Standard-Library',
+            'info_dict': {
+                'id': 'Events/CPP/CppCon-2015/Ranges-for-the-Standard-Library',
+                'ext': 'mp4',
+                'title': 'Ranges for the Standard Library',
+                'description': 'md5:2e6b4917677af3728c5f6d63784c4c5d',
+                'duration': 5646,
+                'thumbnail': 're:http://.*\.jpg',
+            },
+            'params': {
+                'skip_download': True,
+            },
+        }
+    ]

    _RSS_URL = 'http://channel9.msdn.com/%s/RSS'

@@ -264,30 +254,22 @@ class Channel9IE(InfoExtractor):

        return self.playlist_result(contents)

-    def _extract_list(self, video_id, rss_url=None):
-        if not rss_url:
-            rss_url = self._RSS_URL % video_id
-        rss = self._download_xml(rss_url, video_id, 'Downloading RSS')
+    def _extract_list(self, content_path):
+        rss = self._download_xml(self._RSS_URL % content_path, content_path, 'Downloading RSS')
        entries = [self.url_result(session_url.text, 'Channel9')
                   for session_url in rss.findall('./channel/item/link')]
        title_text = rss.find('./channel/title').text
-        return self.playlist_result(entries, video_id, title_text)
+        return self.playlist_result(entries, content_path, title_text)

    def _real_extract(self, url):
        mobj = re.match(self._VALID_URL, url)
        content_path = mobj.group('contentpath')
-        rss = mobj.group('rss')

-        if rss:
-            return self._extract_list(content_path, url)
+        webpage = self._download_webpage(url, content_path, 'Downloading web page')

-        webpage = self._download_webpage(
-            url, content_path, 'Downloading web page')
-
-        page_type = self._search_regex(
-            r'<meta[^>]+name=(["\'])WT\.entryid\1[^>]+content=(["\'])(?P<pagetype>[^:]+).+?\2',
-            webpage, 'page type', default=None, group='pagetype')
-        if page_type:
+        page_type_m = re.search(r'<meta name="WT.entryid" content="(?P<pagetype>[^:]+)[^"]+"/>', webpage)
+        if page_type_m is not None:
+            page_type = page_type_m.group('pagetype')
            if page_type == 'Entry':      # Any 'item'-like page, may contain downloadable content
                return self._extract_entry_item(webpage, content_path)
            elif page_type == 'Session':  # Event session page, may contain downloadable content
@@ -296,5 +278,6 @@ class Channel9IE(InfoExtractor):
                return self._extract_list(content_path)
            else:
                raise ExtractorError('Unexpected WT.entryid %s' % page_type, expected=True)
+
        else:  # Assuming list
            return self._extract_list(content_path)
--- a/youtube_dl/extractor/libraryofcongress.py
+++ b/youtube_dl/extractor/libraryofcongress.py
@@ -1,24 +1,20 @@
 # coding: utf-8
 from __future__ import unicode_literals

-import re
-
 from .common import InfoExtractor

 from ..utils import (
    determine_ext,
    float_or_none,
    int_or_none,
-    parse_filesize,
 )


 class LibraryOfCongressIE(InfoExtractor):
    IE_NAME = 'loc'
    IE_DESC = 'Library of Congress'
-    _VALID_URL = r'https?://(?:www\.)?loc\.gov/(?:item/|today/cyberlc/feature_wdesc\.php\?.*\brec=)(?P<id>[0-9]+)'
-    _TESTS = [{
-        # embedded via <div class="media-player"
+    _VALID_URL = r'https?://(?:www\.)?loc\.gov/item/(?P<id>[0-9]+)'
+    _TEST = {
        'url': 'http://loc.gov/item/90716351/',
        'md5': '353917ff7f0255aa6d4b80a034833de8',
        'info_dict': {
@@ -29,35 +25,7 @@ class LibraryOfCongressIE(InfoExtractor):
            'duration': 0,
            'view_count': int,
        },
-    }, {
-        # webcast embedded via mediaObjectId
-        'url': 'https://www.loc.gov/today/cyberlc/feature_wdesc.php?rec=5578',
-        'info_dict': {
-            'id': '5578',
-            'ext': 'mp4',
-            'title': 'Help! Preservation Training Needs Here, There & Everywhere',
-            'duration': 3765,
-            'view_count': int,
-            'subtitles': 'mincount:1',
-        },
-        'params': {
-            'skip_download': True,
-        },
-    }, {
-        # with direct download links
-        'url': 'https://www.loc.gov/item/78710669/',
-        'info_dict': {
-            'id': '78710669',
-            'ext': 'mp4',
-            'title': 'La vie et la passion de Jesus-Christ',
-            'duration': 0,
-            'view_count': int,
-            'formats': 'mincount:4',
-        },
-        'params': {
-            'skip_download': True,
-        },
-    }]
+    }

    def _real_extract(self, url):
        video_id = self._match_id(url)
@@ -66,20 +34,18 @@ class LibraryOfCongressIE(InfoExtractor):
        media_id = self._search_regex(
            (r'id=(["\'])media-player-(?P<id>.+?)\1',
             r'<video[^>]+id=(["\'])uuid-(?P<id>.+?)\1',
-             r'<video[^>]+data-uuid=(["\'])(?P<id>.+?)\1',
-             r'mediaObjectId\s*:\s*(["\'])(?P<id>.+?)\1'),
+             r'<video[^>]+data-uuid=(["\'])(?P<id>.+?)\1'),
            webpage, 'media id', group='id')

-        data = self._download_json(
-            'https://media.loc.gov/services/v1/media?id=%s&context=json' % media_id,
+        data = self._parse_json(
+            self._download_webpage(
+                'https://media.loc.gov/services/v1/media?id=%s&context=json' % media_id,
+                video_id),
            video_id)['mediaObject']

        derivative = data['derivatives'][0]
        media_url = derivative['derivativeUrl']

-        title = derivative.get('shortName') or data.get('shortName') or self._og_search_title(
-            webpage)
-
        # Following algorithm was extracted from setAVSource js function
        # found in webpage
        media_url = media_url.replace('rtmp', 'https')
@@ -95,7 +61,6 @@ class LibraryOfCongressIE(InfoExtractor):
                'format_id': 'hls',
                'ext': 'mp4',
                'protocol': 'm3u8_native',
-                'quality': 1,
            }]
        elif 'vod/mp3:' in media_url:
            formats = [{
@@ -103,41 +68,17 @@ class LibraryOfCongressIE(InfoExtractor):
                'vcodec': 'none',
            }]

-        download_urls = set()
-        for m in re.finditer(
-                r'<option[^>]+value=(["\'])(?P<url>.+?)\1[^>]+data-file-download=[^>]+>\s*(?P<id>.+?)(?:(?:&nbsp;|\s+)\((?P<size>.+?)\))?\s*<', webpage):
-            format_id = m.group('id').lower()
-            if format_id == 'gif':
-                continue
-            download_url = m.group('url')
-            if download_url in download_urls:
-                continue
-            download_urls.add(download_url)
-            formats.append({
-                'url': download_url,
-                'format_id': format_id,
-                'filesize_approx': parse_filesize(m.group('size')),
-            })
-
        self._sort_formats(formats)

+        title = derivative.get('shortName') or data.get('shortName') or self._og_search_title(webpage)
        duration = float_or_none(data.get('duration'))
        view_count = int_or_none(data.get('viewCount'))

-        subtitles = {}
-        cc_url = data.get('ccUrl')
-        if cc_url:
-            subtitles.setdefault('en', []).append({
-                'url': cc_url,
-                'ext': 'ttml',
-            })
-
        return {
            'id': video_id,
            'title': title,
-            'thumbnail': self._og_search_thumbnail(webpage, default=None),
+            'thumbnail': self._og_search_thumbnail(webpage),
            'duration': duration,
            'view_count': view_count,
            'formats': formats,
-            'subtitles': subtitles,
        }
--- a/youtube_dl/version.py
+++ b/youtube_dl/version.py
@@ -1,3 +1,3 @@
 from __future__ import unicode_literals

-__version__ = '2016.06.03'
+__version__ = '2016.06.04'