youtube-dl/youtube_dl/extractor/spiegeltv.py

# coding: utf-8
from __future__ import unicode_literals

from .common import InfoExtractor
from ..compat import compat_urllib_parse_urlparse
from ..utils import (
    determine_ext,
    float_or_none,
)


class SpiegeltvIE(InfoExtractor):
    _VALID_URL = r'https?://(?:www\.)?spiegel\.tv/(?:#/)?filme/(?P<id>[\-a-z0-9]+)'
    _TESTS = [{
        'url': 'http://www.spiegel.tv/filme/flug-mh370/',
        'info_dict': {
            'id': 'flug-mh370',
            'ext': 'm4v',
            'title': 'Flug MH370',
            'description': 'Das Rätsel um die Boeing 777 der Malaysia-Airlines',
            'thumbnail': 're:http://.*\.jpg$',
        },
        'params': {
            # m3u8 download
            'skip_download': True,
        }
    }, {
        'url': 'http://www.spiegel.tv/#/filme/alleskino-die-wahrheit-ueber-maenner/',
        'only_matching': True,
    }]

    def _real_extract(self, url):
        if '/#/' in url:
            url = url.replace('/#/', '/')
        video_id = self._match_id(url)
        webpage = self._download_webpage(url, video_id)
        title = self._html_search_regex(r'<h1.*?>(.*?)</h1>', webpage, 'title')

        apihost = 'http://spiegeltv-ivms2-restapi.s3.amazonaws.com'
        version_json = self._download_json(
            '%s/version.json' % apihost, video_id,
            note='Downloading version information')
        version_name = version_json['version_name']

        slug_json = self._download_json(
            '%s/%s/restapi/slugs/%s.json' % (apihost, version_name, video_id),
            video_id,
            note='Downloading object information')
        oid = slug_json['object_id']

        media_json = self._download_json(
            '%s/%s/restapi/media/%s.json' % (apihost, version_name, oid),
            video_id, note='Downloading media information')
        uuid = media_json['uuid']
        is_wide = media_json['is_wide']

        server_json = self._download_json(
            'http://spiegeltv-prod-static.s3.amazonaws.com/projectConfigs/projectConfig.json',
            video_id, note='Downloading server information')

        format = '16x9' if is_wide else '4x3'

        formats = []
        for streamingserver in server_json['streamingserver']:
            endpoint = streamingserver.get('endpoint')
            if not endpoint:
                continue
            play_path = 'mp4:%s_spiegeltv_0500_%s.m4v' % (uuid, format)
            if endpoint.startswith('rtmp'):
                formats.append({
                    'url': endpoint,
                    'format_id': 'rtmp',
                    'app': compat_urllib_parse_urlparse(endpoint).path[1:],
                    'play_path': play_path,
                    'player_path': 'http://prod-static.spiegel.tv/frontend-076.swf',
                    'ext': 'flv',
                    'rtmp_live': True,
                })
            elif determine_ext(endpoint) == 'm3u8':
                formats.append({
                    'url': endpoint.replace('[video]', play_path),
                    'ext': 'm4v',
                    'format_id': 'hls',  # Prefer hls since it allows to workaround georestriction
                    'protocol': 'm3u8',
                    'preference': 1,
                    'http_headers': {
                        'Accept-Encoding': 'deflate',  # gzip causes trouble on the server side
                    },
                })
            else:
                formats.append({
                    'url': endpoint,
                })
        self._check_formats(formats, video_id)

        thumbnails = []
        for image in media_json['images']:
            thumbnails.append({
                'url': image['url'],
                'width': image['width'],
                'height': image['height'],
            })

        description = media_json['subtitle']
        duration = float_or_none(media_json.get('duration_in_ms'), scale=1000)

        return {
            'id': video_id,
            'title': title,
            'description': description,
            'duration': duration,
            'thumbnails': thumbnails,
            'formats': formats,
        }
added spiegel.tv 2014-05-30 22:35:17 +08:00			`# coding: utf-8`
			`from __future__ import unicode_literals`

			`from .common import InfoExtractor`
[spiegeltv] Extract all formats and prefer hls (Closes #5843) 2015-06-09 22:36:08 +08:00			`from ..compat import compat_urllib_parse_urlparse`
			`from ..utils import (`
			`determine_ext,`
			`float_or_none,`
			`)`
added spiegel.tv 2014-05-30 22:35:17 +08:00
[spiegeltv] Simplify and PEP8 2014-06-07 21:33:45 +08:00
added spiegel.tv 2014-05-30 22:35:17 +08:00			`class SpiegeltvIE(InfoExtractor):`
[spiegeltv] Match hash-style URLs (Closes #4210) 2014-11-16 07:40:09 +08:00			`_VALID_URL = r'https?://(?:www\.)?spiegel\.tv/(?:#/)?filme/(?P<id>[\-a-z0-9]+)'`
[spiegeltv] Modernize 2014-11-16 07:33:51 +08:00			`_TESTS = [{`
added spiegel.tv 2014-05-30 22:35:17 +08:00			`'url': 'http://www.spiegel.tv/filme/flug-mh370/',`
			`'info_dict': {`
			`'id': 'flug-mh370',`
			`'ext': 'm4v',`
			`'title': 'Flug MH370',`
			`'description': 'Das Rätsel um die Boeing 777 der Malaysia-Airlines',`
[spiegeltv] Simplify and PEP8 2014-06-07 21:33:45 +08:00			`'thumbnail': 're:http://.*\.jpg$',`
[Spiegeltv] skip rtmp download to pass Travis test build 2014-06-03 22:50:54 +08:00			`},`
			`'params': {`
[spiegeltv] Extract all formats and prefer hls (Closes #5843) 2015-06-09 22:36:08 +08:00			`# m3u8 download`
[Spiegeltv] skip rtmp download to pass Travis test build 2014-06-03 22:50:54 +08:00			`'skip_download': True,`
added spiegel.tv 2014-05-30 22:35:17 +08:00			`}`
[spiegeltv] Match hash-style URLs (Closes #4210) 2014-11-16 07:40:09 +08:00			`}, {`
			`'url': 'http://www.spiegel.tv/#/filme/alleskino-die-wahrheit-ueber-maenner/',`
			`'only_matching': True,`
[spiegeltv] Modernize 2014-11-16 07:33:51 +08:00			`}]`
added spiegel.tv 2014-05-30 22:35:17 +08:00
			`def _real_extract(self, url):`
[spiegeltv] Match hash-style URLs (Closes #4210) 2014-11-16 07:40:09 +08:00			`if '/#/' in url:`
			`url = url.replace('/#/', '/')`
[spiegeltv] Modernize 2014-11-16 07:33:51 +08:00			`video_id = self._match_id(url)`
added spiegel.tv 2014-05-30 22:35:17 +08:00			`webpage = self._download_webpage(url, video_id)`
			`title = self._html_search_regex(r'<h1.?>(.?)</h1>', webpage, 'title')`

[spiegeltv] Simplify and PEP8 2014-06-07 21:33:45 +08:00			`apihost = 'http://spiegeltv-ivms2-restapi.s3.amazonaws.com'`
			`version_json = self._download_json(`
			`'%s/version.json' % apihost, video_id,`
			`note='Downloading version information')`
			`version_name = version_json['version_name']`
added spiegel.tv 2014-05-30 22:35:17 +08:00
[spiegeltv] Simplify and PEP8 2014-06-07 21:33:45 +08:00			`slug_json = self._download_json(`
			`'%s/%s/restapi/slugs/%s.json' % (apihost, version_name, video_id),`
			`video_id,`
			`note='Downloading object information')`
			`oid = slug_json['object_id']`
added spiegel.tv 2014-05-30 22:35:17 +08:00
[spiegeltv] Simplify and PEP8 2014-06-07 21:33:45 +08:00			`media_json = self._download_json(`
			`'%s/%s/restapi/media/%s.json' % (apihost, version_name, oid),`
			`video_id, note='Downloading media information')`
			`uuid = media_json['uuid']`
			`is_wide = media_json['is_wide']`
added spiegel.tv 2014-05-30 22:35:17 +08:00
[spiegeltv] Simplify and PEP8 2014-06-07 21:33:45 +08:00			`server_json = self._download_json(`
[spiegeltv] Changed RTMP server (fixes #5788 and fixes #5843) Thanks to @brickleroux for finding out the problem 2015-05-30 13:23:09 +08:00			`'http://spiegeltv-prod-static.s3.amazonaws.com/projectConfigs/projectConfig.json',`
			`video_id, note='Downloading server information')`
[spiegeltv] Extract all formats and prefer hls (Closes #5843) 2015-06-09 22:36:08 +08:00
			`format = '16x9' if is_wide else '4x3'`

			`formats = []`
			`for streamingserver in server_json['streamingserver']:`
			`endpoint = streamingserver.get('endpoint')`
			`if not endpoint:`
			`continue`
			`play_path = 'mp4:%s_spiegeltv_0500_%s.m4v' % (uuid, format)`
			`if endpoint.startswith('rtmp'):`
			`formats.append({`
			`'url': endpoint,`
			`'format_id': 'rtmp',`
			`'app': compat_urllib_parse_urlparse(endpoint).path[1:],`
			`'play_path': play_path,`
			`'player_path': 'http://prod-static.spiegel.tv/frontend-076.swf',`
			`'ext': 'flv',`
			`'rtmp_live': True,`
			`})`
			`elif determine_ext(endpoint) == 'm3u8':`
[spiegeltv] Do not extract m3u8 formats since it's already a format 2015-10-24 18:24:08 +08:00			`formats.append({`
			`'url': endpoint.replace('[video]', play_path),`
			`'ext': 'm4v',`
			`'format_id': 'hls', # Prefer hls since it allows to workaround georestriction`
			`'protocol': 'm3u8',`
			`'preference': 1,`
			`'http_headers': {`
[spiegeltv] Fix style issue Use two spaces before comment. 2015-10-24 18:41:41 +08:00			`'Accept-Encoding': 'deflate', # gzip causes trouble on the server side`
[spiegeltv] Do not extract m3u8 formats since it's already a format 2015-10-24 18:24:08 +08:00			`},`
			`})`
[spiegeltv] Extract all formats and prefer hls (Closes #5843) 2015-06-09 22:36:08 +08:00			`else:`
			`formats.append({`
			`'url': endpoint,`
			`})`
[spiegeltv] Check formats 2015-10-24 18:25:44 +08:00			`self._check_formats(formats, video_id)`
added spiegel.tv 2014-05-30 22:35:17 +08:00
			`thumbnails = []`
			`for image in media_json['images']:`
[spiegeltv] Simplify and PEP8 2014-06-07 21:33:45 +08:00			`thumbnails.append({`
			`'url': image['url'],`
			`'width': image['width'],`
			`'height': image['height'],`
			`})`
added spiegel.tv 2014-05-30 22:35:17 +08:00
			`description = media_json['subtitle']`
[spiegeltv] Modernize 2014-11-16 07:33:51 +08:00			`duration = float_or_none(media_json.get('duration_in_ms'), scale=1000)`
added spiegel.tv 2014-05-30 22:35:17 +08:00
[spiegeltv] Simplify and PEP8 2014-06-07 21:33:45 +08:00			`return {`
added spiegel.tv 2014-05-30 22:35:17 +08:00			`'id': video_id,`
			`'title': title,`
			`'description': description,`
			`'duration': duration,`
[spiegeltv] Changed RTMP server (fixes #5788 and fixes #5843) Thanks to @brickleroux for finding out the problem 2015-05-30 13:23:09 +08:00			`'thumbnails': thumbnails,`
[spiegeltv] Extract all formats and prefer hls (Closes #5843) 2015-06-09 22:36:08 +08:00			`'formats': formats,`
PEP8 applied 2014-11-24 03:41:03 +08:00			`}`