youtube-dl/youtube_dl/extractor/nbc.py

from __future__ import unicode_literals

import re
import json

from .common import InfoExtractor
from ..compat import (
    compat_str,
    compat_HTTPError,
)
from ..utils import (
    ExtractorError,
    find_xpath_attr,
)


class NBCIE(InfoExtractor):
    _VALID_URL = r'http://www\.nbc\.com/(?:[^/]+/)+(?P<id>n?\d+)'

    _TESTS = [
        {
            'url': 'http://www.nbc.com/chicago-fire/video/i-am-a-firefighter/2734188',
            # md5 checksum is not stable
            'info_dict': {
                'id': 'bTmnLCvIbaaH',
                'ext': 'flv',
                'title': 'I Am a Firefighter',
                'description': 'An emergency puts Dawson\'sf irefighter skills to the ultimate test in this four-part digital series.',
            },
        },
        {
            'url': 'http://www.nbc.com/the-tonight-show/episodes/176',
            'info_dict': {
                'id': 'XwU9KZkp98TH',
                'ext': 'flv',
                'title': 'Ricky Gervais, Steven Van Zandt, ILoveMakonnen',
                'description': 'A brand new episode of The Tonight Show welcomes Ricky Gervais, Steven Van Zandt and ILoveMakonnen.',
            },
            'skip': 'Only works from US',
        },
    ]

    def _real_extract(self, url):
        video_id = self._match_id(url)
        webpage = self._download_webpage(url, video_id)
        theplatform_url = self._search_regex(
            '(?:class="video-player video-player-full" data-mpx-url|class="player" src)="(.*?)"',
            webpage, 'theplatform url').replace('_no_endcard', '')
        if theplatform_url.startswith('//'):
            theplatform_url = 'http:' + theplatform_url
        return self.url_result(theplatform_url)


class NBCNewsIE(InfoExtractor):
    _VALID_URL = r'''(?x)https?://www\.nbcnews\.com/
        ((video/.+?/(?P<id>\d+))|
        (feature/[^/]+/(?P<title>.+)))
        '''

    _TESTS = [
        {
            'url': 'http://www.nbcnews.com/video/nbc-news/52753292',
            'md5': '47abaac93c6eaf9ad37ee6c4463a5179',
            'info_dict': {
                'id': '52753292',
                'ext': 'flv',
                'title': 'Crew emerges after four-month Mars food study',
                'description': 'md5:24e632ffac72b35f8b67a12d1b6ddfc1',
            },
        },
        {
            'url': 'http://www.nbcnews.com/feature/edward-snowden-interview/how-twitter-reacted-snowden-interview-n117236',
            'md5': 'b2421750c9f260783721d898f4c42063',
            'info_dict': {
                'id': 'I1wpAI_zmhsQ',
                'ext': 'mp4',
                'title': 'How Twitter Reacted To The Snowden Interview',
                'description': 'md5:65a0bd5d76fe114f3c2727aa3a81fe64',
            },
            'add_ie': ['ThePlatform'],
        },
        {
            'url': 'http://www.nbcnews.com/feature/dateline-full-episodes/full-episode-family-business-n285156',
            'md5': 'fdbf39ab73a72df5896b6234ff98518a',
            'info_dict': {
                'id': 'Wjf9EDR3A_60',
                'ext': 'mp4',
                'title': 'FULL EPISODE: Family Business',
                'description': 'md5:757988edbaae9d7be1d585eb5d55cc04',
            },
        },
    ]

    def _real_extract(self, url):
        mobj = re.match(self._VALID_URL, url)
        video_id = mobj.group('id')
        if video_id is not None:
            all_info = self._download_xml('http://www.nbcnews.com/id/%s/displaymode/1219' % video_id, video_id)
            info = all_info.find('video')

            return {
                'id': video_id,
                'title': info.find('headline').text,
                'ext': 'flv',
                'url': find_xpath_attr(info, 'media', 'type', 'flashVideo').text,
                'description': compat_str(info.find('caption').text),
                'thumbnail': find_xpath_attr(info, 'media', 'type', 'thumbnail').text,
            }
        else:
            # "feature" pages use theplatform.com
            title = mobj.group('title')
            webpage = self._download_webpage(url, title)
            bootstrap_json = self._search_regex(
                r'var bootstrapJson = ({.+})\s*$', webpage, 'bootstrap json',
                flags=re.MULTILINE)
            bootstrap = json.loads(bootstrap_json)
            info = bootstrap['results'][0]['video']
            mpxid = info['mpxId']

            base_urls = [
                info['fallbackPlaylistUrl'],
                info['associatedPlaylistUrl'],
            ]

            for base_url in base_urls:
                if not base_url:
                    continue
                playlist_url = base_url + '?form=MPXNBCNewsAPI'

                try:
                    all_videos = self._download_json(playlist_url, title)
                except ExtractorError as ee:
                    if isinstance(ee.cause, compat_HTTPError):
                        continue
                    raise

                if not all_videos or 'videos' not in all_videos:
                    continue

                try:
                    info = next(v for v in all_videos['videos'] if v['mpxId'] == mpxid)
                    break
                except StopIteration:
                    continue

            if info is None:
                raise ExtractorError('Could not find video in playlists')

            return {
                '_type': 'url',
                # We get the best quality video
                'url': info['videoAssets'][-1]['publicUrl'],
                'ie_key': 'ThePlatform',
            }
[nbc] Modernize 2014-02-24 21:00:31 +08:00			`from __future__ import unicode_literals`

Add an extractor for NBC news (closes #1320) 2013-08-27 18:38:30 +08:00			`import re`
[nbcnews] Add support for /feature/* pages (closes #3007) 2014-05-30 06:38:57 +08:00			`import json`
Add an extractor for NBC news (closes #1320) 2013-08-27 18:38:30 +08:00
			`from .common import InfoExtractor`
Fix imports and general cleanup · Import from compat what comes from compat. Yes, some names are available in utils too, but that's an implementation detail. · Use _match_id consistently whenever possible · Fix some outdated tests · Use consistent valid URL (always match the whole protocol, no ^ at start required) · Use modern test definitions 2014-12-13 19:24:42 +08:00			`from ..compat import (`
[nbc] Add missing import 2014-07-23 07:47:18 +08:00			`compat_str,`
[nbcnews] Ignore HTTP errors while coping with playlists (Closes #4749) 2015-01-20 23:23:51 +08:00			`compat_HTTPError,`
Fix imports and general cleanup · Import from compat what comes from compat. Yes, some names are available in utils too, but that's an implementation detail. · Use _match_id consistently whenever possible · Fix some outdated tests · Use consistent valid URL (always match the whole protocol, no ^ at start required) · Use modern test definitions 2014-12-13 19:24:42 +08:00			`)`
			`from ..utils import (`
[nbc] Add missing import 2014-07-23 07:47:18 +08:00			`ExtractorError,`
			`find_xpath_attr,`
			`)`
Add an extractor for NBC news (closes #1320) 2013-08-27 18:38:30 +08:00

[nbc] Add an extractor for the main nbc.com site Some of the videos are encrypted, the f4m downloader doesn’t support them. 2014-02-26 06:57:54 +08:00			`class NBCIE(InfoExtractor):`
[nbc] Fix extraction (Closes #4441) 2014-12-13 00:10:32 +08:00			`_VALID_URL = r'http://www\.nbc\.com/(?:[^/]+/)+(?P<id>n?\d+)'`

			`_TESTS = [`
			`{`
			`'url': 'http://www.nbc.com/chicago-fire/video/i-am-a-firefighter/2734188',`
			`# md5 checksum is not stable`
			`'info_dict': {`
			`'id': 'bTmnLCvIbaaH',`
			`'ext': 'flv',`
			`'title': 'I Am a Firefighter',`
			`'description': 'An emergency puts Dawson\'sf irefighter skills to the ultimate test in this four-part digital series.',`
			`},`
[nbc] Add an extractor for the main nbc.com site Some of the videos are encrypted, the f4m downloader doesn’t support them. 2014-02-26 06:57:54 +08:00			`},`
[nbc] Fix extraction (Closes #4441) 2014-12-13 00:10:32 +08:00			`{`
			`'url': 'http://www.nbc.com/the-tonight-show/episodes/176',`
			`'info_dict': {`
			`'id': 'XwU9KZkp98TH',`
			`'ext': 'flv',`
			`'title': 'Ricky Gervais, Steven Van Zandt, ILoveMakonnen',`
			`'description': 'A brand new episode of The Tonight Show welcomes Ricky Gervais, Steven Van Zandt and ILoveMakonnen.',`
			`},`
			`'skip': 'Only works from US',`
			`},`
			`]`
[nbc] Add an extractor for the main nbc.com site Some of the videos are encrypted, the f4m downloader doesn’t support them. 2014-02-26 06:57:54 +08:00
			`def _real_extract(self, url):`
[nbc] Fix ThePlatform embedded videos 2014-10-27 08:14:17 +08:00			`video_id = self._match_id(url)`
[nbc] Add an extractor for the main nbc.com site Some of the videos are encrypted, the f4m downloader doesn’t support them. 2014-02-26 06:57:54 +08:00			`webpage = self._download_webpage(url, video_id)`
[nbc] Fix extraction (Closes #4441) 2014-12-13 00:10:32 +08:00			`theplatform_url = self._search_regex(`
			`'(?:class="video-player video-player-full" data-mpx-url\|class="player" src)="(.*?)"',`
			`webpage, 'theplatform url').replace('_no_endcard', '')`
[nbc] Add an extractor for the main nbc.com site Some of the videos are encrypted, the f4m downloader doesn’t support them. 2014-02-26 06:57:54 +08:00			`if theplatform_url.startswith('//'):`
			`theplatform_url = 'http:' + theplatform_url`
			`return self.url_result(theplatform_url)`


Add an extractor for NBC news (closes #1320) 2013-08-27 18:38:30 +08:00			`class NBCNewsIE(InfoExtractor):`
[nbcnews] Add support for /feature/* pages (closes #3007) 2014-05-30 06:38:57 +08:00			`_VALID_URL = r'''(?x)https?://www\.nbcnews\.com/`
			`((video/.+?/(?P<id>\d+))\|`
			`(feature/[^/]+/(?P<title>.+)))`
			`'''`
Add an extractor for NBC news (closes #1320) 2013-08-27 18:38:30 +08:00
[nbcnews] Add support for /feature/* pages (closes #3007) 2014-05-30 06:38:57 +08:00			`_TESTS = [`
			`{`
			`'url': 'http://www.nbcnews.com/video/nbc-news/52753292',`
			`'md5': '47abaac93c6eaf9ad37ee6c4463a5179',`
			`'info_dict': {`
			`'id': '52753292',`
			`'ext': 'flv',`
			`'title': 'Crew emerges after four-month Mars food study',`
			`'description': 'md5:24e632ffac72b35f8b67a12d1b6ddfc1',`
			`},`
Add an extractor for NBC news (closes #1320) 2013-08-27 18:38:30 +08:00			`},`
[nbcnews] Add support for /feature/* pages (closes #3007) 2014-05-30 06:38:57 +08:00			`{`
			`'url': 'http://www.nbcnews.com/feature/edward-snowden-interview/how-twitter-reacted-snowden-interview-n117236',`
			`'md5': 'b2421750c9f260783721d898f4c42063',`
			`'info_dict': {`
			`'id': 'I1wpAI_zmhsQ',`
[nbc] Fix ThePlatform embedded videos 2014-10-27 08:14:17 +08:00			`'ext': 'mp4',`
[nbcnews] Add support for /feature/* pages (closes #3007) 2014-05-30 06:38:57 +08:00			`'title': 'How Twitter Reacted To The Snowden Interview',`
			`'description': 'md5:65a0bd5d76fe114f3c2727aa3a81fe64',`
			`},`
			`'add_ie': ['ThePlatform'],`
			`},`
[nbcnews] Ignore HTTP errors while coping with playlists (Closes #4749) 2015-01-20 23:23:51 +08:00			`{`
			`'url': 'http://www.nbcnews.com/feature/dateline-full-episodes/full-episode-family-business-n285156',`
			`'md5': 'fdbf39ab73a72df5896b6234ff98518a',`
			`'info_dict': {`
			`'id': 'Wjf9EDR3A_60',`
			`'ext': 'mp4',`
			`'title': 'FULL EPISODE: Family Business',`
			`'description': 'md5:757988edbaae9d7be1d585eb5d55cc04',`
			`},`
			`},`
[nbcnews] Add support for /feature/* pages (closes #3007) 2014-05-30 06:38:57 +08:00			`]`
Add an extractor for NBC news (closes #1320) 2013-08-27 18:38:30 +08:00
			`def _real_extract(self, url):`
			`mobj = re.match(self._VALID_URL, url)`
			`video_id = mobj.group('id')`
[nbcnews] Add support for /feature/* pages (closes #3007) 2014-05-30 06:38:57 +08:00			`if video_id is not None:`
			`all_info = self._download_xml('http://www.nbcnews.com/id/%s/displaymode/1219' % video_id, video_id)`
			`info = all_info.find('video')`
Add an extractor for NBC news (closes #1320) 2013-08-27 18:38:30 +08:00
[nbcnews] Add support for /feature/* pages (closes #3007) 2014-05-30 06:38:57 +08:00			`return {`
			`'id': video_id,`
			`'title': info.find('headline').text,`
			`'ext': 'flv',`
			`'url': find_xpath_attr(info, 'media', 'type', 'flashVideo').text,`
			`'description': compat_str(info.find('caption').text),`
			`'thumbnail': find_xpath_attr(info, 'media', 'type', 'thumbnail').text,`
			`}`
			`else:`
			`# "feature" pages use theplatform.com`
			`title = mobj.group('title')`
			`webpage = self._download_webpage(url, title)`
			`bootstrap_json = self._search_regex(`
			`r'var bootstrapJson = ({.+})\s*$', webpage, 'bootstrap json',`
			`flags=re.MULTILINE)`
			`bootstrap = json.loads(bootstrap_json)`
			`info = bootstrap['results'][0]['video']`
			`mpxid = info['mpxId']`
[nbcnews] Look in all playlists for video 2014-07-22 00:06:21 +08:00
			`base_urls = [`
			`info['fallbackPlaylistUrl'],`
			`info['associatedPlaylistUrl'],`
			`]`

			`for base_url in base_urls:`
[nbc] Fix ThePlatform embedded videos 2014-10-27 08:14:17 +08:00			`if not base_url:`
			`continue`
[nbcnews] Look in all playlists for video 2014-07-22 00:06:21 +08:00			`playlist_url = base_url + '?form=MPXNBCNewsAPI'`

			`try:`
[nbcnews] Ignore HTTP errors while coping with playlists (Closes #4749) 2015-01-20 23:23:51 +08:00			`all_videos = self._download_json(playlist_url, title)`
			`except ExtractorError as ee:`
			`if isinstance(ee.cause, compat_HTTPError):`
			`continue`
			`raise`

[nbc] Fix pep8 issue 2015-01-21 17:36:15 +08:00			`if not all_videos or 'videos' not in all_videos:`
[nbcnews] Ignore HTTP errors while coping with playlists (Closes #4749) 2015-01-20 23:23:51 +08:00			`continue`

			`try:`
			`info = next(v for v in all_videos['videos'] if v['mpxId'] == mpxid)`
[nbcnews] Look in all playlists for video 2014-07-22 00:06:21 +08:00			`break`
			`except StopIteration:`
			`continue`

			`if info is None:`
			`raise ExtractorError('Could not find video in playlists')`
[nbcnews] Add support for /feature/* pages (closes #3007) 2014-05-30 06:38:57 +08:00
			`return {`
			`'_type': 'url',`
			`# We get the best quality video`
			`'url': info['videoAssets'][-1]['publicUrl'],`
			`'ie_key': 'ThePlatform',`
			`}`