youtube-dl/youtube_dl/extractor/cspan.py

from __future__ import unicode_literals

import json
import re

from .common import InfoExtractor
from ..utils import (
    unescapeHTML,
)


class CSpanIE(InfoExtractor):
    _VALID_URL = r'http://(?:www\.)?c-spanvideo\.org/program/(?P<name>.*)'
    IE_DESC = 'C-SPAN'
    _TEST = {
        'url': 'http://www.c-spanvideo.org/program/HolderonV',
        'file': '315139.mp4',
        'md5': '8e44ce11f0f725527daccc453f553eb0',
        'info_dict': {
            'title': 'Attorney General Eric Holder on Voting Rights Act Decision',
            'description': 'Attorney General Eric Holder spoke to reporters following the Supreme Court decision in [Shelby County v. Holder] in which the court ruled that the preclearance provisions of the Voting Rights Act could not be enforced until Congress established new guidelines for review.',
        },
        'skip': 'Regularly fails on travis, for unknown reasons',
    }

    def _real_extract(self, url):
        mobj = re.match(self._VALID_URL, url)
        prog_name = mobj.group('name')
        webpage = self._download_webpage(url, prog_name)
        video_id = self._search_regex(r'prog(?:ram)?id=(.*?)&', webpage, 'video id')

        title = self._html_search_regex(
            r'<!-- title -->\n\s*<h1[^>]*>(.*?)</h1>', webpage, 'title')
        description = self._og_search_description(webpage)

        info_url = 'http://c-spanvideo.org/videoLibrary/assets/player/ajax-player.php?os=android&html5=program&id=' + video_id
        data_json = self._download_webpage(
            info_url, video_id, 'Downloading video info')
        data = json.loads(data_json)

        url = unescapeHTML(data['video']['files'][0]['path']['#text'])

        return {
            'id': video_id,
            'title': title,
            'url': url,
            'description': description,
            'thumbnail': self._og_search_thumbnail(webpage),
        }
[cspan] Use HTTP download (Fixes #2098) 2014-01-05 11:30:00 +08:00			`from __future__ import unicode_literals`

			`import json`
Add CSpanIE (closes #312) 2013-06-26 23:55:54 +08:00			`import re`

			`from .common import InfoExtractor`
			`from ..utils import (`
[cspan] Use HTTP download (Fixes #2098) 2014-01-05 11:30:00 +08:00			`unescapeHTML,`
Add CSpanIE (closes #312) 2013-06-26 23:55:54 +08:00			`)`

[cspan] Use HTTP download (Fixes #2098) 2014-01-05 11:30:00 +08:00
Add CSpanIE (closes #312) 2013-06-26 23:55:54 +08:00			`class CSpanIE(InfoExtractor):`
[cspan] Make ‘www’ optional and improve the regex for extracting the id (fixes #2194) 2014-01-22 18:06:03 +08:00			`_VALID_URL = r'http://(?:www\.)?c-spanvideo\.org/program/(?P<name>.*)'`
[cspan] Use HTTP download (Fixes #2098) 2014-01-05 11:30:00 +08:00			`IE_DESC = 'C-SPAN'`
Move tests to the IE definitions 2013-06-28 02:46:46 +08:00			`_TEST = {`
[cspan] Use HTTP download (Fixes #2098) 2014-01-05 11:30:00 +08:00			`'url': 'http://www.c-spanvideo.org/program/HolderonV',`
			`'file': '315139.mp4',`
			`'md5': '8e44ce11f0f725527daccc453f553eb0',`
			`'info_dict': {`
			`'title': 'Attorney General Eric Holder on Voting Rights Act Decision',`
			`'description': 'Attorney General Eric Holder spoke to reporters following the Supreme Court decision in [Shelby County v. Holder] in which the court ruled that the preclearance provisions of the Voting Rights Act could not be enforced until Congress established new guidelines for review.',`
Move tests to the IE definitions 2013-06-28 02:46:46 +08:00			`},`
[cspan] Disable test It works fine from all my machines, no matter where, but from travis, we get lots of 403s. Maybe another project is scraping CSPAN from travis and they're blocking the travis machines? 2014-01-22 22:10:00 +08:00			`'skip': 'Regularly fails on travis, for unknown reasons',`
Move tests to the IE definitions 2013-06-28 02:46:46 +08:00			`}`
Add CSpanIE (closes #312) 2013-06-26 23:55:54 +08:00
			`def _real_extract(self, url):`
			`mobj = re.match(self._VALID_URL, url)`
[cspan] Make ‘www’ optional and improve the regex for extracting the id (fixes #2194) 2014-01-22 18:06:03 +08:00			`prog_name = mobj.group('name')`
Add CSpanIE (closes #312) 2013-06-26 23:55:54 +08:00			`webpage = self._download_webpage(url, prog_name)`
[cspan] Make ‘www’ optional and improve the regex for extracting the id (fixes #2194) 2014-01-22 18:06:03 +08:00			`video_id = self._search_regex(r'prog(?:ram)?id=(.*?)&', webpage, 'video id')`
[cspan] Use HTTP download (Fixes #2098) 2014-01-05 11:30:00 +08:00
			`title = self._html_search_regex(`
			`r'<!-- title -->\n\s<h1[^>]>(.*?)</h1>', webpage, 'title')`
			`description = self._og_search_description(webpage)`

			`info_url = 'http://c-spanvideo.org/videoLibrary/assets/player/ajax-player.php?os=android&html5=program&id=' + video_id`
			`data_json = self._download_webpage(`
			`info_url, video_id, 'Downloading video info')`
			`data = json.loads(data_json)`

			`url = unescapeHTML(data['video']['files'][0]['path']['#text'])`

			`return {`
			`'id': video_id,`
			`'title': title,`
			`'url': url,`
			`'description': description,`
			`'thumbnail': self._og_search_thumbnail(webpage),`
			`}`