youtube-dl/youtube_dl/extractor/naver.py

# encoding: utf-8
from __future__ import unicode_literals

import re

from .common import InfoExtractor
from ..compat import (
    compat_urllib_parse,
    compat_urlparse,
)
from ..utils import (
    ExtractorError,
    clean_html,
)


class NaverIE(InfoExtractor):
    _VALID_URL = r'https?://(?:m\.)?tvcast\.naver\.com/v/(?P<id>\d+)'

    _TESTS = [{
        'url': 'http://tvcast.naver.com/v/81652',
        'info_dict': {
            'id': '81652',
            'ext': 'mp4',
            'title': '[9월 모의고사 해설강의][수학_김상희] 수학 A형 16~20번',
            'description': '합격불변의 법칙 메가스터디 | 메가스터디 수학 김상희 선생님이 9월 모의고사 수학A형 16번에서 20번까지 해설강의를 공개합니다.',
            'upload_date': '20130903',
        },
    }, {
        'url': 'http://tvcast.naver.com/v/395837',
        'md5': '638ed4c12012c458fefcddfd01f173cd',
        'info_dict': {
            'id': '395837',
            'ext': 'mp4',
            'title': '9년이 지나도 아픈 기억, 전효성의 아버지',
            'description': 'md5:5bf200dcbf4b66eb1b350d1eb9c753f7',
            'upload_date': '20150519',
        },
        'skip': 'Georestricted',
    }]

    def _real_extract(self, url):
        video_id = self._match_id(url)
        webpage = self._download_webpage(url, video_id)

        m_id = re.search(r'var rmcPlayer = new nhn.rmcnmv.RMCVideoPlayer\("(.+?)", "(.+?)"',
                         webpage)
        if m_id is None:
            m_error = re.search(
                r'(?s)<div class="(?:nation_error|nation_box)">\s*(?:<!--.*?-->)?\s*<p class="[^"]+">(?P<msg>.+?)</p>\s*</div>',
                webpage)
            if m_error:
                raise ExtractorError(clean_html(m_error.group('msg')), expected=True)
            raise ExtractorError('couldn\'t extract vid and key')
        vid = m_id.group(1)
        key = m_id.group(2)
        query = compat_urllib_parse.urlencode({'vid': vid, 'inKey': key, })
        query_urls = compat_urllib_parse.urlencode({
            'masterVid': vid,
            'protocol': 'p2p',
            'inKey': key,
        })
        info = self._download_xml(
            'http://serviceapi.rmcnmv.naver.com/flash/videoInfo.nhn?' + query,
            video_id, 'Downloading video info')
        urls = self._download_xml(
            'http://serviceapi.rmcnmv.naver.com/flash/playableEncodingOption.nhn?' + query_urls,
            video_id, 'Downloading video formats info')

        formats = []
        for format_el in urls.findall('EncodingOptions/EncodingOption'):
            domain = format_el.find('Domain').text
            uri = format_el.find('uri').text
            f = {
                'url': compat_urlparse.urljoin(domain, uri),
                'ext': 'mp4',
                'width': int(format_el.find('width').text),
                'height': int(format_el.find('height').text),
            }
            if domain.startswith('rtmp'):
                # urlparse does not support custom schemes
                # https://bugs.python.org/issue18828
                f.update({
                    'url': domain + uri,
                    'ext': 'flv',
                    'rtmp_protocol': '1',  # rtmpt
                })
            formats.append(f)
        self._sort_formats(formats)

        return {
            'id': video_id,
            'title': info.find('Subject').text,
            'formats': formats,
            'description': self._og_search_description(webpage),
            'thumbnail': self._og_search_thumbnail(webpage),
            'upload_date': info.find('WriteDate').text.replace('.', ''),
            'view_count': int(info.find('PlayCount').text),
        }
Add extractor for tvcast.naver.com (closes #1331) 2013-09-05 16:53:40 +08:00			`# encoding: utf-8`
[naver] Modernize 2014-06-06 20:57:37 +08:00			`from __future__ import unicode_literals`

Add extractor for tvcast.naver.com (closes #1331) 2013-09-05 16:53:40 +08:00			`import re`

			`from .common import InfoExtractor`
Fix imports and general cleanup · Import from compat what comes from compat. Yes, some names are available in utils too, but that's an implementation detail. · Use _match_id consistently whenever possible · Fix some outdated tests · Use consistent valid URL (always match the whole protocol, no ^ at start required) · Use modern test definitions 2014-12-13 19:24:42 +08:00			`from ..compat import (`
Add extractor for tvcast.naver.com (closes #1331) 2013-09-05 16:53:40 +08:00			`compat_urllib_parse,`
[naver] Fix video url (fixes #5809) RTMP urls in test:naver does not work. Need more investigation. 2015-05-27 14:44:08 +08:00			`compat_urlparse,`
Fix imports and general cleanup · Import from compat what comes from compat. Yes, some names are available in utils too, but that's an implementation detail. · Use _match_id consistently whenever possible · Fix some outdated tests · Use consistent valid URL (always match the whole protocol, no ^ at start required) · Use modern test definitions 2014-12-13 19:24:42 +08:00			`)`
			`from ..utils import (`
Add extractor for tvcast.naver.com (closes #1331) 2013-09-05 16:53:40 +08:00			`ExtractorError,`
[naver] Capture and output error message (#4057) 2014-10-29 22:50:37 +08:00			`clean_html,`
Add extractor for tvcast.naver.com (closes #1331) 2013-09-05 16:53:40 +08:00			`)`


			`class NaverIE(InfoExtractor):`
[naver] Recognize mobile urls (fixes #1951) 2013-12-12 20:04:02 +08:00			`_VALID_URL = r'https?://(?:m\.)?tvcast\.naver\.com/v/(?P<id>\d+)'`
Add extractor for tvcast.naver.com (closes #1331) 2013-09-05 16:53:40 +08:00
[naver] Fix video url (fixes #5809) RTMP urls in test:naver does not work. Need more investigation. 2015-05-27 14:44:08 +08:00			`_TESTS = [{`
[naver] Modernize 2014-06-06 20:57:37 +08:00			`'url': 'http://tvcast.naver.com/v/81652',`
			`'info_dict': {`
			`'id': '81652',`
			`'ext': 'mp4',`
			`'title': '[9월 모의고사 해설강의][수학_김상희] 수학 A형 16~20번',`
			`'description': '합격불변의 법칙 메가스터디 \| 메가스터디 수학 김상희 선생님이 9월 모의고사 수학A형 16번에서 20번까지 해설강의를 공개합니다.',`
			`'upload_date': '20130903',`
Add extractor for tvcast.naver.com (closes #1331) 2013-09-05 16:53:40 +08:00			`},`
[naver] Fix video url (fixes #5809) RTMP urls in test:naver does not work. Need more investigation. 2015-05-27 14:44:08 +08:00			`}, {`
			`'url': 'http://tvcast.naver.com/v/395837',`
			`'md5': '638ed4c12012c458fefcddfd01f173cd',`
			`'info_dict': {`
			`'id': '395837',`
			`'ext': 'mp4',`
			`'title': '9년이 지나도 아픈 기억, 전효성의 아버지',`
			`'description': 'md5:5bf200dcbf4b66eb1b350d1eb9c753f7',`
			`'upload_date': '20150519',`
			`},`
			`'skip': 'Georestricted',`
			`}]`
Add extractor for tvcast.naver.com (closes #1331) 2013-09-05 16:53:40 +08:00
			`def _real_extract(self, url):`
Fix imports and general cleanup · Import from compat what comes from compat. Yes, some names are available in utils too, but that's an implementation detail. · Use _match_id consistently whenever possible · Fix some outdated tests · Use consistent valid URL (always match the whole protocol, no ^ at start required) · Use modern test definitions 2014-12-13 19:24:42 +08:00			`video_id = self._match_id(url)`
Add extractor for tvcast.naver.com (closes #1331) 2013-09-05 16:53:40 +08:00			`webpage = self._download_webpage(url, video_id)`
Fix imports and general cleanup · Import from compat what comes from compat. Yes, some names are available in utils too, but that's an implementation detail. · Use _match_id consistently whenever possible · Fix some outdated tests · Use consistent valid URL (always match the whole protocol, no ^ at start required) · Use modern test definitions 2014-12-13 19:24:42 +08:00
Add extractor for tvcast.naver.com (closes #1331) 2013-09-05 16:53:40 +08:00			`m_id = re.search(r'var rmcPlayer = new nhn.rmcnmv.RMCVideoPlayer\("(.+?)", "(.+?)"',`
PEP8: applied even more rules 2014-11-24 04:39:15 +08:00			`webpage)`
Add extractor for tvcast.naver.com (closes #1331) 2013-09-05 16:53:40 +08:00			`if m_id is None:`
[naver] Capture and output error message (#4057) 2014-10-29 22:50:37 +08:00			`m_error = re.search(`
[naver] Enhanced error detection 2015-05-27 14:20:29 +08:00			`r'(?s)<div class="(?:nation_error\|nation_box)">\s(?:<!--.?-->)?\s<p class="[^"]+">(?P<msg>.+?)</p>\s</div>',`
[naver] Capture and output error message (#4057) 2014-10-29 22:50:37 +08:00			`webpage)`
			`if m_error:`
			`raise ExtractorError(clean_html(m_error.group('msg')), expected=True)`
[naver] Modernize 2014-06-06 20:57:37 +08:00			`raise ExtractorError('couldn\'t extract vid and key')`
Add extractor for tvcast.naver.com (closes #1331) 2013-09-05 16:53:40 +08:00			`vid = m_id.group(1)`
			`key = m_id.group(2)`
PEP8 applied 2014-11-24 03:41:03 +08:00			`query = compat_urllib_parse.urlencode({'vid': vid, 'inKey': key, })`
Add extractor for tvcast.naver.com (closes #1331) 2013-09-05 16:53:40 +08:00			`query_urls = compat_urllib_parse.urlencode({`
			`'masterVid': vid,`
			`'protocol': 'p2p',`
			`'inKey': key,`
			`})`
Use the new '_download_xml' helper in more extractors 2013-11-27 01:48:52 +08:00			`info = self._download_xml(`
Add extractor for tvcast.naver.com (closes #1331) 2013-09-05 16:53:40 +08:00			`'http://serviceapi.rmcnmv.naver.com/flash/videoInfo.nhn?' + query,`
[naver] Modernize 2014-06-06 20:57:37 +08:00			`video_id, 'Downloading video info')`
Use the new '_download_xml' helper in more extractors 2013-11-27 01:48:52 +08:00			`urls = self._download_xml(`
Add extractor for tvcast.naver.com (closes #1331) 2013-09-05 16:53:40 +08:00			`'http://serviceapi.rmcnmv.naver.com/flash/playableEncodingOption.nhn?' + query_urls,`
[naver] Modernize 2014-06-06 20:57:37 +08:00			`video_id, 'Downloading video formats info')`
Add extractor for tvcast.naver.com (closes #1331) 2013-09-05 16:53:40 +08:00
			`formats = []`
			`for format_el in urls.findall('EncodingOptions/EncodingOption'):`
			`domain = format_el.find('Domain').text`
[naver] Fix video url (fixes #5809) RTMP urls in test:naver does not work. Need more investigation. 2015-05-27 14:44:08 +08:00			`uri = format_el.find('uri').text`
[naver] Add rtmp formats (fixes #3054) 2014-06-06 20:55:19 +08:00			`f = {`
[naver] Fix video url (fixes #5809) RTMP urls in test:naver does not work. Need more investigation. 2015-05-27 14:44:08 +08:00			`'url': compat_urlparse.urljoin(domain, uri),`
Add extractor for tvcast.naver.com (closes #1331) 2013-09-05 16:53:40 +08:00			`'ext': 'mp4',`
			`'width': int(format_el.find('width').text),`
			`'height': int(format_el.find('height').text),`
[naver] Add rtmp formats (fixes #3054) 2014-06-06 20:55:19 +08:00			`}`
			`if domain.startswith('rtmp'):`
[naver] Fix video url (fixes #5809) RTMP urls in test:naver does not work. Need more investigation. 2015-05-27 14:44:08 +08:00			`# urlparse does not support custom schemes`
			`# https://bugs.python.org/issue18828`
[naver] Add rtmp formats (fixes #3054) 2014-06-06 20:55:19 +08:00			`f.update({`
[naver] Fix video url (fixes #5809) RTMP urls in test:naver does not work. Need more investigation. 2015-05-27 14:44:08 +08:00			`'url': domain + uri,`
[naver] Add rtmp formats (fixes #3054) 2014-06-06 20:55:19 +08:00			`'ext': 'flv',`
PEP8 applied 2014-11-24 03:41:03 +08:00			`'rtmp_protocol': '1', # rtmpt`
[naver] Add rtmp formats (fixes #3054) 2014-06-06 20:55:19 +08:00			`})`
			`formats.append(f)`
			`self._sort_formats(formats)`
Add extractor for tvcast.naver.com (closes #1331) 2013-09-05 16:53:40 +08:00
Remove the compatibility code used before the new format system was implemented 2013-12-03 21:21:06 +08:00			`return {`
Add extractor for tvcast.naver.com (closes #1331) 2013-09-05 16:53:40 +08:00			`'id': video_id,`
			`'title': info.find('Subject').text,`
			`'formats': formats,`
			`'description': self._og_search_description(webpage),`
			`'thumbnail': self._og_search_thumbnail(webpage),`
			`'upload_date': info.find('WriteDate').text.replace('.', ''),`
			`'view_count': int(info.find('PlayCount').text),`
			`}`