youtube-dl/youtube_dl/extractor/viewster.py

# coding: utf-8
from __future__ import unicode_literals

from .common import InfoExtractor
from ..compat import (
    compat_HTTPError,
    compat_urllib_request,
    compat_urllib_parse,
    compat_urllib_parse_unquote,
)
from ..utils import (
    determine_ext,
    ExtractorError,
    int_or_none,
    parse_iso8601,
    HEADRequest,
)


class ViewsterIE(InfoExtractor):
    _VALID_URL = r'https?://(?:www\.)?viewster\.com/(?:serie|movie)/(?P<id>\d+-\d+-\d+)'
    _TESTS = [{
        # movie, Type=Movie
        'url': 'http://www.viewster.com/movie/1140-11855-000/the-listening-project/',
        'md5': 'e642d1b27fcf3a4ffa79f194f5adde36',
        'info_dict': {
            'id': '1140-11855-000',
            'ext': 'mp4',
            'title': 'The listening Project',
            'description': 'md5:bac720244afd1a8ea279864e67baa071',
            'timestamp': 1214870400,
            'upload_date': '20080701',
            'duration': 4680,
        },
    }, {
        # series episode, Type=Episode
        'url': 'http://www.viewster.com/serie/1284-19427-001/the-world-and-a-wall/',
        'md5': '9243079a8531809efe1b089db102c069',
        'info_dict': {
            'id': '1284-19427-001',
            'ext': 'mp4',
            'title': 'The World and a Wall',
            'description': 'md5:24814cf74d3453fdf5bfef9716d073e3',
            'timestamp': 1428192000,
            'upload_date': '20150405',
            'duration': 1500,
        },
    }, {
        # serie, Type=Serie
        'url': 'http://www.viewster.com/serie/1303-19426-000/',
        'info_dict': {
            'id': '1303-19426-000',
            'title': 'Is It Wrong to Try to Pick up Girls in a Dungeon?',
            'description': 'md5:eeda9bef25b0d524b3a29a97804c2f11',
        },
        'playlist_count': 13,
    }, {
        # unfinished serie, no Type
        'url': 'http://www.viewster.com/serie/1284-19427-000/baby-steps-season-2/',
        'info_dict': {
            'id': '1284-19427-000',
            'title': 'Baby Steps—Season 2',
            'description': 'md5:e7097a8fc97151e25f085c9eb7a1cdb1',
        },
        'playlist_mincount': 16,
    }, {
        # geo restricted series
        'url': 'https://www.viewster.com/serie/1280-18794-002/',
        'only_matching': True,
    }, {
        # geo restricted video
        'url': 'https://www.viewster.com/serie/1280-18794-002/what-is-extraterritoriality-lawo/',
        'only_matching': True,
    }]

    _ACCEPT_HEADER = 'application/json, text/javascript, */*; q=0.01'

    def _download_json(self, url, video_id, note='Downloading JSON metadata', fatal=True):
        request = compat_urllib_request.Request(url)
        request.add_header('Accept', self._ACCEPT_HEADER)
        request.add_header('Auth-token', self._AUTH_TOKEN)
        return super(ViewsterIE, self)._download_json(request, video_id, note, fatal=fatal)

    def _real_extract(self, url):
        video_id = self._match_id(url)
        # Get 'api_token' cookie
        self._request_webpage(HEADRequest('http://www.viewster.com/'), video_id)
        cookies = self._get_cookies('http://www.viewster.com/')
        self._AUTH_TOKEN = compat_urllib_parse_unquote(cookies['api_token'].value)

        info = self._download_json(
            'https://public-api.viewster.com/search/%s' % video_id,
            video_id, 'Downloading entry JSON')

        entry_id = info.get('Id') or info['id']

        # unfinished serie has no Type
        if info.get('Type') in ('Serie', None):
            try:
                episodes = self._download_json(
                    'https://public-api.viewster.com/series/%s/episodes' % entry_id,
                    video_id, 'Downloading series JSON')
            except ExtractorError as e:
                if isinstance(e.cause, compat_HTTPError) and e.cause.code == 404:
                    self.raise_geo_restricted()
                else:
                    raise
            entries = [
                self.url_result(
                    'http://www.viewster.com/movie/%s' % episode['OriginId'], 'Viewster')
                for episode in episodes]
            title = (info.get('Title') or info['Synopsis']['Title']).strip()
            description = info.get('Synopsis', {}).get('Detailed')
            return self.playlist_result(entries, video_id, title, description)

        formats = []
        for media_type in ('application/f4m+xml', 'application/x-mpegURL', 'video/mp4'):
            media = self._download_json(
                'https://public-api.viewster.com/movies/%s/video?mediaType=%s'
                % (entry_id, compat_urllib_parse.quote(media_type)),
                video_id, 'Downloading %s JSON' % media_type, fatal=False)
            if not media:
                continue
            video_url = media.get('Uri')
            if not video_url:
                continue
            ext = determine_ext(video_url)
            if ext == 'f4m':
                video_url += '&' if '?' in video_url else '?'
                video_url += 'hdcore=3.2.0&plugin=flowplayer-3.2.0.1'
                formats.extend(self._extract_f4m_formats(
                    video_url, video_id, f4m_id='hds'))
            elif ext == 'm3u8':
                formats.extend(self._extract_m3u8_formats(
                    video_url, video_id, 'mp4', m3u8_id='hls',
                    fatal=False  # m3u8 sometimes fail
                ))
            else:
                format_id = media.get('Bitrate')
                f = {
                    'url': video_url,
                    'format_id': 'mp4-%s' % format_id,
                    'height': int_or_none(media.get('Height')),
                    'width': int_or_none(media.get('Width')),
                    'preference': 1,
                }
                if format_id and not f['height']:
                    f['height'] = int_or_none(self._search_regex(
                        r'^(\d+)[pP]$', format_id, 'height', default=None))
                formats.append(f)

        if not formats and not info.get('LanguageSets') and not info.get('VODSettings'):
            self.raise_geo_restricted()

        self._sort_formats(formats)

        synopsis = info.get('Synopsis', {})
        # Prefer title outside synopsis since it's less messy
        title = (info.get('Title') or synopsis['Title']).strip()
        description = synopsis.get('Detailed') or info.get('Synopsis', {}).get('Short')
        duration = int_or_none(info.get('Duration'))
        timestamp = parse_iso8601(info.get('ReleaseDate'))

        return {
            'id': video_id,
            'title': title,
            'description': description,
            'timestamp': timestamp,
            'duration': duration,
            'formats': formats,
        }
[viewster] Rewrite for new API (Closes #6317) 2015-07-21 20:00:21 +00:00			`# coding: utf-8`
[viewster] Add extractor 2015-03-13 20:12:11 +00:00			`from __future__ import unicode_literals`

			`from .common import InfoExtractor`
[viewster] Rewrite for new API (Closes #6317) 2015-07-21 20:00:21 +00:00			`from ..compat import (`
[voewster] Detect series geo restriction 2015-09-22 15:52:41 +00:00			`compat_HTTPError,`
[viewster] Rewrite for new API (Closes #6317) 2015-07-21 20:00:21 +00:00			`compat_urllib_request,`
			`compat_urllib_parse,`
[viewster] Use 'compat_urllib_parse_unquote' 2015-07-30 17:12:37 +00:00			`compat_urllib_parse_unquote,`
[viewster] Rewrite for new API (Closes #6317) 2015-07-21 20:00:21 +00:00			`)`
			`from ..utils import (`
			`determine_ext,`
[voewster] Detect series geo restriction 2015-09-22 15:52:41 +00:00			`ExtractorError,`
[viewster] Rewrite for new API (Closes #6317) 2015-07-21 20:00:21 +00:00			`int_or_none,`
			`parse_iso8601,`
[viewster] use head request to extract api token Closes #6419. 2015-07-31 13:41:30 +00:00			`HEADRequest,`
[viewster] Rewrite for new API (Closes #6317) 2015-07-21 20:00:21 +00:00			`)`
[viewster] Add extractor 2015-03-13 20:12:11 +00:00

			`class ViewsterIE(InfoExtractor):`
[viewster] accept https links and fix api_token extraction and extract mp4 video link(fixes #6787) 2015-09-20 21:26:23 +00:00			`_VALID_URL = r'https?://(?:www\.)?viewster\.com/(?:serie\|movie)/(?P<id>\d+-\d+-\d+)'`
[viewster] Improve extraction 2015-03-13 21:18:04 +00:00			`_TESTS = [{`
[viewster] Rewrite for new API (Closes #6317) 2015-07-21 20:00:21 +00:00			`# movie, Type=Movie`
[viewster] Improve extraction 2015-03-13 21:18:04 +00:00			`'url': 'http://www.viewster.com/movie/1140-11855-000/the-listening-project/',`
[voewster] Update tests 2015-09-22 15:49:29 +00:00			`'md5': 'e642d1b27fcf3a4ffa79f194f5adde36',`
[viewster] Improve extraction 2015-03-13 21:18:04 +00:00			`'info_dict': {`
			`'id': '1140-11855-000',`
[voewster] Update tests 2015-09-22 15:49:29 +00:00			`'ext': 'mp4',`
[viewster] Rewrite for new API (Closes #6317) 2015-07-21 20:00:21 +00:00			`'title': 'The listening Project',`
			`'description': 'md5:bac720244afd1a8ea279864e67baa071',`
			`'timestamp': 1214870400,`
			`'upload_date': '20080701',`
			`'duration': 4680,`
			`},`
[viewster] Improve extraction 2015-03-13 21:18:04 +00:00			`}, {`
[viewster] Rewrite for new API (Closes #6317) 2015-07-21 20:00:21 +00:00			`# series episode, Type=Episode`
			`'url': 'http://www.viewster.com/serie/1284-19427-001/the-world-and-a-wall/',`
[voewster] Update tests 2015-09-22 15:49:29 +00:00			`'md5': '9243079a8531809efe1b089db102c069',`
[viewster] Improve extraction 2015-03-13 21:18:04 +00:00			`'info_dict': {`
[viewster] Rewrite for new API (Closes #6317) 2015-07-21 20:00:21 +00:00			`'id': '1284-19427-001',`
[voewster] Update tests 2015-09-22 15:49:29 +00:00			`'ext': 'mp4',`
[viewster] Rewrite for new API (Closes #6317) 2015-07-21 20:00:21 +00:00			`'title': 'The World and a Wall',`
			`'description': 'md5:24814cf74d3453fdf5bfef9716d073e3',`
			`'timestamp': 1428192000,`
			`'upload_date': '20150405',`
			`'duration': 1500,`
			`},`
			`}, {`
			`# serie, Type=Serie`
			`'url': 'http://www.viewster.com/serie/1303-19426-000/',`
			`'info_dict': {`
			`'id': '1303-19426-000',`
			`'title': 'Is It Wrong to Try to Pick up Girls in a Dungeon?',`
			`'description': 'md5:eeda9bef25b0d524b3a29a97804c2f11',`
			`},`
			`'playlist_count': 13,`
			`}, {`
			`# unfinished serie, no Type`
			`'url': 'http://www.viewster.com/serie/1284-19427-000/baby-steps-season-2/',`
			`'info_dict': {`
			`'id': '1284-19427-000',`
			`'title': 'Baby Steps—Season 2',`
			`'description': 'md5:e7097a8fc97151e25f085c9eb7a1cdb1',`
			`},`
			`'playlist_mincount': 16,`
[viewster] Add geo restricted tests 2015-09-22 15:55:04 +00:00			`}, {`
			`# geo restricted series`
			`'url': 'https://www.viewster.com/serie/1280-18794-002/',`
			`'only_matching': True,`
			`}, {`
			`# geo restricted video`
			`'url': 'https://www.viewster.com/serie/1280-18794-002/what-is-extraterritoriality-lawo/',`
			`'only_matching': True,`
[viewster] Improve extraction 2015-03-13 21:18:04 +00:00			`}]`
[viewster] Add extractor 2015-03-13 20:12:11 +00:00
			`_ACCEPT_HEADER = 'application/json, text/javascript, /; q=0.01'`

[viewster] Rewrite for new API (Closes #6317) 2015-07-21 20:00:21 +00:00			`def _download_json(self, url, video_id, note='Downloading JSON metadata', fatal=True):`
			`request = compat_urllib_request.Request(url)`
[viewster] Add extractor 2015-03-13 20:12:11 +00:00			`request.add_header('Accept', self._ACCEPT_HEADER)`
[viewster] Rewrite for new API (Closes #6317) 2015-07-21 20:00:21 +00:00			`request.add_header('Auth-token', self._AUTH_TOKEN)`
			`return super(ViewsterIE, self)._download_json(request, video_id, note, fatal=fatal)`
[viewster] Add extractor 2015-03-13 20:12:11 +00:00
[viewster] Rewrite for new API (Closes #6317) 2015-07-21 20:00:21 +00:00			`def _real_extract(self, url):`
			`video_id = self._match_id(url)`
[viewster] extract the api auth token Closes #6406. 2015-07-29 22:20:37 +00:00			`# Get 'api_token' cookie`
[viewster] accept https links and fix api_token extraction and extract mp4 video link(fixes #6787) 2015-09-20 21:26:23 +00:00			`self._request_webpage(HEADRequest('http://www.viewster.com/'), video_id)`
			`cookies = self._get_cookies('http://www.viewster.com/')`
[viewster] Use 'compat_urllib_parse_unquote' 2015-07-30 17:12:37 +00:00			`self._AUTH_TOKEN = compat_urllib_parse_unquote(cookies['api_token'].value)`
[viewster] Improve extraction 2015-03-13 21:18:04 +00:00
[viewster] Rewrite for new API (Closes #6317) 2015-07-21 20:00:21 +00:00			`info = self._download_json(`
			`'https://public-api.viewster.com/search/%s' % video_id,`
			`video_id, 'Downloading entry JSON')`
[viewster] Improve extraction 2015-03-13 21:18:04 +00:00
[viewster] Rewrite for new API (Closes #6317) 2015-07-21 20:00:21 +00:00			`entry_id = info.get('Id') or info['id']`
[viewster] Improve extraction 2015-03-13 21:18:04 +00:00
[viewster] Rewrite for new API (Closes #6317) 2015-07-21 20:00:21 +00:00			`# unfinished serie has no Type`
[viewster] Use tuple 2015-09-22 16:00:50 +00:00			`if info.get('Type') in ('Serie', None):`
[voewster] Detect series geo restriction 2015-09-22 15:52:41 +00:00			`try:`
			`episodes = self._download_json(`
			`'https://public-api.viewster.com/series/%s/episodes' % entry_id,`
			`video_id, 'Downloading series JSON')`
			`except ExtractorError as e:`
			`if isinstance(e.cause, compat_HTTPError) and e.cause.code == 404:`
			`self.raise_geo_restricted()`
			`else:`
			`raise`
[viewster] Rewrite for new API (Closes #6317) 2015-07-21 20:00:21 +00:00			`entries = [`
			`self.url_result(`
			`'http://www.viewster.com/movie/%s' % episode['OriginId'], 'Viewster')`
			`for episode in episodes]`
[viewster] Strip titles 2015-07-21 20:08:25 +00:00			`title = (info.get('Title') or info['Synopsis']['Title']).strip()`
[viewster] Rewrite for new API (Closes #6317) 2015-07-21 20:00:21 +00:00			`description = info.get('Synopsis', {}).get('Detailed')`
			`return self.playlist_result(entries, video_id, title, description)`
[viewster] Improve extraction 2015-03-13 21:18:04 +00:00
[viewster] Rewrite for new API (Closes #6317) 2015-07-21 20:00:21 +00:00			`formats = []`
[viewster] accept https links and fix api_token extraction and extract mp4 video link(fixes #6787) 2015-09-20 21:26:23 +00:00			`for media_type in ('application/f4m+xml', 'application/x-mpegURL', 'video/mp4'):`
[viewster] Rewrite for new API (Closes #6317) 2015-07-21 20:00:21 +00:00			`media = self._download_json(`
			`'https://public-api.viewster.com/movies/%s/video?mediaType=%s'`
			`% (entry_id, compat_urllib_parse.quote(media_type)),`
			`video_id, 'Downloading %s JSON' % media_type, fatal=False)`
			`if not media:`
			`continue`
			`video_url = media.get('Uri')`
			`if not video_url:`
			`continue`
			`ext = determine_ext(video_url)`
			`if ext == 'f4m':`
			`video_url += '&' if '?' in video_url else '?'`
			`video_url += 'hdcore=3.2.0&plugin=flowplayer-3.2.0.1'`
			`formats.extend(self._extract_f4m_formats(`
			`video_url, video_id, f4m_id='hds'))`
			`elif ext == 'm3u8':`
			`formats.extend(self._extract_m3u8_formats(`
			`video_url, video_id, 'mp4', m3u8_id='hls',`
			`fatal=False # m3u8 sometimes fail`
			`))`
			`else:`
[viewster] Extract height from bitrate and prefer mp4 videos 2015-09-22 15:47:56 +00:00			`format_id = media.get('Bitrate')`
			`f = {`
[viewster] Rewrite for new API (Closes #6317) 2015-07-21 20:00:21 +00:00			`'url': video_url,`
[viewster] Extract height from bitrate and prefer mp4 videos 2015-09-22 15:47:56 +00:00			`'format_id': 'mp4-%s' % format_id,`
[viewster] accept https links and fix api_token extraction and extract mp4 video link(fixes #6787) 2015-09-20 21:26:23 +00:00			`'height': int_or_none(media.get('Height')),`
			`'width': int_or_none(media.get('Width')),`
[viewster] Extract height from bitrate and prefer mp4 videos 2015-09-22 15:47:56 +00:00			`'preference': 1,`
			`}`
			`if format_id and not f['height']:`
			`f['height'] = int_or_none(self._search_regex(`
			`r'^(\d+)[pP]$', format_id, 'height', default=None))`
			`formats.append(f)`
[viewster] Detect video geo restriction 2015-09-22 15:54:32 +00:00
			`if not formats and not info.get('LanguageSets') and not info.get('VODSettings'):`
			`self.raise_geo_restricted()`

[viewster] Rewrite for new API (Closes #6317) 2015-07-21 20:00:21 +00:00			`self._sort_formats(formats)`
[viewster] Improve extraction 2015-03-13 21:18:04 +00:00
[viewster] Rewrite for new API (Closes #6317) 2015-07-21 20:00:21 +00:00			`synopsis = info.get('Synopsis', {})`
			`# Prefer title outside synopsis since it's less messy`
[viewster] Strip titles 2015-07-21 20:08:25 +00:00			`title = (info.get('Title') or synopsis['Title']).strip()`
[viewster] Rewrite for new API (Closes #6317) 2015-07-21 20:00:21 +00:00			`description = synopsis.get('Detailed') or info.get('Synopsis', {}).get('Short')`
			`duration = int_or_none(info.get('Duration'))`
			`timestamp = parse_iso8601(info.get('ReleaseDate'))`

			`return {`
			`'id': video_id,`
			`'title': title,`
			`'description': description,`
			`'timestamp': timestamp,`
			`'duration': duration,`
			`'formats': formats,`
			`}`