yt-dlc/youtube_dl/extractor/indavideo.py

# coding: utf-8
from __future__ import unicode_literals

import re

from .common import InfoExtractor
from ..compat import compat_str
from ..utils import (
    int_or_none,
    parse_age_limit,
    parse_iso8601,
    update_url_query,
)


class IndavideoEmbedIE(InfoExtractor):
    _VALID_URL = r'https?://(?:(?:embed\.)?indavideo\.hu/player/video/|assets\.indavideo\.hu/swf/player\.swf\?.*\b(?:v(?:ID|id))=)(?P<id>[\da-f]+)'
    _TESTS = [{
        'url': 'http://indavideo.hu/player/video/1bdc3c6d80/',
        'md5': 'c8a507a1c7410685f83a06eaeeaafeab',
        'info_dict': {
            'id': '1837039',
            'ext': 'mp4',
            'title': 'Cicatánc',
            'description': '',
            'thumbnail': r're:^https?://.*\.jpg$',
            'uploader': 'cukiajanlo',
            'uploader_id': '83729',
            'timestamp': 1439193826,
            'upload_date': '20150810',
            'duration': 72,
            'age_limit': 0,
            'tags': ['tánc', 'cica', 'cuki', 'cukiajanlo', 'newsroom'],
        },
    }, {
        'url': 'http://embed.indavideo.hu/player/video/1bdc3c6d80?autostart=1&hide=1',
        'only_matching': True,
    }, {
        'url': 'http://assets.indavideo.hu/swf/player.swf?v=fe25e500&vID=1bdc3c6d80&autostart=1&hide=1&i=1',
        'only_matching': True,
    }]

    # Some example URLs covered by generic extractor:
    #   http://indavideo.hu/video/Vicces_cica_1
    #   http://index.indavideo.hu/video/2015_0728_beregszasz
    #   http://auto.indavideo.hu/video/Sajat_utanfutoban_a_kis_tacsko
    #   http://erotika.indavideo.hu/video/Amator_tini_punci
    #   http://film.indavideo.hu/video/f_hrom_nagymamm_volt
    #   http://palyazat.indavideo.hu/video/Embertelen_dal_Dodgem_egyuttes

    @staticmethod
    def _extract_urls(webpage):
        return re.findall(
            r'<iframe[^>]+\bsrc=["\'](?P<url>(?:https?:)?//embed\.indavideo\.hu/player/video/[\da-f]+)',
            webpage)

    def _real_extract(self, url):
        video_id = self._match_id(url)

        video = self._download_json(
            'http://amfphp.indavideo.hu/SYm0json.php/player.playerHandler.getVideoData/%s' % video_id,
            video_id)['data']

        title = video['title']

        video_urls = []

        video_files = video.get('video_files')
        if isinstance(video_files, list):
            video_urls.extend(video_files)
        elif isinstance(video_files, dict):
            video_urls.extend(video_files.values())

        video_file = video.get('video_file')
        if video:
            video_urls.append(video_file)
        video_urls = list(set(video_urls))

        video_prefix = video_urls[0].rsplit('/', 1)[0]

        for flv_file in video.get('flv_files', []):
            flv_url = '%s/%s' % (video_prefix, flv_file)
            if flv_url not in video_urls:
                video_urls.append(flv_url)

        filesh = video.get('filesh')

        formats = []
        for video_url in video_urls:
            height = int_or_none(self._search_regex(
                r'\.(\d{3,4})\.mp4(?:\?|$)', video_url, 'height', default=None))
            if filesh:
                if not height:
                    continue
                token = filesh.get(compat_str(height))
                if token is None:
                    continue
                video_url = update_url_query(video_url, {'token': token})
            formats.append({
                'url': video_url,
                'height': height,
            })
        self._sort_formats(formats)

        timestamp = video.get('date')
        if timestamp:
            # upload date is in CEST
            timestamp = parse_iso8601(timestamp + ' +0200', ' ')

        thumbnails = [{
            'url': self._proto_relative_url(thumbnail)
        } for thumbnail in video.get('thumbnails', [])]

        tags = [tag['title'] for tag in video.get('tags') or []]

        return {
            'id': video.get('id') or video_id,
            'title': title,
            'description': video.get('description'),
            'thumbnails': thumbnails,
            'uploader': video.get('user_name'),
            'uploader_id': video.get('user_id'),
            'timestamp': timestamp,
            'duration': int_or_none(video.get('length')),
            'age_limit': parse_age_limit(video.get('age_limit')),
            'tags': tags,
            'formats': formats,
        }
[indavideo] Add new extractor Closes #2147. 2015-08-10 17:27:16 +00:00			`# coding: utf-8`
			`from __future__ import unicode_literals`

[indavideo] Add support for generic embeds (closes #11989) 2018-05-25 18:25:40 +00:00			`import re`

[indavideo] Add new extractor Closes #2147. 2015-08-10 17:27:16 +00:00			`from .common import InfoExtractor`
[indavideo] Sign download URLs 2018-05-25 17:46:05 +00:00			`from ..compat import compat_str`
[indavideo] Split in two extractors, extract all formats and fix timestamp 2015-08-13 17:25:47 +00:00			`from ..utils import (`
			`int_or_none,`
			`parse_age_limit,`
			`parse_iso8601,`
[indavideo] Sign download URLs 2018-05-25 17:46:05 +00:00			`update_url_query,`
[indavideo] Split in two extractors, extract all formats and fix timestamp 2015-08-13 17:25:47 +00:00			`)`
[indavideo] Add new extractor Closes #2147. 2015-08-10 17:27:16 +00:00

[indavideo] Split in two extractors, extract all formats and fix timestamp 2015-08-13 17:25:47 +00:00			`class IndavideoEmbedIE(InfoExtractor):`
			`_VALID_URL = r'https?://(?:(?:embed\.)?indavideo\.hu/player/video/\|assets\.indavideo\.hu/swf/player\.swf\?.*\b(?:v(?:ID\|id))=)(?P<id>[\da-f]+)'`
			`_TESTS = [{`
			`'url': 'http://indavideo.hu/player/video/1bdc3c6d80/',`
[indavideo] Fix extraction (closes #11221) 2018-05-25 18:09:44 +00:00			`'md5': 'c8a507a1c7410685f83a06eaeeaafeab',`
[indavideo] Split in two extractors, extract all formats and fix timestamp 2015-08-13 17:25:47 +00:00			`'info_dict': {`
			`'id': '1837039',`
			`'ext': 'mp4',`
			`'title': 'Cicatánc',`
			`'description': '',`
Fix "invalid escape sequences" error on Python 3.6 2017-01-02 12:08:07 +00:00			`'thumbnail': r're:^https?://.*\.jpg$',`
[indavideo] Split in two extractors, extract all formats and fix timestamp 2015-08-13 17:25:47 +00:00			`'uploader': 'cukiajanlo',`
			`'uploader_id': '83729',`
			`'timestamp': 1439193826,`
			`'upload_date': '20150810',`
			`'duration': 72,`
			`'age_limit': 0,`
			`'tags': ['tánc', 'cica', 'cuki', 'cukiajanlo', 'newsroom'],`
[indavideo] Add new extractor Closes #2147. 2015-08-10 17:27:16 +00:00			`},`
[indavideo] Split in two extractors, extract all formats and fix timestamp 2015-08-13 17:25:47 +00:00			`}, {`
			`'url': 'http://embed.indavideo.hu/player/video/1bdc3c6d80?autostart=1&hide=1',`
			`'only_matching': True,`
			`}, {`
			`'url': 'http://assets.indavideo.hu/swf/player.swf?v=fe25e500&vID=1bdc3c6d80&autostart=1&hide=1&i=1',`
			`'only_matching': True,`
			`}]`
[indavideo] Add new extractor Closes #2147. 2015-08-10 17:27:16 +00:00
[indavideo] Add support for generic embeds (closes #11989) 2018-05-25 18:25:40 +00:00			`# Some example URLs covered by generic extractor:`
			`# http://indavideo.hu/video/Vicces_cica_1`
			`# http://index.indavideo.hu/video/2015_0728_beregszasz`
			`# http://auto.indavideo.hu/video/Sajat_utanfutoban_a_kis_tacsko`
			`# http://erotika.indavideo.hu/video/Amator_tini_punci`
			`# http://film.indavideo.hu/video/f_hrom_nagymamm_volt`
			`# http://palyazat.indavideo.hu/video/Embertelen_dal_Dodgem_egyuttes`

			`@staticmethod`
			`def _extract_urls(webpage):`
			`return re.findall(`
			`r'<iframe[^>]+\bsrc=["\'](?P<url>(?:https?:)?//embed\.indavideo\.hu/player/video/[\da-f]+)',`
			`webpage)`

[indavideo] Add new extractor Closes #2147. 2015-08-10 17:27:16 +00:00			`def _real_extract(self, url):`
[indavideo] Split in two extractors, extract all formats and fix timestamp 2015-08-13 17:25:47 +00:00			`video_id = self._match_id(url)`

			`video = self._download_json(`
			`'http://amfphp.indavideo.hu/SYm0json.php/player.playerHandler.getVideoData/%s' % video_id,`
			`video_id)['data']`

			`title = video['title']`
[indavideo] Add new extractor Closes #2147. 2015-08-10 17:27:16 +00:00
[indavideo] Fix extraction (closes #11221) 2018-05-25 18:09:44 +00:00			`video_urls = []`

			`video_files = video.get('video_files')`
			`if isinstance(video_files, list):`
			`video_urls.extend(video_files)`
			`elif isinstance(video_files, dict):`
			`video_urls.extend(video_files.values())`

[indavideo] Split in two extractors, extract all formats and fix timestamp 2015-08-13 17:25:47 +00:00			`video_file = video.get('video_file')`
			`if video:`
			`video_urls.append(video_file)`
			`video_urls = list(set(video_urls))`
[indavideo] Add new extractor Closes #2147. 2015-08-10 17:27:16 +00:00
[indavideo] Split in two extractors, extract all formats and fix timestamp 2015-08-13 17:25:47 +00:00			`video_prefix = video_urls[0].rsplit('/', 1)[0]`
[indavideo] Add new extractor Closes #2147. 2015-08-10 17:27:16 +00:00
[indavideo] Split in two extractors, extract all formats and fix timestamp 2015-08-13 17:25:47 +00:00			`for flv_file in video.get('flv_files', []):`
			`flv_url = '%s/%s' % (video_prefix, flv_file)`
			`if flv_url not in video_urls:`
			`video_urls.append(flv_url)`
[indavideo] Add new extractor Closes #2147. 2015-08-10 17:27:16 +00:00
[indavideo] Sign download URLs 2018-05-25 17:46:05 +00:00			`filesh = video.get('filesh')`
[indavideo] Fix extraction (closes #11221) 2018-05-25 18:09:44 +00:00
			`formats = []`
			`for video_url in video_urls:`
			`height = int_or_none(self._search_regex(`
			`r'\.(\d{3,4})\.mp4(?:\?\|$)', video_url, 'height', default=None))`
			`if filesh:`
			`if not height:`
			`continue`
			`token = filesh.get(compat_str(height))`
			`if token is None:`
			`continue`
			`video_url = update_url_query(video_url, {'token': token})`
			`formats.append({`
			`'url': video_url,`
			`'height': height,`
			`})`
[indavideo] Split in two extractors, extract all formats and fix timestamp 2015-08-13 17:25:47 +00:00			`self._sort_formats(formats)`

			`timestamp = video.get('date')`
			`if timestamp:`
			`# upload date is in CEST`
			`timestamp = parse_iso8601(timestamp + ' +0200', ' ')`

			`thumbnails = [{`
			`'url': self._proto_relative_url(thumbnail)`
			`} for thumbnail in video.get('thumbnails', [])]`

[indavideo:embed] Fix tags extraction (Closes #8738) 2016-03-03 18:09:40 +00:00			`tags = [tag['title'] for tag in video.get('tags') or []]`
[indavideo] Add new extractor Closes #2147. 2015-08-10 17:27:16 +00:00
			`return {`
[indavideo] Relax _VALID_URL to match subdomains and add tests 2015-08-13 17:40:20 +00:00			`'id': video.get('id') or video_id,`
[indavideo] Split in two extractors, extract all formats and fix timestamp 2015-08-13 17:25:47 +00:00			`'title': title,`
			`'description': video.get('description'),`
[indavideo] Add new extractor Closes #2147. 2015-08-10 17:27:16 +00:00			`'thumbnails': thumbnails,`
[indavideo] Split in two extractors, extract all formats and fix timestamp 2015-08-13 17:25:47 +00:00			`'uploader': video.get('user_name'),`
			`'uploader_id': video.get('user_id'),`
			`'timestamp': timestamp,`
			`'duration': int_or_none(video.get('length')),`
			`'age_limit': parse_age_limit(video.get('age_limit')),`
[indavideo] Add new extractor Closes #2147. 2015-08-10 17:27:16 +00:00			`'tags': tags,`
[indavideo] Split in two extractors, extract all formats and fix timestamp 2015-08-13 17:25:47 +00:00			`'formats': formats,`
			`}`