[yt-dlp.git] / youtube_dlc / extractor / picarto.py

# coding: utf-8
from __future__ import unicode_literals

import re
import time

from .common import InfoExtractor
from ..compat import compat_str
from ..utils import (
    ExtractorError,
    js_to_json,
    try_get,
    update_url_query,
    urlencode_postdata,
)


class PicartoIE(InfoExtractor):
    _VALID_URL = r'https?://(?:www.)?picarto\.tv/(?P<id>[a-zA-Z0-9]+)(?:/(?P<token>[a-zA-Z0-9]+))?'
    _TEST = {
        'url': 'https://picarto.tv/Setz',
        'info_dict': {
            'id': 'Setz',
            'ext': 'mp4',
            'title': 're:^Setz [0-9]{4}-[0-9]{2}-[0-9]{2} [0-9]{2}:[0-9]{2}$',
            'timestamp': int,
            'is_live': True
        },
        'skip': 'Stream is offline',
    }

    @classmethod
    def suitable(cls, url):
        return False if PicartoVodIE.suitable(url) else super(PicartoIE, cls).suitable(url)

    def _real_extract(self, url):
        mobj = re.match(self._VALID_URL, url)
        channel_id = mobj.group('id')

        metadata = self._download_json(
            'https://api.picarto.tv/v1/channel/name/' + channel_id,
            channel_id)

        if metadata.get('online') is False:
            raise ExtractorError('Stream is offline', expected=True)

        cdn_data = self._download_json(
            'https://picarto.tv/process/channel', channel_id,
            data=urlencode_postdata({'loadbalancinginfo': channel_id}),
            note='Downloading load balancing info')

        token = mobj.group('token') or 'public'
        params = {
            'con': int(time.time() * 1000),
            'token': token,
        }

        prefered_edge = cdn_data.get('preferedEdge')
        formats = []

        for edge in cdn_data['edges']:
            edge_ep = edge.get('ep')
            if not edge_ep or not isinstance(edge_ep, compat_str):
                continue
            edge_id = edge.get('id')
            for tech in cdn_data['techs']:
                tech_label = tech.get('label')
                tech_type = tech.get('type')
                preference = 0
                if edge_id == prefered_edge:
                    preference += 1
                format_id = []
                if edge_id:
                    format_id.append(edge_id)
                if tech_type == 'application/x-mpegurl' or tech_label == 'HLS':
                    format_id.append('hls')
                    formats.extend(self._extract_m3u8_formats(
                        update_url_query(
                            'https://%s/hls/%s/index.m3u8'
                            % (edge_ep, channel_id), params),
                        channel_id, 'mp4', preference=preference,
                        m3u8_id='-'.join(format_id), fatal=False))
                    continue
                elif tech_type == 'video/mp4' or tech_label == 'MP4':
                    format_id.append('mp4')
                    formats.append({
                        'url': update_url_query(
                            'https://%s/mp4/%s.mp4' % (edge_ep, channel_id),
                            params),
                        'format_id': '-'.join(format_id),
                        'preference': preference,
                    })
                else:
                    # rtmp format does not seem to work
                    continue
        self._sort_formats(formats)

        mature = metadata.get('adult')
        if mature is None:
            age_limit = None
        else:
            age_limit = 18 if mature is True else 0

        return {
            'id': channel_id,
            'title': self._live_title(metadata.get('title') or channel_id),
            'is_live': True,
            'thumbnail': try_get(metadata, lambda x: x['thumbnails']['web']),
            'channel': channel_id,
            'channel_url': 'https://picarto.tv/%s' % channel_id,
            'age_limit': age_limit,
            'formats': formats,
        }


class PicartoVodIE(InfoExtractor):
    _VALID_URL = r'https?://(?:www.)?picarto\.tv/videopopout/(?P<id>[^/?#&]+)'
    _TESTS = [{
        'url': 'https://picarto.tv/videopopout/ArtofZod_2017.12.12.00.13.23.flv',
        'md5': '3ab45ba4352c52ee841a28fb73f2d9ca',
        'info_dict': {
            'id': 'ArtofZod_2017.12.12.00.13.23.flv',
            'ext': 'mp4',
            'title': 'ArtofZod_2017.12.12.00.13.23.flv',
            'thumbnail': r're:^https?://.*\.jpg'
        },
    }, {
        'url': 'https://picarto.tv/videopopout/Plague',
        'only_matching': True,
    }]

    def _real_extract(self, url):
        video_id = self._match_id(url)

        webpage = self._download_webpage(url, video_id)

        vod_info = self._parse_json(
            self._search_regex(
                r'(?s)#vod-player["\']\s*,\s*(\{.+?\})\s*\)', webpage,
                video_id),
            video_id, transform_source=js_to_json)

        formats = self._extract_m3u8_formats(
            vod_info['vod'], video_id, 'mp4', entry_protocol='m3u8_native',
            m3u8_id='hls')
        self._sort_formats(formats)

        return {
            'id': video_id,
            'title': video_id,
            'thumbnail': vod_info.get('vodThumb'),
            'formats': formats,
        }
Commit	Line	Data
d6166a76 PG	1	# coding: utf-8
	2	from __future__ import unicode_literals
	3
730c0d12	4	import re
a42839e5 S	5	import time
a42839e5 S	6
d6166a76	7	from .common import InfoExtractor
a42839e5 S	8	from ..compat import compat_str
	9	from ..utils import (
	10	ExtractorError,
	11	js_to_json,
730c0d12	12	try_get,
a42839e5 S	13	update_url_query,
	14	urlencode_postdata,
	15	)
d6166a76 PG	16
	17
	18	class PicartoIE(InfoExtractor):
f17a24a6	19	_VALID_URL = r'https?://(?:www.)?picarto\.tv/(?P<id>[a-zA-Z0-9]+)(?:/(?P<token>[a-zA-Z0-9]+))?'
d6166a76 PG	20	_TEST = {
	21	'url': 'https://picarto.tv/Setz',
	22	'info_dict': {
	23	'id': 'Setz',
	24	'ext': 'mp4',
	25	'title': 're:^Setz [0-9]{4}-[0-9]{2}-[0-9]{2} [0-9]{2}:[0-9]{2}$',
	26	'timestamp': int,
	27	'is_live': True
	28	},
a42839e5	29	'skip': 'Stream is offline',
d6166a76 PG	30	}
d6166a76 PG	31
a42839e5 S	32	@classmethod
	33	def suitable(cls, url):
	34	return False if PicartoVodIE.suitable(url) else super(PicartoIE, cls).suitable(url)
	35
d6166a76	36	def _real_extract(self, url):
730c0d12 S	37	mobj = re.match(self._VALID_URL, url)
	38	channel_id = mobj.group('id')
	39
f17a24a6 PG	40	metadata = self._download_json(
	41	'https://api.picarto.tv/v1/channel/name/' + channel_id,
	42	channel_id)
d6166a76	43
f17a24a6	44	if metadata.get('online') is False:
d6166a76 PG	45	raise ExtractorError('Stream is offline', expected=True)
d6166a76 PG	46
a42839e5 S	47	cdn_data = self._download_json(
a42839e5 S	48	'https://picarto.tv/process/channel', channel_id,
d6166a76	49	data=urlencode_postdata({'loadbalancinginfo': channel_id}),
a42839e5 S	50	note='Downloading load balancing info')
a42839e5 S	51
730c0d12	52	token = mobj.group('token') or 'public'
a42839e5	53	params = {
a42839e5	54	'con': int(time.time() * 1000),
f17a24a6	55	'token': token,
a42839e5 S	56	}
	57
	58	prefered_edge = cdn_data.get('preferedEdge')
a42839e5 S	59	formats = []
	60
	61	for edge in cdn_data['edges']:
	62	edge_ep = edge.get('ep')
	63	if not edge_ep or not isinstance(edge_ep, compat_str):
	64	continue
	65	edge_id = edge.get('id')
	66	for tech in cdn_data['techs']:
	67	tech_label = tech.get('label')
	68	tech_type = tech.get('type')
	69	preference = 0
	70	if edge_id == prefered_edge:
	71	preference += 1
a42839e5 S	72	format_id = []
	73	if edge_id:
	74	format_id.append(edge_id)
	75	if tech_type == 'application/x-mpegurl' or tech_label == 'HLS':
	76	format_id.append('hls')
	77	formats.extend(self._extract_m3u8_formats(
	78	update_url_query(
	79	'https://%s/hls/%s/index.m3u8'
	80	% (edge_ep, channel_id), params),
	81	channel_id, 'mp4', preference=preference,
	82	m3u8_id='-'.join(format_id), fatal=False))
	83	continue
	84	elif tech_type == 'video/mp4' or tech_label == 'MP4':
	85	format_id.append('mp4')
	86	formats.append({
	87	'url': update_url_query(
	88	'https://%s/mp4/%s.mp4' % (edge_ep, channel_id),
	89	params),
	90	'format_id': '-'.join(format_id),
	91	'preference': preference,
	92	})
	93	else:
	94	# rtmp format does not seem to work
	95	continue
d6166a76 PG	96	self._sort_formats(formats)
d6166a76 PG	97
f17a24a6	98	mature = metadata.get('adult')
a42839e5 S	99	if mature is None:
	100	age_limit = None
	101	else:
	102	age_limit = 18 if mature is True else 0
	103
d6166a76 PG	104	return {
d6166a76 PG	105	'id': channel_id,
730c0d12	106	'title': self._live_title(metadata.get('title') or channel_id),
d6166a76	107	'is_live': True,
730c0d12 S	108	'thumbnail': try_get(metadata, lambda x: x['thumbnails']['web']),
	109	'channel': channel_id,
	110	'channel_url': 'https://picarto.tv/%s' % channel_id,
a42839e5 S	111	'age_limit': age_limit,
a42839e5 S	112	'formats': formats,
d6166a76 PG	113	}
	114
	115
	116	class PicartoVodIE(InfoExtractor):
a42839e5 S	117	_VALID_URL = r'https?://(?:www.)?picarto\.tv/videopopout/(?P<id>[^/?#&]+)'
	118	_TESTS = [{
	119	'url': 'https://picarto.tv/videopopout/ArtofZod_2017.12.12.00.13.23.flv',
	120	'md5': '3ab45ba4352c52ee841a28fb73f2d9ca',
d6166a76	121	'info_dict': {
a42839e5	122	'id': 'ArtofZod_2017.12.12.00.13.23.flv',
d6166a76	123	'ext': 'mp4',
a42839e5 S	124	'title': 'ArtofZod_2017.12.12.00.13.23.flv',
	125	'thumbnail': r're:^https?://.*\.jpg'
	126	},
	127	}, {
	128	'url': 'https://picarto.tv/videopopout/Plague',
	129	'only_matching': True,
	130	}]
d6166a76 PG	131
	132	def _real_extract(self, url):
	133	video_id = self._match_id(url)
a42839e5	134
d6166a76 PG	135	webpage = self._download_webpage(url, video_id)
d6166a76 PG	136
a42839e5 S	137	vod_info = self._parse_json(
	138	self._search_regex(
	139	r'(?s)#vod-player["\']\s,\s(\{.+?\})\s*\)', webpage,
	140	video_id),
	141	video_id, transform_source=js_to_json)
	142
	143	formats = self._extract_m3u8_formats(
	144	vod_info['vod'], video_id, 'mp4', entry_protocol='m3u8_native',
	145	m3u8_id='hls')
	146	self._sort_formats(formats)
d6166a76 PG	147
	148	return {
	149	'id': video_id,
	150	'title': video_id,
d6166a76	151	'thumbnail': vod_info.get('vodThumb'),
a42839e5	152	'formats': formats,
d6166a76	153	}