[yt-dlp.git] / yt_dlp / extractor / apa.py

# coding: utf-8
from __future__ import unicode_literals

import re

from .common import InfoExtractor
from ..utils import (
    determine_ext,
    int_or_none,
    url_or_none,
)


class APAIE(InfoExtractor):
    _VALID_URL = r'(?P<base_url>https?://[^/]+\.apa\.at)/embed/(?P<id>[\da-f]{8}-[\da-f]{4}-[\da-f]{4}-[\da-f]{4}-[\da-f]{12})'
    _TESTS = [{
        'url': 'http://uvp.apa.at/embed/293f6d17-692a-44e3-9fd5-7b178f3a1029',
        'md5': '2b12292faeb0a7d930c778c7a5b4759b',
        'info_dict': {
            'id': '293f6d17-692a-44e3-9fd5-7b178f3a1029',
            'ext': 'mp4',
            'title': '293f6d17-692a-44e3-9fd5-7b178f3a1029',
            'thumbnail': r're:^https?://.*\.jpg$',
        },
    }, {
        'url': 'https://uvp-apapublisher.sf.apa.at/embed/2f94e9e6-d945-4db2-9548-f9a41ebf7b78',
        'only_matching': True,
    }, {
        'url': 'http://uvp-rma.sf.apa.at/embed/70404cca-2f47-4855-bbb8-20b1fae58f76',
        'only_matching': True,
    }, {
        'url': 'http://uvp-kleinezeitung.sf.apa.at/embed/f1c44979-dba2-4ebf-b021-e4cf2cac3c81',
        'only_matching': True,
    }]

    @staticmethod
    def _extract_urls(webpage):
        return [
            mobj.group('url')
            for mobj in re.finditer(
                r'<iframe[^>]+\bsrc=(["\'])(?P<url>(?:https?:)?//[^/]+\.apa\.at/embed/[\da-f]{8}-[\da-f]{4}-[\da-f]{4}-[\da-f]{4}-[\da-f]{12}.*?)\1',
                webpage)]

    def _real_extract(self, url):
        mobj = self._match_valid_url(url)
        video_id, base_url = mobj.group('id', 'base_url')

        webpage = self._download_webpage(
            '%s/player/%s' % (base_url, video_id), video_id)

        jwplatform_id = self._search_regex(
            r'media[iI]d\s*:\s*["\'](?P<id>[a-zA-Z0-9]{8})', webpage,
            'jwplatform id', default=None)

        if jwplatform_id:
            return self.url_result(
                'jwplatform:' + jwplatform_id, ie='JWPlatform',
                video_id=video_id)

        def extract(field, name=None):
            return self._search_regex(
                r'\b%s["\']\s*:\s*(["\'])(?P<value>(?:(?!\1).)+)\1' % field,
                webpage, name or field, default=None, group='value')

        title = extract('title') or video_id
        description = extract('description')
        thumbnail = extract('poster', 'thumbnail')

        formats = []
        for format_id in ('hls', 'progressive'):
            source_url = url_or_none(extract(format_id))
            if not source_url:
                continue
            ext = determine_ext(source_url)
            if ext == 'm3u8':
                formats.extend(self._extract_m3u8_formats(
                    source_url, video_id, 'mp4', entry_protocol='m3u8_native',
                    m3u8_id='hls', fatal=False))
            else:
                height = int_or_none(self._search_regex(
                    r'(\d+)\.mp4', source_url, 'height', default=None))
                formats.append({
                    'url': source_url,
                    'format_id': format_id,
                    'height': height,
                })
        self._sort_formats(formats)

        return {
            'id': video_id,
            'title': title,
            'description': description,
            'thumbnail': thumbnail,
            'formats': formats,
        }
Commit	Line	Data
cfd7f2a6 S	1	# coding: utf-8
	2	from __future__ import unicode_literals
	3
	4	import re
	5
	6	from .common import InfoExtractor
cfd7f2a6 S	7	from ..utils import (
cfd7f2a6 S	8	determine_ext,
7c60c33e	9	int_or_none,
3052a30d	10	url_or_none,
cfd7f2a6 S	11	)
	12
	13
	14	class APAIE(InfoExtractor):
7c60c33e	15	_VALID_URL = r'(?P<base_url>https?://[^/]+\.apa\.at)/embed/(?P<id>[\da-f]{8}-[\da-f]{4}-[\da-f]{4}-[\da-f]{4}-[\da-f]{12})'
cfd7f2a6 S	16	_TESTS = [{
	17	'url': 'http://uvp.apa.at/embed/293f6d17-692a-44e3-9fd5-7b178f3a1029',
	18	'md5': '2b12292faeb0a7d930c778c7a5b4759b',
	19	'info_dict': {
7c60c33e	20	'id': '293f6d17-692a-44e3-9fd5-7b178f3a1029',
cfd7f2a6	21	'ext': 'mp4',
7c60c33e	22	'title': '293f6d17-692a-44e3-9fd5-7b178f3a1029',
cfd7f2a6	23	'thumbnail': r're:^https?://.*\.jpg$',
cfd7f2a6 S	24	},
	25	}, {
	26	'url': 'https://uvp-apapublisher.sf.apa.at/embed/2f94e9e6-d945-4db2-9548-f9a41ebf7b78',
	27	'only_matching': True,
	28	}, {
	29	'url': 'http://uvp-rma.sf.apa.at/embed/70404cca-2f47-4855-bbb8-20b1fae58f76',
	30	'only_matching': True,
	31	}, {
	32	'url': 'http://uvp-kleinezeitung.sf.apa.at/embed/f1c44979-dba2-4ebf-b021-e4cf2cac3c81',
	33	'only_matching': True,
	34	}]
	35
	36	@staticmethod
	37	def _extract_urls(webpage):
	38	return [
	39	mobj.group('url')
	40	for mobj in re.finditer(
	41	r'<iframe[^>]+\bsrc=(["\'])(?P<url>(?:https?:)?//[^/]+\.apa\.at/embed/[\da-f]{8}-[\da-f]{4}-[\da-f]{4}-[\da-f]{4}-[\da-f]{12}.*?)\1',
	42	webpage)]
	43
	44	def _real_extract(self, url):
5ad28e7f	45	mobj = self._match_valid_url(url)
7c60c33e	46	video_id, base_url = mobj.group('id', 'base_url')
cfd7f2a6	47
7c60c33e	48	webpage = self._download_webpage(
7c60c33e	49	'%s/player/%s' % (base_url, video_id), video_id)
cfd7f2a6 S	50
	51	jwplatform_id = self._search_regex(
	52	r'media[iI]d\s:\s["\'](?P<id>[a-zA-Z0-9]{8})', webpage,
	53	'jwplatform id', default=None)
	54
	55	if jwplatform_id:
	56	return self.url_result(
	57	'jwplatform:' + jwplatform_id, ie='JWPlatform',
	58	video_id=video_id)
	59
7c60c33e	60	def extract(field, name=None):
	61	return self._search_regex(
	62	r'\b%s["\']\s:\s(["\'])(?P<value>(?:(?!\1).)+)\1' % field,
	63	webpage, name or field, default=None, group='value')
	64
	65	title = extract('title') or video_id
	66	description = extract('description')
	67	thumbnail = extract('poster', 'thumbnail')
cfd7f2a6 S	68
cfd7f2a6 S	69	formats = []
7c60c33e	70	for format_id in ('hls', 'progressive'):
7c60c33e	71	source_url = url_or_none(extract(format_id))
3052a30d	72	if not source_url:
cfd7f2a6 S	73	continue
	74	ext = determine_ext(source_url)
	75	if ext == 'm3u8':
	76	formats.extend(self._extract_m3u8_formats(
	77	source_url, video_id, 'mp4', entry_protocol='m3u8_native',
	78	m3u8_id='hls', fatal=False))
	79	else:
7c60c33e	80	height = int_or_none(self._search_regex(
7c60c33e	81	r'(\d+)\.mp4', source_url, 'height', default=None))
cfd7f2a6 S	82	formats.append({
cfd7f2a6 S	83	'url': source_url,
7c60c33e	84	'format_id': format_id,
7c60c33e	85	'height': height,
cfd7f2a6 S	86	})
	87	self._sort_formats(formats)
	88
cfd7f2a6 S	89	return {
cfd7f2a6 S	90	'id': video_id,
7c60c33e	91	'title': title,
7c60c33e	92	'description': description,
cfd7f2a6 S	93	'thumbnail': thumbnail,
	94	'formats': formats,
	95	}