[yt-dlp.git] / youtube_dl / extractor / minhateca.py

# coding: utf-8
from __future__ import unicode_literals

from .common import InfoExtractor
from ..utils import (
    int_or_none,
    parse_duration,
    parse_filesize,
    sanitized_Request,
    urlencode_postdata,
)


class MinhatecaIE(InfoExtractor):
    _VALID_URL = r'https?://minhateca\.com\.br/[^?#]+,(?P<id>[0-9]+)\.'
    _TEST = {
        'url': 'http://minhateca.com.br/pereba/misc/youtube-dl+test+video,125848331.mp4(video)',
        'info_dict': {
            'id': '125848331',
            'ext': 'mp4',
            'title': 'youtube-dl test video',
            'thumbnail': 're:^https?://.*\.jpg$',
            'filesize_approx': 1530000,
            'duration': 9,
            'view_count': int,
        }
    }

    def _real_extract(self, url):
        video_id = self._match_id(url)
        webpage = self._download_webpage(url, video_id)

        token = self._html_search_regex(
            r'<input name="__RequestVerificationToken".*?value="([^"]+)"',
            webpage, 'request token')
        token_data = [
            ('fileId', video_id),
            ('__RequestVerificationToken', token),
        ]
        req = sanitized_Request(
            'http://minhateca.com.br/action/License/Download',
            data=urlencode_postdata(token_data))
        req.add_header('Content-Type', 'application/x-www-form-urlencoded')
        data = self._download_json(
            req, video_id, note='Downloading metadata')

        video_url = data['redirectUrl']
        title_str = self._html_search_regex(
            r'<h1.*?>(.*?)</h1>', webpage, 'title')
        title, _, ext = title_str.rpartition('.')
        filesize_approx = parse_filesize(self._html_search_regex(
            r'<p class="fileSize">(.*?)</p>',
            webpage, 'file size approximation', fatal=False))
        duration = parse_duration(self._html_search_regex(
            r'(?s)<p class="fileLeng[ht][th]">.*?class="bold">(.*?)<',
            webpage, 'duration', fatal=False))
        view_count = int_or_none(self._html_search_regex(
            r'<p class="downloadsCounter">([0-9]+)</p>',
            webpage, 'view count', fatal=False))

        return {
            'id': video_id,
            'url': video_url,
            'title': title,
            'ext': ext,
            'filesize_approx': filesize_approx,
            'duration': duration,
            'view_count': view_count,
            'thumbnail': self._og_search_thumbnail(webpage),
        }
Commit	Line	Data
4349c07d PH	1	# coding: utf-8
	2	from __future__ import unicode_literals
	3
	4	from .common import InfoExtractor
4349c07d PH	5	from ..utils import (
4349c07d PH	6	int_or_none,
e8df5cee	7	parse_duration,
4349c07d	8	parse_filesize,
5c2266df	9	sanitized_Request,
6e6bc8da	10	urlencode_postdata,
4349c07d PH	11	)
	12
	13
	14	class MinhatecaIE(InfoExtractor):
	15	_VALID_URL = r'https?://minhateca\.com\.br/[^?#]+,(?P<id>[0-9]+)\.'
	16	_TEST = {
	17	'url': 'http://minhateca.com.br/pereba/misc/youtube-dl+test+video,125848331.mp4(video)',
	18	'info_dict': {
	19	'id': '125848331',
	20	'ext': 'mp4',
	21	'title': 'youtube-dl test video',
	22	'thumbnail': 're:^https?://.*\.jpg$',
	23	'filesize_approx': 1530000,
	24	'duration': 9,
	25	'view_count': int,
	26	}
	27	}
	28
	29	def _real_extract(self, url):
	30	video_id = self._match_id(url)
	31	webpage = self._download_webpage(url, video_id)
	32
	33	token = self._html_search_regex(
	34	r'<input name="__RequestVerificationToken".*?value="([^"]+)"',
	35	webpage, 'request token')
	36	token_data = [
	37	('fileId', video_id),
	38	('__RequestVerificationToken', token),
	39	]
5c2266df	40	req = sanitized_Request(
4349c07d	41	'http://minhateca.com.br/action/License/Download',
6e6bc8da	42	data=urlencode_postdata(token_data))
4349c07d PH	43	req.add_header('Content-Type', 'application/x-www-form-urlencoded')
	44	data = self._download_json(
	45	req, video_id, note='Downloading metadata')
	46
	47	video_url = data['redirectUrl']
	48	title_str = self._html_search_regex(
	49	r'<h1.?>(.?)</h1>', webpage, 'title')
	50	title, _, ext = title_str.rpartition('.')
	51	filesize_approx = parse_filesize(self._html_search_regex(
	52	r'<p class="fileSize">(.*?)</p>',
	53	webpage, 'file size approximation', fatal=False))
e8df5cee PH	54	duration = parse_duration(self._html_search_regex(
e8df5cee PH	55	r'(?s)<p class="fileLeng[ht][th]">.?class="bold">(.?)<',
4349c07d PH	56	webpage, 'duration', fatal=False))
	57	view_count = int_or_none(self._html_search_regex(
	58	r'<p class="downloadsCounter">([0-9]+)</p>',
	59	webpage, 'view count', fatal=False))
	60
	61	return {
	62	'id': video_id,
	63	'url': video_url,
	64	'title': title,
	65	'ext': ext,
	66	'filesize_approx': filesize_approx,
	67	'duration': duration,
	68	'view_count': view_count,
	69	'thumbnail': self._og_search_thumbnail(webpage),
	70	}