[yt-dlp.git] / youtube_dl / extractor / youporn.py

from __future__ import unicode_literals

import re

from .common import InfoExtractor
from ..utils import (
    int_or_none,
    sanitized_Request,
    str_to_int,
    unescapeHTML,
    unified_strdate,
)
from ..aes import aes_decrypt_text


class YouPornIE(InfoExtractor):
    _VALID_URL = r'https?://(?:www\.)?youporn\.com/watch/(?P<id>\d+)/(?P<display_id>[^/?#&]+)'
    _TESTS = [{
        'url': 'http://www.youporn.com/watch/505835/sex-ed-is-it-safe-to-masturbate-daily/',
        'md5': '71ec5fcfddacf80f495efa8b6a8d9a89',
        'info_dict': {
            'id': '505835',
            'display_id': 'sex-ed-is-it-safe-to-masturbate-daily',
            'ext': 'mp4',
            'title': 'Sex Ed: Is It Safe To Masturbate Daily?',
            'description': 'Love & Sex Answers: http://bit.ly/DanAndJenn -- Is It Unhealthy To Masturbate Daily?',
            'thumbnail': 're:^https?://.*\.jpg$',
            'uploader': 'Ask Dan And Jennifer',
            'upload_date': '20101221',
            'average_rating': int,
            'view_count': int,
            'comment_count': int,
            'categories': list,
            'tags': list,
            'age_limit': 18,
        },
    }, {
        # Anonymous User uploader
        'url': 'http://www.youporn.com/watch/561726/big-tits-awesome-brunette-on-amazing-webcam-show/?from=related3&al=2&from_id=561726&pos=4',
        'info_dict': {
            'id': '561726',
            'display_id': 'big-tits-awesome-brunette-on-amazing-webcam-show',
            'ext': 'mp4',
            'title': 'Big Tits Awesome Brunette On amazing webcam show',
            'description': 'http://sweetlivegirls.com Big Tits Awesome Brunette On amazing webcam show.mp4',
            'thumbnail': 're:^https?://.*\.jpg$',
            'uploader': 'Anonymous User',
            'upload_date': '20111125',
            'average_rating': int,
            'view_count': int,
            'comment_count': int,
            'categories': list,
            'tags': list,
            'age_limit': 18,
        },
        'params': {
            'skip_download': True,
        },
    }]

    def _real_extract(self, url):
        mobj = re.match(self._VALID_URL, url)
        video_id = mobj.group('id')
        display_id = mobj.group('display_id')

        request = sanitized_Request(url)
        request.add_header('Cookie', 'age_verified=1')
        webpage = self._download_webpage(request, display_id)

        title = self._search_regex(
            [r'(?:video_titles|videoTitle)\s*[:=]\s*(["\'])(?P<title>.+?)\1',
             r'<h1[^>]+class=["\']heading\d?["\'][^>]*>([^<])<'],
            webpage, 'title', group='title')

        links = []

        sources = self._search_regex(
            r'sources\s*:\s*({.+?})', webpage, 'sources', default=None)
        if sources:
            for _, link in re.findall(r'[^:]+\s*:\s*(["\'])(http.+?)\1', sources):
                links.append(link)

        # Fallback #1
        for _, link in re.findall(
                r'(?:videoUrl|videoSrc|videoIpadUrl|html5PlayerSrc)\s*[:=]\s*(["\'])(http.+?)\1', webpage):
            links.append(link)

        # Fallback #2, this also contains extra low quality 180p format
        for _, link in re.findall(r'<a[^>]+href=(["\'])(http.+?)\1[^>]+title=["\']Download [Vv]ideo', webpage):
            links.append(link)

        # Fallback #3, encrypted links
        for _, encrypted_link in re.findall(
                r'encryptedQuality\d{3,4}URL\s*=\s*(["\'])([\da-zA-Z+/=]+)\1', webpage):
            links.append(aes_decrypt_text(encrypted_link, title, 32).decode('utf-8'))

        formats = []
        for video_url in set(unescapeHTML(link) for link in links):
            f = {
                'url': video_url,
            }
            # Video URL's path looks like this:
            #  /201012/17/505835/720p_1500k_505835/YouPorn%20-%20Sex%20Ed%20Is%20It%20Safe%20To%20Masturbate%20Daily.mp4
            # We will benefit from it by extracting some metadata
            mobj = re.search(r'/(?P<height>\d{3,4})[pP]_(?P<bitrate>\d+)[kK]_\d+/', video_url)
            if mobj:
                height = int(mobj.group('height'))
                bitrate = int(mobj.group('bitrate'))
                f.update({
                    'format_id': '%dp-%dk' % (height, bitrate),
                    'height': height,
                    'tbr': bitrate,
                })
            formats.append(f)
        self._sort_formats(formats)

        description = self._html_search_regex(
            r'(?s)<div[^>]+class=["\']video-description["\'][^>]*>(.+?)</div>',
            webpage, 'description', default=None)
        thumbnail = self._search_regex(
            r'(?:imageurl\s*=|poster\s*:)\s*(["\'])(?P<thumbnail>.+?)\1',
            webpage, 'thumbnail', fatal=False, group='thumbnail')

        uploader = self._html_search_regex(
            r'(?s)<div[^>]+class=["\']videoInfoBy["\'][^>]*>\s*By:\s*</div>(.+?)</(?:a|div)>',
            webpage, 'uploader', fatal=False)
        upload_date = unified_strdate(self._html_search_regex(
            r'(?s)<div[^>]+class=["\']videoInfoTime["\'][^>]*>(.+?)</div>',
            webpage, 'upload date', fatal=False))

        age_limit = self._rta_search(webpage)

        average_rating = int_or_none(self._search_regex(
            r'<div[^>]+class=["\']videoInfoRating["\'][^>]*>\s*<div[^>]+class=["\']videoRatingPercentage["\'][^>]*>(\d+)%</div>',
            webpage, 'average rating', fatal=False))

        view_count = str_to_int(self._search_regex(
            r'(?s)<div[^>]+class=["\']videoInfoViews["\'][^>]*>.*?([\d,.]+)\s*</div>',
            webpage, 'view count', fatal=False))
        comment_count = str_to_int(self._search_regex(
            r'>All [Cc]omments? \(([\d,.]+)\)',
            webpage, 'comment count', fatal=False))

        def extract_tag_box(title):
            tag_box = self._search_regex(
                (r'<div[^>]+class=["\']tagBoxTitle["\'][^>]*>\s*%s\b.*?</div>\s*'
                 '<div[^>]+class=["\']tagBoxContent["\']>(.+?)</div>') % re.escape(title),
                webpage, '%s tag box' % title, default=None)
            if not tag_box:
                return []
            return re.findall(r'<a[^>]+href=[^>]+>([^<]+)', tag_box)

        categories = extract_tag_box('Category')
        tags = extract_tag_box('Tags')

        return {
            'id': video_id,
            'display_id': display_id,
            'title': title,
            'description': description,
            'thumbnail': thumbnail,
            'uploader': uploader,
            'upload_date': upload_date,
            'average_rating': average_rating,
            'view_count': view_count,
            'comment_count': comment_count,
            'categories': categories,
            'tags': tags,
            'age_limit': age_limit,
            'formats': formats,
        }
Commit	Line	Data
f24e9833 PH	1	from __future__ import unicode_literals
f24e9833 PH	2
0143dc02	3	import re
0143dc02 PH	4
0143dc02 PH	5	from .common import InfoExtractor
1cc79574	6	from ..utils import (
589c33da	7	int_or_none,
5c2266df	8	sanitized_Request,
589c33da	9	str_to_int,
0143dc02 PH	10	unescapeHTML,
	11	unified_strdate,
	12	)
589c33da	13	from ..aes import aes_decrypt_text
0143dc02	14
bfe9de85	15
0143dc02	16	class YouPornIE(InfoExtractor):
589c33da	17	_VALID_URL = r'https?://(?:www\.)?youporn\.com/watch/(?P<id>\d+)/(?P<display_id>[^/?#&]+)'
4f13f8f7	18	_TESTS = [{
f24e9833	19	'url': 'http://www.youporn.com/watch/505835/sex-ed-is-it-safe-to-masturbate-daily/',
4f13f8f7	20	'md5': '71ec5fcfddacf80f495efa8b6a8d9a89',
f24e9833 PH	21	'info_dict': {
f24e9833 PH	22	'id': '505835',
589c33da	23	'display_id': 'sex-ed-is-it-safe-to-masturbate-daily',
f24e9833	24	'ext': 'mp4',
f24e9833	25	'title': 'Sex Ed: Is It Safe To Masturbate Daily?',
589c33da S	26	'description': 'Love & Sex Answers: http://bit.ly/DanAndJenn -- Is It Unhealthy To Masturbate Daily?',
589c33da S	27	'thumbnail': 're:^https?://.*\.jpg$',
e572a101	28	'uploader': 'Ask Dan And Jennifer',
589c33da S	29	'upload_date': '20101221',
	30	'average_rating': int,
	31	'view_count': int,
755ff8d2	32	'comment_count': int,
589c33da S	33	'categories': list,
589c33da S	34	'tags': list,
f24e9833	35	'age_limit': 18,
4f13f8f7 S	36	},
	37	}, {
	38	# Anonymous User uploader
	39	'url': 'http://www.youporn.com/watch/561726/big-tits-awesome-brunette-on-amazing-webcam-show/?from=related3&al=2&from_id=561726&pos=4',
	40	'info_dict': {
	41	'id': '561726',
	42	'display_id': 'big-tits-awesome-brunette-on-amazing-webcam-show',
	43	'ext': 'mp4',
	44	'title': 'Big Tits Awesome Brunette On amazing webcam show',
	45	'description': 'http://sweetlivegirls.com Big Tits Awesome Brunette On amazing webcam show.mp4',
	46	'thumbnail': 're:^https?://.*\.jpg$',
	47	'uploader': 'Anonymous User',
	48	'upload_date': '20111125',
	49	'average_rating': int,
	50	'view_count': int,
755ff8d2	51	'comment_count': int,
4f13f8f7 S	52	'categories': list,
	53	'tags': list,
	54	'age_limit': 18,
	55	},
	56	'params': {
	57	'skip_download': True,
	58	},
	59	}]
0143dc02	60
0143dc02 PH	61	def _real_extract(self, url):
0143dc02 PH	62	mobj = re.match(self._VALID_URL, url)
589c33da S	63	video_id = mobj.group('id')
589c33da S	64	display_id = mobj.group('display_id')
0143dc02	65
5c2266df	66	request = sanitized_Request(url)
589c33da S	67	request.add_header('Cookie', 'age_verified=1')
	68	webpage = self._download_webpage(request, display_id)
	69
	70	title = self._search_regex(
	71	[r'(?:video_titles\|videoTitle)\s[:=]\s(["\'])(?P<title>.+?)\1',
	72	r'<h1[^>]+class=["\']heading\d?["\'][^>]*>([^<])<'],
	73	webpage, 'title', group='title')
0143dc02	74
589c33da S	75	links = []
	76
	77	sources = self._search_regex(
	78	r'sources\s:\s({.+?})', webpage, 'sources', default=None)
	79	if sources:
	80	for _, link in re.findall(r'[^:]+\s:\s(["\'])(http.+?)\1', sources):
e572a101	81	links.append(link)
5f6a1245	82
589c33da S	83	# Fallback #1
	84	for _, link in re.findall(
	85	r'(?:videoUrl\|videoSrc\|videoIpadUrl\|html5PlayerSrc)\s[:=]\s(["\'])(http.+?)\1', webpage):
	86	links.append(link)
	87
	88	# Fallback #2, this also contains extra low quality 180p format
	89	for _, link in re.findall(r'<a[^>]+href=(["\'])(http.+?)\1[^>]+title=["\']Download [Vv]ideo', webpage):
	90	links.append(link)
	91
	92	# Fallback #3, encrypted links
	93	for _, encrypted_link in re.findall(
	94	r'encryptedQuality\d{3,4}URL\s=\s(["\'])([\da-zA-Z+/=]+)\1', webpage):
	95	links.append(aes_decrypt_text(encrypted_link, title, 32).decode('utf-8'))
	96
0143dc02	97	formats = []
589c33da S	98	for video_url in set(unescapeHTML(link) for link in links):
589c33da S	99	f = {
0143dc02	100	'url': video_url,
589c33da S	101	}
	102	# Video URL's path looks like this:
	103	# /201012/17/505835/720p_1500k_505835/YouPorn%20-%20Sex%20Ed%20Is%20It%20Safe%20To%20Masturbate%20Daily.mp4
	104	# We will benefit from it by extracting some metadata
	105	mobj = re.search(r'/(?P<height>\d{3,4})[pP]_(?P<bitrate>\d+)[kK]_\d+/', video_url)
	106	if mobj:
	107	height = int(mobj.group('height'))
	108	bitrate = int(mobj.group('bitrate'))
	109	f.update({
	110	'format_id': '%dp-%dk' % (height, bitrate),
	111	'height': height,
	112	'tbr': bitrate,
	113	})
	114	formats.append(f)
bfe9de85 PH	115	self._sort_formats(formats)
bfe9de85 PH	116
589c33da S	117	description = self._html_search_regex(
589c33da S	118	r'(?s)<div[^>]+class=["\']video-description["\'][^>]*>(.+?)</div>',
feb7711c	119	webpage, 'description', default=None)
589c33da S	120	thumbnail = self._search_regex(
	121	r'(?:imageurl\s=\|poster\s:)\s*(["\'])(?P<thumbnail>.+?)\1',
	122	webpage, 'thumbnail', fatal=False, group='thumbnail')
	123
4f13f8f7 S	124	uploader = self._html_search_regex(
4f13f8f7 S	125	r'(?s)<div[^>]+class=["\']videoInfoBy["\'][^>]>\sBy:\s*</div>(.+?)</(?:a\|div)>',
589c33da S	126	webpage, 'uploader', fatal=False)
	127	upload_date = unified_strdate(self._html_search_regex(
	128	r'(?s)<div[^>]+class=["\']videoInfoTime["\'][^>]*>(.+?)</div>',
	129	webpage, 'upload date', fatal=False))
	130
	131	age_limit = self._rta_search(webpage)
	132
	133	average_rating = int_or_none(self._search_regex(
	134	r'<div[^>]+class=["\']videoInfoRating["\'][^>]>\s<div[^>]+class=["\']videoRatingPercentage["\'][^>]*>(\d+)%</div>',
	135	webpage, 'average rating', fatal=False))
	136
	137	view_count = str_to_int(self._search_regex(
	138	r'(?s)<div[^>]+class=["\']videoInfoViews["\'][^>]>.?([\d,.]+)\s*</div>',
	139	webpage, 'view count', fatal=False))
755ff8d2 S	140	comment_count = str_to_int(self._search_regex(
	141	r'>All [Cc]omments? \(([\d,.]+)\)',
	142	webpage, 'comment count', fatal=False))
589c33da S	143
	144	def extract_tag_box(title):
	145	tag_box = self._search_regex(
	146	(r'<div[^>]+class=["\']tagBoxTitle["\'][^>]>\s%s\b.?</div>\s'
	147	'<div[^>]+class=["\']tagBoxContent["\']>(.+?)</div>') % re.escape(title),
	148	webpage, '%s tag box' % title, default=None)
	149	if not tag_box:
	150	return []
	151	return re.findall(r'<a[^>]+href=[^>]+>([^<]+)', tag_box)
	152
	153	categories = extract_tag_box('Category')
	154	tags = extract_tag_box('Tags')
5f6a1245	155
7df28654	156	return {
7df28654	157	'id': video_id,
589c33da S	158	'display_id': display_id,
	159	'title': title,
	160	'description': description,
	161	'thumbnail': thumbnail,
	162	'uploader': uploader,
	163	'upload_date': upload_date,
	164	'average_rating': average_rating,
	165	'view_count': view_count,
755ff8d2	166	'comment_count': comment_count,
589c33da S	167	'categories': categories,
589c33da S	168	'tags': tags,
7df28654	169	'age_limit': age_limit,
	170	'formats': formats,
	171	}