[yt-dlp.git] / youtube_dl / extractor / xfileshare.py

# coding: utf-8
from __future__ import unicode_literals

import re

from .common import InfoExtractor
from ..compat import compat_urllib_parse
from ..utils import (
    ExtractorError,
    encode_dict,
    int_or_none,
    sanitized_Request,
)


class XFileShareIE(InfoExtractor):
    IE_DESC = 'XFileShare based sites: GorillaVid.in, daclips.in, movpod.in, fastvideo.in, realvid.net, filehoot.com and vidto.me'
    _VALID_URL = r'''(?x)
        https?://(?P<host>(?:www\.)?
            (?:daclips\.in|gorillavid\.in|movpod\.in|fastvideo\.in|realvid\.net|filehoot\.com|vidto\.me))/
        (?:embed-)?(?P<id>[0-9a-zA-Z]+)(?:-[0-9]+x[0-9]+\.html)?
    '''

    _FILE_NOT_FOUND_REGEX = r'>(?:404 - )?File Not Found<'

    _TESTS = [{
        'url': 'http://gorillavid.in/06y9juieqpmi',
        'md5': '5ae4a3580620380619678ee4875893ba',
        'info_dict': {
            'id': '06y9juieqpmi',
            'ext': 'flv',
            'title': 'Rebecca Black My Moment Official Music Video Reaction-6GK87Rc8bzQ',
            'thumbnail': 're:http://.*\.jpg',
        },
    }, {
        'url': 'http://gorillavid.in/embed-z08zf8le23c6-960x480.html',
        'only_matching': True,
    }, {
        'url': 'http://daclips.in/3rso4kdn6f9m',
        'md5': '1ad8fd39bb976eeb66004d3a4895f106',
        'info_dict': {
            'id': '3rso4kdn6f9m',
            'ext': 'mp4',
            'title': 'Micro Pig piglets ready on 16th July 2009-bG0PdrCdxUc',
            'thumbnail': 're:http://.*\.jpg',
        }
    }, {
        # video with countdown timeout
        'url': 'http://fastvideo.in/1qmdn1lmsmbw',
        'md5': '8b87ec3f6564a3108a0e8e66594842ba',
        'info_dict': {
            'id': '1qmdn1lmsmbw',
            'ext': 'mp4',
            'title': 'Man of Steel - Trailer',
            'thumbnail': 're:http://.*\.jpg',
        },
    }, {
        'url': 'http://realvid.net/ctn2y6p2eviw',
        'md5': 'b2166d2cf192efd6b6d764c18fd3710e',
        'info_dict': {
            'id': 'ctn2y6p2eviw',
            'ext': 'flv',
            'title': 'rdx 1955',
            'thumbnail': 're:http://.*\.jpg',
        },
    }, {
        'url': 'http://movpod.in/0wguyyxi1yca',
        'only_matching': True,
    }, {
        'url': 'http://filehoot.com/3ivfabn7573c.html',
        'info_dict': {
            'id': '3ivfabn7573c',
            'ext': 'mp4',
            'title': 'youtube-dl test video \'äBaW_jenozKc.mp4.mp4',
            'thumbnail': 're:http://.*\.jpg',
        }
    }, {
        'url': 'http://vidto.me/ku5glz52nqe1.html',
        'info_dict': {
            'id': 'ku5glz52nqe1',
            'ext': 'mp4',
            'title': 'test'
        }
    }]

    def _real_extract(self, url):
        mobj = re.match(self._VALID_URL, url)
        video_id = mobj.group('id')

        url = 'http://%s/%s' % (mobj.group('host'), video_id)
        webpage = self._download_webpage(url, video_id)

        if re.search(self._FILE_NOT_FOUND_REGEX, webpage) is not None:
            raise ExtractorError('Video %s does not exist' % video_id, expected=True)

        fields = self._hidden_inputs(webpage)

        if fields['op'] == 'download1':
            countdown = int_or_none(self._search_regex(
                r'<span id="countdown_str">(?:[Ww]ait)?\s*<span id="cxc">(\d+)</span>\s*(?:seconds?)?</span>',
                webpage, 'countdown', default=None))
            if countdown:
                self._sleep(countdown, video_id)

            post = compat_urllib_parse.urlencode(encode_dict(fields))

            req = sanitized_Request(url, post)
            req.add_header('Content-type', 'application/x-www-form-urlencoded')

            webpage = self._download_webpage(req, video_id, 'Downloading video page')

        title = (self._search_regex(
            [r'style="z-index: [0-9]+;">([^<]+)</span>',
             r'<td nowrap>([^<]+)</td>',
             r'>Watch (.+) ',
             r'<h2 class="video-page-head">([^<]+)</h2>'],
            webpage, 'title', default=None) or self._og_search_title(webpage)).strip()
        video_url = self._search_regex(
            [r'file\s*:\s*["\'](http[^"\']+)["\'],',
             r'file_link\s*=\s*\'(https?:\/\/[0-9a-zA-z.\/\-_]+)'],
            webpage, 'file url')
        thumbnail = self._search_regex(
            r'image\s*:\s*["\'](http[^"\']+)["\'],', webpage, 'thumbnail', default=None)

        formats = [{
            'format_id': 'sd',
            'url': video_url,
            'quality': 1,
        }]

        return {
            'id': video_id,
            'title': title,
            'thumbnail': thumbnail,
            'formats': formats,
        }
Commit	Line	Data
031ec536	1	# coding: utf-8
617c0b22	2	from __future__ import unicode_literals
	3
	4	import re
	5
	6	from .common import InfoExtractor
5c2266df	7	from ..compat import compat_urllib_parse
1cc79574 PH	8	from ..utils import (
1cc79574 PH	9	ExtractorError,
4abe2144	10	encode_dict,
ceb33673	11	int_or_none,
5c2266df	12	sanitized_Request,
5f28a1ac PP	13	)
5f28a1ac PP	14
617c0b22	15
031ec536 S	16	class XFileShareIE(InfoExtractor):
031ec536 S	17	IE_DESC = 'XFileShare based sites: GorillaVid.in, daclips.in, movpod.in, fastvideo.in, realvid.net, filehoot.com and vidto.me'
953b3586	18	_VALID_URL = r'''(?x)
aaefb347	19	https?://(?P<host>(?:www\.)?
9d584da7	20	(?:daclips\.in\|gorillavid\.in\|movpod\.in\|fastvideo\.in\|realvid\.net\|filehoot\.com\|vidto\.me))/
953b3586 PH	21	(?:embed-)?(?P<id>[0-9a-zA-Z]+)(?:-[0-9]+x[0-9]+\.html)?
953b3586 PH	22	'''
5f28a1ac	23
3ae165aa S	24	_FILE_NOT_FOUND_REGEX = r'>(?:404 - )?File Not Found<'
3ae165aa S	25
5f28a1ac PP	26	_TESTS = [{
	27	'url': 'http://gorillavid.in/06y9juieqpmi',
	28	'md5': '5ae4a3580620380619678ee4875893ba',
	29	'info_dict': {
	30	'id': '06y9juieqpmi',
	31	'ext': 'flv',
e4b85e35	32	'title': 'Rebecca Black My Moment Official Music Video Reaction-6GK87Rc8bzQ',
5f28a1ac PP	33	'thumbnail': 're:http://.*\.jpg',
	34	},
	35	}, {
	36	'url': 'http://gorillavid.in/embed-z08zf8le23c6-960x480.html',
1ed34f3d	37	'only_matching': True,
953b3586 PH	38	}, {
953b3586 PH	39	'url': 'http://daclips.in/3rso4kdn6f9m',
aaefb347	40	'md5': '1ad8fd39bb976eeb66004d3a4895f106',
953b3586 PH	41	'info_dict': {
	42	'id': '3rso4kdn6f9m',
	43	'ext': 'mp4',
2e9ff8f3	44	'title': 'Micro Pig piglets ready on 16th July 2009-bG0PdrCdxUc',
953b3586	45	'thumbnail': 're:http://.*\.jpg',
2e9ff8f3	46	}
ceb33673 S	47	}, {
	48	# video with countdown timeout
	49	'url': 'http://fastvideo.in/1qmdn1lmsmbw',
	50	'md5': '8b87ec3f6564a3108a0e8e66594842ba',
	51	'info_dict': {
	52	'id': '1qmdn1lmsmbw',
	53	'ext': 'mp4',
	54	'title': 'Man of Steel - Trailer',
	55	'thumbnail': 're:http://.*\.jpg',
	56	},
3eec9fef	57	}, {
	58	'url': 'http://realvid.net/ctn2y6p2eviw',
	59	'md5': 'b2166d2cf192efd6b6d764c18fd3710e',
	60	'info_dict': {
	61	'id': 'ctn2y6p2eviw',
	62	'ext': 'flv',
	63	'title': 'rdx 1955',
	64	'thumbnail': 're:http://.*\.jpg',
	65	},
b81f484b PH	66	}, {
	67	'url': 'http://movpod.in/0wguyyxi1yca',
	68	'only_matching': True,
c7c0996d S	69	}, {
	70	'url': 'http://filehoot.com/3ivfabn7573c.html',
	71	'info_dict': {
	72	'id': '3ivfabn7573c',
	73	'ext': 'mp4',
	74	'title': 'youtube-dl test video \'äBaW_jenozKc.mp4.mp4',
	75	'thumbnail': 're:http://.*\.jpg',
	76	}
668db403 S	77	}, {
	78	'url': 'http://vidto.me/ku5glz52nqe1.html',
	79	'info_dict': {
	80	'id': 'ku5glz52nqe1',
	81	'ext': 'mp4',
	82	'title': 'test'
	83	}
5f28a1ac	84	}]
617c0b22	85
	86	def _real_extract(self, url):
	87	mobj = re.match(self._VALID_URL, url)
	88	video_id = mobj.group('id')
	89
e213c98d S	90	url = 'http://%s/%s' % (mobj.group('host'), video_id)
e213c98d S	91	webpage = self._download_webpage(url, video_id)
617c0b22	92
3ae165aa S	93	if re.search(self._FILE_NOT_FOUND_REGEX, webpage) is not None:
	94	raise ExtractorError('Video %s does not exist' % video_id, expected=True)
	95
f8da79f8	96	fields = self._hidden_inputs(webpage)
5f6a1245	97
5f28a1ac	98	if fields['op'] == 'download1':
ceb33673 S	99	countdown = int_or_none(self._search_regex(
	100	r'<span id="countdown_str">(?:[Ww]ait)?\s<span id="cxc">(\d+)</span>\s(?:seconds?)?</span>',
	101	webpage, 'countdown', default=None))
	102	if countdown:
	103	self._sleep(countdown, video_id)
	104
4abe2144	105	post = compat_urllib_parse.urlencode(encode_dict(fields))
5f28a1ac	106
5c2266df	107	req = sanitized_Request(url, post)
5f28a1ac	108	req.add_header('Content-type', 'application/x-www-form-urlencoded')
617c0b22	109
5f28a1ac PP	110	webpage = self._download_webpage(req, video_id, 'Downloading video page')
5f28a1ac PP	111
668db403	112	title = (self._search_regex(
b9ad1019 S	113	[r'style="z-index: [0-9]+;">([^<]+)</span>',
	114	r'<td nowrap>([^<]+)</td>',
	115	r'>Watch (.+) ',
	116	r'<h2 class="video-page-head">([^<]+)</h2>'],
668db403	117	webpage, 'title', default=None) or self._og_search_title(webpage)).strip()
ceb33673	118	video_url = self._search_regex(
b9ad1019 S	119	[r'file\s:\s["\'](http[^"\']+)["\'],',
	120	r'file_link\s=\s\'(https?:\/\/[0-9a-zA-z.\/\-_]+)'],
	121	webpage, 'file url')
ceb33673	122	thumbnail = self._search_regex(
b9ad1019	123	r'image\s:\s["\'](http[^"\']+)["\'],', webpage, 'thumbnail', default=None)
5f28a1ac PP	124
	125	formats = [{
	126	'format_id': 'sd',
e4b85e35	127	'url': video_url,
5f28a1ac PP	128	'quality': 1,
	129	}]
	130
	131	return {
617c0b22	132	'id': video_id,
617c0b22	133	'title': title,
5f28a1ac PP	134	'thumbnail': thumbnail,
5f28a1ac PP	135	'formats': formats,
617c0b22	136	}