[yt-dlp.git] / youtube_dl / extractor / imgur.py

from __future__ import unicode_literals

import re

from .common import InfoExtractor
from ..compat import compat_urlparse
from ..utils import (
    int_or_none,
    js_to_json,
    mimetype2ext,
    ExtractorError,
)


class ImgurIE(InfoExtractor):
    _VALID_URL = r'https?://(?:i\.)?imgur\.com/(gallery/)?(?P<id>[a-zA-Z0-9]{6,})'

    _TESTS = [{
        'url': 'https://i.imgur.com/A61SaA1.gifv',
        'info_dict': {
            'id': 'A61SaA1',
            'ext': 'mp4',
            'title': 're:Imgur GIF$|MRW gifv is up and running without any bugs$',
            'description': 'Imgur: The most awesome images on the Internet.',
        },
    }, {
        'url': 'https://imgur.com/A61SaA1',
        'info_dict': {
            'id': 'A61SaA1',
            'ext': 'mp4',
            'title': 're:Imgur GIF$|MRW gifv is up and running without any bugs$',
            'description': 'Imgur: The most awesome images on the Internet.',
        },
    }, {
        'url': 'https://imgur.com/gallery/YcAQlkx',
        'info_dict': {
            'id': 'YcAQlkx',
            'ext': 'mp4',
            'title': 'Classic Steve Carell gif...cracks me up everytime....damn the repost downvotes....',
            'description': 'Imgur: The most awesome images on the Internet.'

        }
    }]

    def _real_extract(self, url):
        video_id = self._match_id(url)
        webpage = self._download_webpage(
            compat_urlparse.urljoin(url, video_id), video_id)

        width = int_or_none(self._search_regex(
            r'<param name="width" value="([0-9]+)"',
            webpage, 'width', fatal=False))
        height = int_or_none(self._search_regex(
            r'<param name="height" value="([0-9]+)"',
            webpage, 'height', fatal=False))

        video_elements = self._search_regex(
            r'(?s)<div class="video-elements">(.*?)</div>',
            webpage, 'video elements', default=None)
        if not video_elements:
            raise ExtractorError(
                'No sources found for video %s. Maybe an image?' % video_id,
                expected=True)

        formats = []
        for m in re.finditer(r'<source\s+src="(?P<src>[^"]+)"\s+type="(?P<type>[^"]+)"', video_elements):
            formats.append({
                'format_id': m.group('type').partition('/')[2],
                'url': self._proto_relative_url(m.group('src')),
                'ext': mimetype2ext(m.group('type')),
                'acodec': 'none',
                'width': width,
                'height': height,
                'http_headers': {
                    'User-Agent': 'youtube-dl (like wget)',
                },
            })

        gif_json = self._search_regex(
            r'(?s)var\s+videoItem\s*=\s*(\{.*?\})',
            webpage, 'GIF code', fatal=False)
        if gif_json:
            gifd = self._parse_json(
                gif_json, video_id, transform_source=js_to_json)
            formats.append({
                'format_id': 'gif',
                'preference': -10,
                'width': width,
                'height': height,
                'ext': 'gif',
                'acodec': 'none',
                'vcodec': 'gif',
                'container': 'gif',
                'url': self._proto_relative_url(gifd['gifUrl']),
                'filesize': gifd.get('size'),
                'http_headers': {
                    'User-Agent': 'youtube-dl (like wget)',
                },
            })

        self._sort_formats(formats)

        return {
            'id': video_id,
            'formats': formats,
            'description': self._og_search_description(webpage),
            'title': self._og_search_title(webpage),
        }


class ImgurAlbumIE(InfoExtractor):
    _VALID_URL = r'https?://(?:i\.)?imgur\.com/(gallery/)?(?P<id>[a-zA-Z0-9]{5})(?![a-zA-Z0-9])'

    _TEST = {
        'url': 'http://imgur.com/gallery/Q95ko',
        'info_dict': {
            'id': 'Q95ko',
        },
        'playlist_count': 25,
    }

    def _real_extract(self, url):
        album_id = self._match_id(url)

        album_img_data = self._download_json(
            'http://imgur.com/gallery/%s/album_images/hit.json?all=true' % album_id, album_id)['data']

        if len(album_img_data) == 0:
            return self.url_result('http://imgur.com/%s' % album_id)
        else:
            album_images = album_img_data['images']
            entries = [
                self.url_result('http://imgur.com/%s' % image['hash'])
                for image in album_images if image.get('hash')]

        return self.playlist_result(entries, album_id)
Commit	Line	Data
3bf57053 PH	1	from __future__ import unicode_literals
	2
	3	import re
	4
	5	from .common import InfoExtractor
96b96909	6	from ..compat import compat_urlparse
3bf57053 PH	7	from ..utils import (
	8	int_or_none,
	9	js_to_json,
	10	mimetype2ext,
1a13940c	11	ExtractorError,
3bf57053 PH	12	)
3bf57053 PH	13
b88ba053	14
3bf57053	15	class ImgurIE(InfoExtractor):
dbee18b5	16	_VALID_URL = r'https?://(?:i\.)?imgur\.com/(gallery/)?(?P<id>[a-zA-Z0-9]{6,})'
3bf57053 PH	17
	18	_TESTS = [{
	19	'url': 'https://i.imgur.com/A61SaA1.gifv',
	20	'info_dict': {
	21	'id': 'A61SaA1',
	22	'ext': 'mp4',
5e9a033e	23	'title': 're:Imgur GIF$\|MRW gifv is up and running without any bugs$',
dbee18b5	24	'description': 'Imgur: The most awesome images on the Internet.',
3bf57053	25	},
1a13940c JB	26	}, {
	27	'url': 'https://imgur.com/A61SaA1',
	28	'info_dict': {
	29	'id': 'A61SaA1',
	30	'ext': 'mp4',
5e9a033e	31	'title': 're:Imgur GIF$\|MRW gifv is up and running without any bugs$',
dbee18b5	32	'description': 'Imgur: The most awesome images on the Internet.',
1a13940c	33	},
dbee18b5 AK	34	}, {
	35	'url': 'https://imgur.com/gallery/YcAQlkx',
	36	'info_dict': {
	37	'id': 'YcAQlkx',
	38	'ext': 'mp4',
	39	'title': 'Classic Steve Carell gif...cracks me up everytime....damn the repost downvotes....',
	40	'description': 'Imgur: The most awesome images on the Internet.'
	41
	42	}
3bf57053 PH	43	}]
	44
	45	def _real_extract(self, url):
	46	video_id = self._match_id(url)
96b96909 S	47	webpage = self._download_webpage(
96b96909 S	48	compat_urlparse.urljoin(url, video_id), video_id)
3bf57053 PH	49
	50	width = int_or_none(self._search_regex(
	51	r'<param name="width" value="([0-9]+)"',
	52	webpage, 'width', fatal=False))
	53	height = int_or_none(self._search_regex(
	54	r'<param name="height" value="([0-9]+)"',
	55	webpage, 'height', fatal=False))
	56
b88ba053	57	video_elements = self._search_regex(
3bf57053	58	r'(?s)<div class="video-elements">(.*?)</div>',
b88ba053	59	webpage, 'video elements', default=None)
9e2d7dca JB	60	if not video_elements:
9e2d7dca JB	61	raise ExtractorError(
b88ba053 PH	62	'No sources found for video %s. Maybe an image?' % video_id,
b88ba053 PH	63	expected=True)
9e2d7dca	64
3bf57053 PH	65	formats = []
	66	for m in re.finditer(r'<source\s+src="(?P<src>[^"]+)"\s+type="(?P<type>[^"]+)"', video_elements):
	67	formats.append({
	68	'format_id': m.group('type').partition('/')[2],
	69	'url': self._proto_relative_url(m.group('src')),
	70	'ext': mimetype2ext(m.group('type')),
	71	'acodec': 'none',
	72	'width': width,
	73	'height': height,
	74	'http_headers': {
	75	'User-Agent': 'youtube-dl (like wget)',
	76	},
	77	})
	78
	79	gif_json = self._search_regex(
	80	r'(?s)var\s+videoItem\s=\s(\{.*?\})',
	81	webpage, 'GIF code', fatal=False)
	82	if gif_json:
	83	gifd = self._parse_json(
	84	gif_json, video_id, transform_source=js_to_json)
	85	formats.append({
	86	'format_id': 'gif',
	87	'preference': -10,
	88	'width': width,
	89	'height': height,
	90	'ext': 'gif',
	91	'acodec': 'none',
	92	'vcodec': 'gif',
	93	'container': 'gif',
	94	'url': self._proto_relative_url(gifd['gifUrl']),
	95	'filesize': gifd.get('size'),
	96	'http_headers': {
	97	'User-Agent': 'youtube-dl (like wget)',
	98	},
	99	})
	100
	101	self._sort_formats(formats)
	102
	103	return {
	104	'id': video_id,
	105	'formats': formats,
	106	'description': self._og_search_description(webpage),
	107	'title': self._og_search_title(webpage),
	108	}
8875b3d5 S	109
	110
	111	class ImgurAlbumIE(InfoExtractor):
dbee18b5	112	_VALID_URL = r'https?://(?:i\.)?imgur\.com/(gallery/)?(?P<id>[a-zA-Z0-9]{5})(?![a-zA-Z0-9])'
8875b3d5 S	113
	114	_TEST = {
	115	'url': 'http://imgur.com/gallery/Q95ko',
	116	'info_dict': {
	117	'id': 'Q95ko',
	118	},
	119	'playlist_count': 25,
	120	}
	121
	122	def _real_extract(self, url):
	123	album_id = self._match_id(url)
	124
dbee18b5 AK	125	album_img_data = self._download_json(
dbee18b5 AK	126	'http://imgur.com/gallery/%s/album_images/hit.json?all=true' % album_id, album_id)['data']
8875b3d5	127
dbee18b5 AK	128	if len(album_img_data) == 0:
	129	return self.url_result('http://imgur.com/%s' % album_id)
	130	else:
	131	album_images = album_img_data['images']
	132	entries = [
	133	self.url_result('http://imgur.com/%s' % image['hash'])
	134	for image in album_images if image.get('hash')]
8875b3d5 S	135
8875b3d5 S	136	return self.playlist_result(entries, album_id)