[yt-dlp.git] / test / test_download.py

#!/usr/bin/env python3

from __future__ import unicode_literals

# Allow direct execution
import os
import sys
import unittest
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))

from test.helper import (
    assertGreaterEqual,
    expect_info_dict,
    expect_warnings,
    get_params,
    gettestcases,
    is_download_test,
    report_warning,
    try_rm,
)


import hashlib
import io
import json
import socket

import yt_dlp.YoutubeDL
from yt_dlp.compat import (
    compat_http_client,
    compat_urllib_error,
    compat_HTTPError,
)
from yt_dlp.utils import (
    DownloadError,
    ExtractorError,
    format_bytes,
    UnavailableVideoError,
)
from yt_dlp.extractor import get_info_extractor

RETRIES = 3


class YoutubeDL(yt_dlp.YoutubeDL):
    def __init__(self, *args, **kwargs):
        self.to_stderr = self.to_screen
        self.processed_info_dicts = []
        super(YoutubeDL, self).__init__(*args, **kwargs)

    def report_warning(self, message):
        # Don't accept warnings during tests
        raise ExtractorError(message)

    def process_info(self, info_dict):
        self.processed_info_dicts.append(info_dict)
        return super(YoutubeDL, self).process_info(info_dict)


def _file_md5(fn):
    with open(fn, 'rb') as f:
        return hashlib.md5(f.read()).hexdigest()


defs = gettestcases()


@is_download_test
class TestDownload(unittest.TestCase):
    # Parallel testing in nosetests. See
    # http://nose.readthedocs.org/en/latest/doc_tests/test_multiprocess/multiprocess.html
    _multiprocess_shared_ = True

    maxDiff = None

    def __str__(self):
        """Identify each test with the `add_ie` attribute, if available."""

        def strclass(cls):
            """From 2.7's unittest; 2.6 had _strclass so we can't import it."""
            return '%s.%s' % (cls.__module__, cls.__name__)

        add_ie = getattr(self, self._testMethodName).add_ie
        return '%s (%s)%s:' % (self._testMethodName,
                               strclass(self.__class__),
                               ' [%s]' % add_ie if add_ie else '')

    def setUp(self):
        self.defs = defs

# Dynamically generate tests


def generator(test_case, tname):

    def test_template(self):
        ie = yt_dlp.extractor.get_info_extractor(test_case['name'])()
        other_ies = [get_info_extractor(ie_key)() for ie_key in test_case.get('add_ie', [])]
        is_playlist = any(k.startswith('playlist') for k in test_case)
        test_cases = test_case.get(
            'playlist', [] if is_playlist else [test_case])

        def print_skipping(reason):
            print('Skipping %s: %s' % (test_case['name'], reason))
        if not ie.working():
            print_skipping('IE marked as not _WORKING')
            return

        for tc in test_cases:
            info_dict = tc.get('info_dict', {})
            if not (info_dict.get('id') and info_dict.get('ext')):
                raise Exception('Test definition incorrect. The output file cannot be known. Are both \'id\' and \'ext\' keys present?')

        if 'skip' in test_case:
            print_skipping(test_case['skip'])
            return
        for other_ie in other_ies:
            if not other_ie.working():
                print_skipping('test depends on %sIE, marked as not WORKING' % other_ie.ie_key())
                return

        params = get_params(test_case.get('params', {}))
        params['outtmpl'] = tname + '_' + params['outtmpl']
        if is_playlist and 'playlist' not in test_case:
            params.setdefault('extract_flat', 'in_playlist')
            params.setdefault('playlistend', test_case.get('playlist_mincount'))
            params.setdefault('skip_download', True)

        ydl = YoutubeDL(params, auto_init=False)
        ydl.add_default_info_extractors()
        finished_hook_called = set()

        def _hook(status):
            if status['status'] == 'finished':
                finished_hook_called.add(status['filename'])
        ydl.add_progress_hook(_hook)
        expect_warnings(ydl, test_case.get('expected_warnings', []))

        def get_tc_filename(tc):
            return ydl.prepare_filename(tc.get('info_dict', {}))

        res_dict = None

        def try_rm_tcs_files(tcs=None):
            if tcs is None:
                tcs = test_cases
            for tc in tcs:
                tc_filename = get_tc_filename(tc)
                try_rm(tc_filename)
                try_rm(tc_filename + '.part')
                try_rm(os.path.splitext(tc_filename)[0] + '.info.json')
        try_rm_tcs_files()
        try:
            try_num = 1
            while True:
                try:
                    # We're not using .download here since that is just a shim
                    # for outside error handling, and returns the exit code
                    # instead of the result dict.
                    res_dict = ydl.extract_info(
                        test_case['url'],
                        force_generic_extractor=params.get('force_generic_extractor', False))
                except (DownloadError, ExtractorError) as err:
                    # Check if the exception is not a network related one
                    if not err.exc_info[0] in (compat_urllib_error.URLError, socket.timeout, UnavailableVideoError, compat_http_client.BadStatusLine) or (err.exc_info[0] == compat_HTTPError and err.exc_info[1].code == 503):
                        raise

                    if try_num == RETRIES:
                        report_warning('%s failed due to network errors, skipping...' % tname)
                        return

                    print('Retrying: {0} failed tries\n\n##########\n\n'.format(try_num))

                    try_num += 1
                else:
                    break

            if is_playlist:
                self.assertTrue(res_dict['_type'] in ['playlist', 'multi_video'])
                self.assertTrue('entries' in res_dict)
                expect_info_dict(self, res_dict, test_case.get('info_dict', {}))

            if 'playlist_mincount' in test_case:
                assertGreaterEqual(
                    self,
                    len(res_dict['entries']),
                    test_case['playlist_mincount'],
                    'Expected at least %d in playlist %s, but got only %d' % (
                        test_case['playlist_mincount'], test_case['url'],
                        len(res_dict['entries'])))
            if 'playlist_count' in test_case:
                self.assertEqual(
                    len(res_dict['entries']),
                    test_case['playlist_count'],
                    'Expected %d entries in playlist %s, but got %d.' % (
                        test_case['playlist_count'],
                        test_case['url'],
                        len(res_dict['entries']),
                    ))
            if 'playlist_duration_sum' in test_case:
                got_duration = sum(e['duration'] for e in res_dict['entries'])
                self.assertEqual(
                    test_case['playlist_duration_sum'], got_duration)

            # Generalize both playlists and single videos to unified format for
            # simplicity
            if 'entries' not in res_dict:
                res_dict['entries'] = [res_dict]

            for tc_num, tc in enumerate(test_cases):
                tc_res_dict = res_dict['entries'][tc_num]
                # First, check test cases' data against extracted data alone
                expect_info_dict(self, tc_res_dict, tc.get('info_dict', {}))
                # Now, check downloaded file consistency
                tc_filename = get_tc_filename(tc)
                if not test_case.get('params', {}).get('skip_download', False):
                    self.assertTrue(os.path.exists(tc_filename), msg='Missing file ' + tc_filename)
                    self.assertTrue(tc_filename in finished_hook_called)
                    expected_minsize = tc.get('file_minsize', 10000)
                    if expected_minsize is not None:
                        if params.get('test'):
                            expected_minsize = max(expected_minsize, 10000)
                        got_fsize = os.path.getsize(tc_filename)
                        assertGreaterEqual(
                            self, got_fsize, expected_minsize,
                            'Expected %s to be at least %s, but it\'s only %s ' %
                            (tc_filename, format_bytes(expected_minsize),
                                format_bytes(got_fsize)))
                    if 'md5' in tc:
                        md5_for_file = _file_md5(tc_filename)
                        self.assertEqual(tc['md5'], md5_for_file)
                # Finally, check test cases' data again but this time against
                # extracted data from info JSON file written during processing
                info_json_fn = os.path.splitext(tc_filename)[0] + '.info.json'
                self.assertTrue(
                    os.path.exists(info_json_fn),
                    'Missing info file %s' % info_json_fn)
                with io.open(info_json_fn, encoding='utf-8') as infof:
                    info_dict = json.load(infof)
                expect_info_dict(self, info_dict, tc.get('info_dict', {}))
        finally:
            try_rm_tcs_files()
            if is_playlist and res_dict is not None and res_dict.get('entries'):
                # Remove all other files that may have been extracted if the
                # extractor returns full results even with extract_flat
                res_tcs = [{'info_dict': e} for e in res_dict['entries']]
                try_rm_tcs_files(res_tcs)

    return test_template


# And add them to TestDownload
for n, test_case in enumerate(defs):
    tname = 'test_' + str(test_case['name'])
    i = 1
    while hasattr(TestDownload, tname):
        tname = 'test_%s_%d' % (test_case['name'], i)
        i += 1
    test_method = generator(test_case, tname)
    test_method.__name__ = str(tname)
    ie_list = test_case.get('add_ie')
    test_method.add_ie = ie_list and ','.join(ie_list)
    setattr(TestDownload, test_method.__name__, test_method)
    del test_method


if __name__ == '__main__':
    unittest.main()
Commit	Line	Data
cc52de43	1	#!/usr/bin/env python3
fd5ff020	2
a0f59cdc PH	3	from __future__ import unicode_literals
a0f59cdc PH	4
44a5f171 PH	5	# Allow direct execution
	6	import os
	7	import sys
	8	import unittest
	9	sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
	10
dd508b7c	11	from test.helper import (
0990305d	12	assertGreaterEqual,
060ac762	13	expect_info_dict,
70b7e3fb	14	expect_warnings,
dd508b7c	15	get_params,
ff14fc49	16	gettestcases,
060ac762	17	is_download_test,
257cfebf	18	report_warning,
060ac762	19	try_rm,
dd508b7c	20	)
44a5f171 PH	21
44a5f171 PH	22
efe8902f	23	import hashlib
fd5ff020	24	import io
7f60b5aa	25	import json
6b3aef80	26	import socket
cdab8aa3	27
7a5c1cfe P	28	import yt_dlp.YoutubeDL
7a5c1cfe P	29	from yt_dlp.compat import (
dcf3eec4	30	compat_http_client,
44a5f171	31	compat_urllib_error,
f6cc16f5	32	compat_HTTPError,
42f7d2f5	33	)
7a5c1cfe	34	from yt_dlp.utils import (
44a5f171 PH	35	DownloadError,
44a5f171 PH	36	ExtractorError,
753727cd	37	format_bytes,
44a5f171 PH	38	UnavailableVideoError,
44a5f171 PH	39	)
7a5c1cfe	40	from yt_dlp.extractor import get_info_extractor
fd5ff020	41
8cc83b8d FV	42	RETRIES = 3
8cc83b8d FV	43
5f6a1245	44
7a5c1cfe	45	class YoutubeDL(yt_dlp.YoutubeDL):
fd5ff020	46	def __init__(self, args, *kwargs):
fd5ff020	47	self.to_stderr = self.to_screen
0eaf520d	48	self.processed_info_dicts = []
8222d8de	49	super(YoutubeDL, self).__init__(args, *kwargs)
5f6a1245	50
476203d0	51	def report_warning(self, message):
be95cac1 FV	52	# Don't accept warnings during tests
be95cac1 FV	53	raise ExtractorError(message)
5f6a1245	54
0eaf520d FV	55	def process_info(self, info_dict):
0eaf520d FV	56	self.processed_info_dicts.append(info_dict)
8222d8de	57	return super(YoutubeDL, self).process_info(info_dict)
1535ac2a	58
5f6a1245	59
fd5ff020 FV	60	def _file_md5(fn):
	61	with open(fn, 'rb') as f:
	62	return hashlib.md5(f.read()).hexdigest()
	63
582be358	64
ff14fc49	65	defs = gettestcases()
6b47c7f2	66
0eaf520d	67
060ac762	68	@is_download_test
1535ac2a	69	class TestDownload(unittest.TestCase):
8936f68a YCH	70	# Parallel testing in nosetests. See
	71	# http://nose.readthedocs.org/en/latest/doc_tests/test_multiprocess/multiprocess.html
	72	_multiprocess_shared_ = True
	73
744435f2	74	maxDiff = None
5f6a1245	75
c6c22e98 JH	76	def __str__(self):
	77	"""Identify each test with the `add_ie` attribute, if available."""
	78
	79	def strclass(cls):
	80	"""From 2.7's unittest; 2.6 had _strclass so we can't import it."""
	81	return '%s.%s' % (cls.__module__, cls.__name__)
	82
	83	add_ie = getattr(self, self._testMethodName).add_ie
	84	return '%s (%s)%s:' % (self._testMethodName,
	85	strclass(self.__class__),
	86	' [%s]' % add_ie if add_ie else '')
	87
fd5ff020	88	def setUp(self):
fd5ff020 FV	89	self.defs = defs
fd5ff020 FV	90
5f6a1245 JW	91	# Dynamically generate tests
	92
	93
8936f68a	94	def generator(test_case, tname):
5d01a647	95
1535ac2a	96	def test_template(self):
7a5c1cfe	97	ie = yt_dlp.extractor.get_info_extractor(test_case['name'])()
655c4100	98	other_ies = [get_info_extractor(ie_key)() for ie_key in test_case.get('add_ie', [])]
e8ee972c PH	99	is_playlist = any(k.startswith('playlist') for k in test_case)
	100	test_cases = test_case.get(
	101	'playlist', [] if is_playlist else [test_case])
	102
bc2884af JMF	103	def print_skipping(reason):
bc2884af JMF	104	print('Skipping %s: %s' % (test_case['name'], reason))
9ee2b5f6	105	if not ie.working():
bc2884af	106	print_skipping('IE marked as not _WORKING')
fd5ff020	107	return
e8ee972c PH	108
	109	for tc in test_cases:
	110	info_dict = tc.get('info_dict', {})
4e980275	111	if not (info_dict.get('id') and info_dict.get('ext')):
2437fbca	112	raise Exception('Test definition incorrect. The output file cannot be known. Are both \'id\' and \'ext\' keys present?')
e8ee972c	113
fd5ff020	114	if 'skip' in test_case:
bc2884af	115	print_skipping(test_case['skip'])
fd5ff020	116	return
9ee2b5f6 JMF	117	for other_ie in other_ies:
9ee2b5f6 JMF	118	if not other_ie.working():
e075a44a	119	print_skipping('test depends on %sIE, marked as not WORKING' % other_ie.ie_key())
9ee2b5f6	120	return
0eaf520d	121
44a5f171	122	params = get_params(test_case.get('params', {}))
8936f68a	123	params['outtmpl'] = tname + '_' + params['outtmpl']
e8ee972c	124	if is_playlist and 'playlist' not in test_case:
65d49afa	125	params.setdefault('extract_flat', 'in_playlist')
6911e11e	126	params.setdefault('playlistend', test_case.get('playlist_mincount'))
e8ee972c	127	params.setdefault('skip_download', True)
0eaf520d	128
ac35c266	129	ydl = YoutubeDL(params, auto_init=False)
023fa8c4	130	ydl.add_default_info_extractors()
bffbd5f0	131	finished_hook_called = set()
5f6a1245	132
bffbd5f0 PH	133	def _hook(status):
	134	if status['status'] == 'finished':
	135	finished_hook_called.add(status['filename'])
933605d7	136	ydl.add_progress_hook(_hook)
70b7e3fb	137	expect_warnings(ydl, test_case.get('expected_warnings', []))
5c892b0b	138
702665c0	139	def get_tc_filename(tc):
4e980275	140	return ydl.prepare_filename(tc.get('info_dict', {}))
702665c0	141
28570840	142	res_dict = None
5f6a1245	143
28570840 PH	144	def try_rm_tcs_files(tcs=None):
	145	if tcs is None:
	146	tcs = test_cases
	147	for tc in tcs:
702665c0 JMF	148	tc_filename = get_tc_filename(tc)
	149	try_rm(tc_filename)
	150	try_rm(tc_filename + '.part')
4eb92208	151	try_rm(os.path.splitext(tc_filename)[0] + '.info.json')
702665c0	152	try_rm_tcs_files()
5c892b0b	153	try:
dd508b7c FV	154	try_num = 1
dd508b7c FV	155	while True:
8cc83b8d	156	try:
3bef10a5	157	# We're not using .download here since that is just a shim
e8ee972c PH	158	# for outside error handling, and returns the exit code
e8ee972c PH	159	# instead of the result dict.
308cfe0a S	160	res_dict = ydl.extract_info(
	161	test_case['url'],
	162	force_generic_extractor=params.get('force_generic_extractor', False))
8cc83b8d	163	except (DownloadError, ExtractorError) as err:
8cc83b8d	164	# Check if the exception is not a network related one
dcf3eec4	165	if not err.exc_info[0] in (compat_urllib_error.URLError, socket.timeout, UnavailableVideoError, compat_http_client.BadStatusLine) or (err.exc_info[0] == compat_HTTPError and err.exc_info[1].code == 503):
8cc83b8d FV	166	raise
8cc83b8d FV	167
dd508b7c	168	if try_num == RETRIES:
8936f68a	169	report_warning('%s failed due to network errors, skipping...' % tname)
dd508b7c FV	170	return
	171
	172	print('Retrying: {0} failed tries\n\n##########\n\n'.format(try_num))
	173
	174	try_num += 1
8cc83b8d FV	175	else:
8cc83b8d FV	176	break
5c892b0b	177
e8ee972c	178	if is_playlist:
880ee801	179	self.assertTrue(res_dict['_type'] in ['playlist', 'multi_video'])
d6e6a422	180	self.assertTrue('entries' in res_dict)
f74b341d	181	expect_info_dict(self, res_dict, test_case.get('info_dict', {}))
d6e6a422	182
e8ee972c	183	if 'playlist_mincount' in test_case:
0990305d PH	184	assertGreaterEqual(
0990305d PH	185	self,
e8ee972c PH	186	len(res_dict['entries']),
	187	test_case['playlist_mincount'],
	188	'Expected at least %d in playlist %s, but got only %d' % (
	189	test_case['playlist_mincount'], test_case['url'],
	190	len(res_dict['entries'])))
829476b8 PH	191	if 'playlist_count' in test_case:
	192	self.assertEqual(
	193	len(res_dict['entries']),
	194	test_case['playlist_count'],
28570840	195	'Expected %d entries in playlist %s, but got %d.' % (
22a6f150	196	test_case['playlist_count'],
28570840	197	test_case['url'],
22a6f150 PH	198	len(res_dict['entries']),
22a6f150 PH	199	))
28570840 PH	200	if 'playlist_duration_sum' in test_case:
	201	got_duration = sum(e['duration'] for e in res_dict['entries'])
	202	self.assertEqual(
	203	test_case['playlist_duration_sum'], got_duration)
e8ee972c	204
364a69e8 S	205	# Generalize both playlists and single videos to unified format for
	206	# simplicity
	207	if 'entries' not in res_dict:
	208	res_dict['entries'] = [res_dict]
	209
80b2fdf9	210	for tc_num, tc in enumerate(test_cases):
364a69e8 S	211	tc_res_dict = res_dict['entries'][tc_num]
364a69e8 S	212	# First, check test cases' data against extracted data alone
80b2fdf9	213	expect_info_dict(self, tc_res_dict, tc.get('info_dict', {}))
364a69e8	214	# Now, check downloaded file consistency
702665c0	215	tc_filename = get_tc_filename(tc)
511eda8e	216	if not test_case.get('params', {}).get('skip_download', False):
702665c0 JMF	217	self.assertTrue(os.path.exists(tc_filename), msg='Missing file ' + tc_filename)
702665c0 JMF	218	self.assertTrue(tc_filename in finished_hook_called)
08a36c35 S	219	expected_minsize = tc.get('file_minsize', 10000)
	220	if expected_minsize is not None:
	221	if params.get('test'):
	222	expected_minsize = max(expected_minsize, 10000)
	223	got_fsize = os.path.getsize(tc_filename)
	224	assertGreaterEqual(
	225	self, got_fsize, expected_minsize,
	226	'Expected %s to be at least %s, but it\'s only %s ' %
	227	(tc_filename, format_bytes(expected_minsize),
	228	format_bytes(got_fsize)))
	229	if 'md5' in tc:
	230	md5_for_file = _file_md5(tc_filename)
374560f0	231	self.assertEqual(tc['md5'], md5_for_file)
364a69e8 S	232	# Finally, check test cases' data again but this time against
364a69e8 S	233	# extracted data from info JSON file written during processing
4eb92208	234	info_json_fn = os.path.splitext(tc_filename)[0] + '.info.json'
f744c0f3 PH	235	self.assertTrue(
	236	os.path.exists(info_json_fn),
	237	'Missing info file %s' % info_json_fn)
4eb92208	238	with io.open(info_json_fn, encoding='utf-8') as infof:
5c892b0b	239	info_dict = json.load(infof)
f74b341d	240	expect_info_dict(self, info_dict, tc.get('info_dict', {}))
5c892b0b	241	finally:
702665c0	242	try_rm_tcs_files()
d6e6a422	243	if is_playlist and res_dict is not None and res_dict.get('entries'):
28570840 PH	244	# Remove all other files that may have been extracted if the
	245	# extractor returns full results even with extract_flat
	246	res_tcs = [{'info_dict': e} for e in res_dict['entries']]
	247	try_rm_tcs_files(res_tcs)
fd5ff020	248
1535ac2a	249	return test_template
fd5ff020	250
582be358	251
5f6a1245	252	# And add them to TestDownload
f7ab6cbe	253	for n, test_case in enumerate(defs):
2eb88d95 PH	254	tname = 'test_' + str(test_case['name'])
	255	i = 1
	256	while hasattr(TestDownload, tname):
a0f59cdc	257	tname = 'test_%s_%d' % (test_case['name'], i)
2eb88d95	258	i += 1
8936f68a	259	test_method = generator(test_case, tname)
a0f59cdc	260	test_method.__name__ = str(tname)
c6c22e98 JH	261	ie_list = test_case.get('add_ie')
c6c22e98 JH	262	test_method.add_ie = ie_list and ','.join(ie_list)
fd5ff020	263	setattr(TestDownload, test_method.__name__, test_method)
5d01a647	264	del test_method
cdab8aa3 PH	265
	266
	267	if __name__ == '__main__':
	268	unittest.main()