Merge pull request #1602 from meeb/tcely-curl-thumbnails
fix: download thumbnails with curl
This commit is contained in:
@@ -43,7 +43,8 @@ from ._migrations import (
|
||||
from ._private import _srctype_dict, _nfo_element
|
||||
from .media__tasks import (
|
||||
copy_thumbnail, download_checklist, download_finished,
|
||||
failed_format, refresh_formats, wait_for_premiere, write_nfo_file,
|
||||
download_thumbnails, failed_format, refresh_formats,
|
||||
wait_for_premiere, write_nfo_file,
|
||||
)
|
||||
from .source import Source
|
||||
|
||||
@@ -1238,6 +1239,7 @@ class Media(models.Model):
|
||||
Media.copy_thumbnail = copy_thumbnail
|
||||
Media.download_checklist = download_checklist
|
||||
Media.download_finished = download_finished
|
||||
Media.download_thumbnails = download_thumbnails
|
||||
Media.failed_format = failed_format
|
||||
Media.refresh_formats = refresh_formats
|
||||
Media.wait_for_premiere = wait_for_premiere
|
||||
|
||||
@@ -1,17 +1,29 @@
|
||||
import io
|
||||
import os
|
||||
import subprocess
|
||||
from collections import defaultdict
|
||||
from pathlib import Path
|
||||
from shutil import copyfile
|
||||
from common.logger import log
|
||||
from pathlib import Path, PurePosixPath
|
||||
from shutil import copyfile, rmtree
|
||||
from tempfile import TemporaryDirectory
|
||||
from urllib.parse import urlparse, urlunparse
|
||||
|
||||
from PIL import Image
|
||||
|
||||
from django.conf import settings
|
||||
from django.core.files.uploadedfile import SimpleUploadedFile
|
||||
from django.utils import timezone
|
||||
from django.utils.translation import gettext_lazy as _
|
||||
|
||||
from common.errors import (
|
||||
NoMetadataException,
|
||||
)
|
||||
from common.utils import multi_key_sort
|
||||
from django.conf import settings
|
||||
from django.utils import timezone
|
||||
from django.utils.translation import gettext_lazy as _
|
||||
from common.logger import log
|
||||
from common.utils import getenv, multi_key_sort
|
||||
from common.yt_dlp import retry_django_db
|
||||
from ..choices import Val, SourceResolution
|
||||
from ..utils import filter_response, write_text_file
|
||||
from ..utils import (
|
||||
filter_response, resize_image_to_height, write_text_file
|
||||
)
|
||||
|
||||
|
||||
def copy_thumbnail(self):
|
||||
@@ -69,12 +81,11 @@ def download_checklist(self, skip_checks=False):
|
||||
return False
|
||||
max_cap_age = media.source.download_cap_date
|
||||
published = media.published
|
||||
if max_cap_age and published:
|
||||
if published <= max_cap_age:
|
||||
log.warn(f'Download task triggered media: {media} (UUID: {media.pk}) but '
|
||||
f'the source has a download cap and the media is now too old, '
|
||||
f'not downloading')
|
||||
return False
|
||||
if max_cap_age and published and published <= max_cap_age:
|
||||
log.warn(f'Download task triggered media: {media} (UUID: {media.pk}) but '
|
||||
f'the source has a download cap and the media is now too old, '
|
||||
f'not downloading')
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
@@ -128,6 +139,309 @@ def download_finished(self, format_str, container, downloaded_filepath=None):
|
||||
self.downloaded_hdr = False
|
||||
|
||||
|
||||
def make_youtube_thumbnail_urls(video_id: str, output_format: str = 'string') -> str | tuple[dict[str, str], ...]:
|
||||
"""
|
||||
Generates YouTube thumbnail URLs using urlunparse.
|
||||
Defaults to raw 'string' output, but can be extended.
|
||||
"""
|
||||
|
||||
# Format serialization functions mapped inside the registry
|
||||
def _format_as_string(urls, headers, rows) -> str:
|
||||
return '\n'.join(urls)
|
||||
|
||||
def _format_as_dicts(urls, headers, rows) -> tuple[dict[str, str], ...]:
|
||||
return tuple({'url': urls[i], 'filename': rows[i][3]} for i in range(len(urls)))
|
||||
|
||||
FORMATTER_MAP = {
|
||||
'string': _format_as_string,
|
||||
'dicts': _format_as_dicts,
|
||||
}
|
||||
|
||||
scheme = 'https'
|
||||
hostname = 'i.ytimg.com'
|
||||
base_names = (
|
||||
'maxresdefault', 'sddefault', 'hqdefault', '1', '2', '3',
|
||||
'oardefault', 'oar1', 'oar2', 'oar3',
|
||||
)
|
||||
|
||||
urls = []
|
||||
for name in base_names:
|
||||
# Construct full clean URLs directly using urlunparse tuples
|
||||
# Tuple format: (scheme, netloc, path, params, query, fragment)
|
||||
jpg = urlunparse((scheme, hostname, f'/vi/{video_id}/{name}.jpg', '', '', ''))
|
||||
webp = urlunparse((scheme, hostname, f'/vi_webp/{video_id}/{name}.webp', '', '', ''))
|
||||
|
||||
urls.extend((jpg, webp))
|
||||
|
||||
headers = ('Scheme', 'Hostname', 'Path', 'Filename')
|
||||
rows = []
|
||||
for url in urls:
|
||||
parsed = urlparse(url)
|
||||
rows.append((
|
||||
parsed.scheme,
|
||||
parsed.hostname,
|
||||
parsed.path,
|
||||
PurePosixPath(parsed.path).name,
|
||||
))
|
||||
|
||||
fmt = output_format.lower().strip()
|
||||
if fmt not in FORMATTER_MAP:
|
||||
raise ValueError(f'Unsupported format "{output_format}".')
|
||||
|
||||
return FORMATTER_MAP[fmt](urls, headers, rows)
|
||||
|
||||
|
||||
def download_thumbnails_pycurl(self) -> Path | None:
|
||||
import pycurl
|
||||
|
||||
def download_thumbnails_parallel(video_id: str, max_connections: int = 2) -> dict[str, io.BytesIO]:
|
||||
"""
|
||||
Executes high-performance parallel downloads via pycurl completely in memory.
|
||||
Returns a dictionary mapping filenames to populated BytesIO buffers.
|
||||
"""
|
||||
url_targets = make_youtube_thumbnail_urls(video_id=video_id, output_format='dicts')
|
||||
|
||||
downloaded_buffers = {}
|
||||
|
||||
# This is a quick and dirty attempt to support proxies.
|
||||
env_proxy = getenv('https_proxy') or getenv('http_proxy') or getenv('all_proxy')
|
||||
no_proxy_setting = getenv('no_proxy')
|
||||
|
||||
with pycurl.CurlMulti() as multi:
|
||||
multi.setopt(pycurl.M_PIPELINING, pycurl.PIPE_NOTHING)
|
||||
multi.setopt(pycurl.M_MAX_HOST_CONNECTIONS, max_connections)
|
||||
|
||||
connections = {}
|
||||
|
||||
for target in url_targets:
|
||||
c = pycurl.Curl()
|
||||
c.setopt(c.URL, target['url'])
|
||||
c.setopt(c.FOLLOWLOCATION, True)
|
||||
c.setopt(c.FAILONERROR, True)
|
||||
|
||||
buffer = io.BytesIO()
|
||||
c.setopt(c.WRITEDATA, buffer)
|
||||
|
||||
if env_proxy:
|
||||
c.setopt(pycurl.PROXY, env_proxy)
|
||||
if no_proxy_setting:
|
||||
c.setopt(pycurl.NOPROXY, no_proxy_setting)
|
||||
|
||||
multi.add_handle(c)
|
||||
connections[c] = {'filename': target['filename'], 'buffer': buffer}
|
||||
c = buffer = None
|
||||
|
||||
num_handles = 1
|
||||
while 0 < num_handles:
|
||||
ret, num_handles = multi.perform()
|
||||
if ret != pycurl.E_CALL_MULTI_PERFORM:
|
||||
multi.select(1.0)
|
||||
|
||||
num_q = 1
|
||||
while 0 < num_q:
|
||||
# err_list never used
|
||||
# ruff: ignore[RUF059]
|
||||
num_q, ok_list, err_list = multi.info_read()
|
||||
for curl in ok_list:
|
||||
info = connections[curl]
|
||||
buf = info['buffer']
|
||||
buf.seek(0, io.SEEK_END)
|
||||
if 0 < buf.tell():
|
||||
buf.seek(0, io.SEEK_SET)
|
||||
downloaded_buffers[info['filename']] = buf
|
||||
|
||||
for curl in connections:
|
||||
curl.close()
|
||||
|
||||
log.debug(f'Parallel download pass completed. Successfully stored {len(downloaded_buffers)} valid buffers for: {video_id}')
|
||||
return downloaded_buffers
|
||||
|
||||
downloaded_data = download_thumbnails_parallel(self.key)
|
||||
if not downloaded_data:
|
||||
return
|
||||
|
||||
chosen_filename = PurePosixPath(self.thumbnail).name
|
||||
width = getattr(settings, 'MEDIA_THUMBNAIL_WIDTH', 430)
|
||||
height = getattr(settings, 'MEDIA_THUMBNAIL_HEIGHT', 240)
|
||||
saved_size = (0, 0)
|
||||
thumb_path = None
|
||||
|
||||
for filename, buffer in downloaded_data.items():
|
||||
filename_path = Path(filename)
|
||||
|
||||
# accept: maxres webp, the filename from self.thumbnail, or any jpg thumbnails
|
||||
if not (
|
||||
chosen_filename == filename_path.name or
|
||||
'.jpg' == filename_path.suffix or
|
||||
'maxresdefault' == filename_path.stem
|
||||
):
|
||||
continue
|
||||
|
||||
image_file = io.BytesIO()
|
||||
with Image.open(buffer) as img:
|
||||
if img.size < saved_size:
|
||||
continue
|
||||
saved_size = img.size
|
||||
if 'RGB' != img.mode:
|
||||
img = img.convert('RGB')
|
||||
if (img.width > width) and (img.height > height):
|
||||
log.debug(f'Resizing {img.width}x{img.height} thumbnail to '
|
||||
f'{width}x{height}: {filename_path.name}')
|
||||
img = resize_image_to_height(img, width, height)
|
||||
img.save(image_file, 'JPEG', quality=85, optimize=True, progressive=True)
|
||||
|
||||
img = None
|
||||
image_file.seek(0, io.SEEK_SET)
|
||||
thumbnail_bytes = image_file.read()
|
||||
image_file = None
|
||||
|
||||
if self.thumb_file_exists:
|
||||
self.thumb.delete(save=False)
|
||||
|
||||
retry_django_db(5)(self.thumb.save)(
|
||||
'thumb',
|
||||
SimpleUploadedFile(
|
||||
'thumb',
|
||||
thumbnail_bytes,
|
||||
'image/jpeg',
|
||||
),
|
||||
save=True,
|
||||
)
|
||||
|
||||
thumbnail_bytes = None
|
||||
thumb_path = filename_path
|
||||
if chosen_filename == filename_path.name:
|
||||
break
|
||||
|
||||
copy_thumbnail(self)
|
||||
|
||||
if thumb_path is None:
|
||||
return
|
||||
|
||||
return thumb_path
|
||||
|
||||
|
||||
def download_thumbnails(self) -> Path | None:
|
||||
def export_urls_to_temp_file(video_id: str) -> str:
|
||||
"""Writes URLs.txt using the 'string' format."""
|
||||
|
||||
prefix = f'i.ytimg.com-thumbnails-[{video_id}]-'
|
||||
with TemporaryDirectory(prefix=prefix, delete=False) as temp_dir:
|
||||
file_path = Path(temp_dir) / 'URLs.txt'
|
||||
with open(file_path, 'w') as f:
|
||||
f.write(make_youtube_thumbnail_urls(video_id=video_id, output_format='string'))
|
||||
f.write('\n')
|
||||
|
||||
return file_path
|
||||
|
||||
def download_thumbnails_parallel(video_id: str, max_connections: int = 2) -> tuple[Path,...]:
|
||||
"""
|
||||
Executes high-performance parallel downloads via curl.
|
||||
Attempts all URLs, but strictly skips writing files for any 404 responses.
|
||||
"""
|
||||
|
||||
file_path = Path(export_urls_to_temp_file(video_id))
|
||||
curl_command = (
|
||||
'curl',
|
||||
'--parallel', '--parallel-immediate',
|
||||
# Debian 13 curl is too old for this option
|
||||
# '--parallel-max-host', str(max_connections),
|
||||
'--parallel-max', str(max_connections),
|
||||
'--remote-time', '--remote-name-all',
|
||||
'--fail', '--verbose', '--show-error',
|
||||
'--dump-header', 'curl.header.log.txt',
|
||||
'--stderr', 'curl.stderr.log.txt',
|
||||
'--url', f'@{file_path.name}',
|
||||
)
|
||||
|
||||
try:
|
||||
subprocess.run(
|
||||
curl_command,
|
||||
cwd=str(file_path.parent),
|
||||
check=False,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
)
|
||||
|
||||
downloaded_files = tuple(
|
||||
e_path for e in os.scandir(file_path.parent)
|
||||
if (e_path := Path(e.path)).suffix in ('.jpg', '.webp') and e.is_file() and 0 < e.stat().st_size
|
||||
)
|
||||
|
||||
log.debug(f'Parallel download pass completed. Successfully stored {len(downloaded_files)} valid files for: {video_id}')
|
||||
|
||||
return downloaded_files
|
||||
|
||||
except FileNotFoundError:
|
||||
raise RuntimeError('Missing dependencies: "curl" executable was not found on your system environment PATH.')
|
||||
|
||||
paths = download_thumbnails_parallel(self.key)
|
||||
if not paths:
|
||||
return
|
||||
|
||||
chosen_filename = PurePosixPath(self.thumbnail).name
|
||||
width = getattr(settings, 'MEDIA_THUMBNAIL_WIDTH', 430)
|
||||
height = getattr(settings, 'MEDIA_THUMBNAIL_HEIGHT', 240)
|
||||
saved_size = (0, 0)
|
||||
thumb_path = None
|
||||
try:
|
||||
for e_path in paths:
|
||||
# accept: maxres webp, the filename from self.thumbnail, or any jpg thumbnails
|
||||
if not (
|
||||
chosen_filename == e_path.name or
|
||||
'.jpg' == e_path.suffix or
|
||||
'maxresdefault' == e_path.stem
|
||||
):
|
||||
continue
|
||||
|
||||
image_file = io.BytesIO()
|
||||
with Image.open(e_path) as img:
|
||||
if img.size < saved_size:
|
||||
continue
|
||||
saved_size = img.size
|
||||
if 'RGB' != img.mode:
|
||||
img = img.convert('RGB')
|
||||
if (img.width > width) and (img.height > height):
|
||||
log.debug(f'Resizing {img.width}x{img.height} thumbnail to '
|
||||
f'{width}x{height}: {e_path.name}')
|
||||
img = resize_image_to_height(img, width, height)
|
||||
img.save(image_file, 'JPEG', quality=85, optimize=True, progressive=True)
|
||||
|
||||
img = None
|
||||
image_file.seek(0, io.SEEK_SET)
|
||||
thumbnail_bytes = image_file.read()
|
||||
image_file = None
|
||||
|
||||
if self.thumb_file_exists:
|
||||
self.thumb.delete(save=False)
|
||||
|
||||
retry_django_db(5)(self.thumb.save)(
|
||||
'thumb',
|
||||
SimpleUploadedFile(
|
||||
'thumb',
|
||||
thumbnail_bytes,
|
||||
'image/jpeg',
|
||||
),
|
||||
save=True,
|
||||
)
|
||||
|
||||
thumbnail_bytes = None
|
||||
thumb_path = e_path
|
||||
if chosen_filename == e_path.name:
|
||||
break
|
||||
except:
|
||||
temp_dir = (next(iter(paths))).parent
|
||||
rmtree(temp_dir, True)
|
||||
raise
|
||||
|
||||
copy_thumbnail(self)
|
||||
|
||||
if thumb_path is None:
|
||||
return
|
||||
|
||||
return thumb_path
|
||||
|
||||
|
||||
def failed_format(self, format_str, /, *, cause=None, exc=None):
|
||||
if not self.has_metadata:
|
||||
return
|
||||
@@ -192,10 +506,10 @@ def refresh_formats(self):
|
||||
|
||||
# select and save our best thumbnail url
|
||||
try:
|
||||
thumbnail = [ thumb.get('url') for thumb in multi_key_sort(
|
||||
thumbnail = next(thumb.get('url') for thumb in multi_key_sort(
|
||||
thumbnails,
|
||||
[('preference', True,)],
|
||||
) if thumb.get('url', '').endswith('.jpg') ][0]
|
||||
) if thumb.get('url', '').endswith('.jpg'))
|
||||
except IndexError:
|
||||
pass
|
||||
else:
|
||||
@@ -232,7 +546,7 @@ def wait_for_premiere(self):
|
||||
else:
|
||||
in_hours = hours(self.published - now)
|
||||
self.manual_skip = True
|
||||
self.title = _(f'Premieres in {in_hours} hours')
|
||||
self.title = _('Premieres in {:d} hours').format(in_hours)
|
||||
|
||||
return (True, in_hours,)
|
||||
|
||||
|
||||
@@ -649,19 +649,6 @@ def index_source(source_id):
|
||||
else:
|
||||
# log the new media instances
|
||||
log.info(f'Indexed new media: {source} / {media}')
|
||||
log.info(f'Scheduling tasks to download thumbnail for: {media.key}')
|
||||
thumbnail_fmt = 'https://i.ytimg.com/vi/{}/{}default.jpg'
|
||||
for num, prefix in enumerate(reversed(('hq', 'sd', 'maxres',))):
|
||||
thumbnail_url = thumbnail_fmt.format(
|
||||
media.key,
|
||||
prefix,
|
||||
)
|
||||
download_media_image(
|
||||
str(media.pk),
|
||||
thumbnail_url,
|
||||
priority=10+(5*num),
|
||||
delay=max(0, 65-(30*num)),
|
||||
)
|
||||
priority = download_media_metadata.settings.get('default_priority', 50)
|
||||
if source.download_media:
|
||||
priority += 5
|
||||
@@ -676,6 +663,13 @@ def index_source(source_id):
|
||||
vn_fmt=_('Downloading metadata for: "{}": {}'),
|
||||
vn_args=(media.key, media.name,),
|
||||
)
|
||||
TaskHistory.schedule(
|
||||
download_media_thumbnails,
|
||||
str(media.pk),
|
||||
remove_duplicates=True,
|
||||
vn_fmt=_('Downloading thumbnails for: "{}": {}'),
|
||||
vn_args=(media.key, media.name,),
|
||||
)
|
||||
# Reset task.verbose_name to the saved value
|
||||
update_task_status(task, None)
|
||||
# Update any remaining items in the batches
|
||||
@@ -979,7 +973,7 @@ def download_media_metadata(media_id):
|
||||
metadata_lock.acquired = False
|
||||
|
||||
|
||||
@db_task(delay=10, priority=90, retries=15, backoff_class=DjangoBackgroundTasksBackoff, task_base=AttemptsTask, queue=Val(TaskQueue.NET))
|
||||
@db_task(delay=10, priority=80, retries=15, backoff_class=DjangoBackgroundTasksBackoff, task_base=AttemptsTask, queue=Val(TaskQueue.NET))
|
||||
def download_media_image(media_id, url):
|
||||
'''
|
||||
Downloads an image from a URL and save it as a local thumbnail attached to a
|
||||
@@ -1016,6 +1010,8 @@ def download_media_image(media_id, url):
|
||||
image_file.seek(0)
|
||||
thumbnail_bytes = image_file.read()
|
||||
i = image_file = None
|
||||
if media.thumb_file_exists:
|
||||
media.thumb.delete(save=False)
|
||||
retry_django_db(3)(media.thumb.save)(
|
||||
'thumb',
|
||||
SimpleUploadedFile(
|
||||
@@ -1043,6 +1039,7 @@ def on_complete_download_media_image(signal_name, task_obj, exception_obj=None,
|
||||
if result is False or result is True:
|
||||
huey.result(preserve=False, id=task_obj.id)
|
||||
|
||||
|
||||
@db_task(delay=60, priority=70, timeout=max(0, settings.MAX_RUN_TIME-600), context=True, queue=Val(TaskQueue.LIMIT))
|
||||
def download_media_file(media_id, override=False, *, task=None):
|
||||
'''
|
||||
@@ -1150,6 +1147,28 @@ def download_media_file(media_id, override=False, *, task=None):
|
||||
)
|
||||
|
||||
|
||||
@db_task(priority=90, retries=5, backoff_class=DjangoBackgroundTasksBackoff, task_base=AttemptsTask, queue=Val(TaskQueue.NET))
|
||||
def download_media_thumbnails(media_id):
|
||||
try:
|
||||
media = Media.objects.get(pk=media_id)
|
||||
except Media.DoesNotExist as e:
|
||||
# Task triggered but the media no longer exists, do nothing
|
||||
raise CancelExecution(_('no such media'), retry=False) from e
|
||||
if media.thumb_file_exists:
|
||||
raise CancelExecution(_('thumbnail exists already'), retry=False)
|
||||
selected_thumbnail = media.download_thumbnails()
|
||||
if selected_thumbnail is not None:
|
||||
selected_thumbnail = Path(selected_thumbnail)
|
||||
log.info(f'Selected thumbnail file {selected_thumbnail.name} for: {media.key}')
|
||||
try:
|
||||
temp_dir = selected_thumbnail.resolve(strict=True).parent
|
||||
except FileNotFoundError:
|
||||
pass
|
||||
else:
|
||||
if '-thumbnails-' in temp_dir.name:
|
||||
rmtree(temp_dir, True)
|
||||
|
||||
|
||||
@db_task(delay=30, expires=210, priority=100, queue=Val(TaskQueue.NET))
|
||||
def rescan_media_server(mediaserver_id):
|
||||
'''
|
||||
|
||||
Reference in New Issue
Block a user