Merge pull request #1602 from meeb/tcely-curl-thumbnails

fix: download thumbnails with curl
This commit is contained in:
tcely
2026-09-24 11:37:21 -04:00
committed by GitHub
3 changed files with 367 additions and 32 deletions

View File

@@ -43,7 +43,8 @@ from ._migrations import (
from ._private import _srctype_dict, _nfo_element
from .media__tasks import (
copy_thumbnail, download_checklist, download_finished,
failed_format, refresh_formats, wait_for_premiere, write_nfo_file,
download_thumbnails, failed_format, refresh_formats,
wait_for_premiere, write_nfo_file,
)
from .source import Source
@@ -1238,6 +1239,7 @@ class Media(models.Model):
Media.copy_thumbnail = copy_thumbnail
Media.download_checklist = download_checklist
Media.download_finished = download_finished
Media.download_thumbnails = download_thumbnails
Media.failed_format = failed_format
Media.refresh_formats = refresh_formats
Media.wait_for_premiere = wait_for_premiere

View File

@@ -1,17 +1,29 @@
import io
import os
import subprocess
from collections import defaultdict
from pathlib import Path
from shutil import copyfile
from common.logger import log
from pathlib import Path, PurePosixPath
from shutil import copyfile, rmtree
from tempfile import TemporaryDirectory
from urllib.parse import urlparse, urlunparse
from PIL import Image
from django.conf import settings
from django.core.files.uploadedfile import SimpleUploadedFile
from django.utils import timezone
from django.utils.translation import gettext_lazy as _
from common.errors import (
NoMetadataException,
)
from common.utils import multi_key_sort
from django.conf import settings
from django.utils import timezone
from django.utils.translation import gettext_lazy as _
from common.logger import log
from common.utils import getenv, multi_key_sort
from common.yt_dlp import retry_django_db
from ..choices import Val, SourceResolution
from ..utils import filter_response, write_text_file
from ..utils import (
filter_response, resize_image_to_height, write_text_file
)
def copy_thumbnail(self):
@@ -69,12 +81,11 @@ def download_checklist(self, skip_checks=False):
return False
max_cap_age = media.source.download_cap_date
published = media.published
if max_cap_age and published:
if published <= max_cap_age:
log.warn(f'Download task triggered media: {media} (UUID: {media.pk}) but '
f'the source has a download cap and the media is now too old, '
f'not downloading')
return False
if max_cap_age and published and published <= max_cap_age:
log.warn(f'Download task triggered media: {media} (UUID: {media.pk}) but '
f'the source has a download cap and the media is now too old, '
f'not downloading')
return False
return True
@@ -128,6 +139,309 @@ def download_finished(self, format_str, container, downloaded_filepath=None):
self.downloaded_hdr = False
def make_youtube_thumbnail_urls(video_id: str, output_format: str = 'string') -> str | tuple[dict[str, str], ...]:
"""
Generates YouTube thumbnail URLs using urlunparse.
Defaults to raw 'string' output, but can be extended.
"""
# Format serialization functions mapped inside the registry
def _format_as_string(urls, headers, rows) -> str:
return '\n'.join(urls)
def _format_as_dicts(urls, headers, rows) -> tuple[dict[str, str], ...]:
return tuple({'url': urls[i], 'filename': rows[i][3]} for i in range(len(urls)))
FORMATTER_MAP = {
'string': _format_as_string,
'dicts': _format_as_dicts,
}
scheme = 'https'
hostname = 'i.ytimg.com'
base_names = (
'maxresdefault', 'sddefault', 'hqdefault', '1', '2', '3',
'oardefault', 'oar1', 'oar2', 'oar3',
)
urls = []
for name in base_names:
# Construct full clean URLs directly using urlunparse tuples
# Tuple format: (scheme, netloc, path, params, query, fragment)
jpg = urlunparse((scheme, hostname, f'/vi/{video_id}/{name}.jpg', '', '', ''))
webp = urlunparse((scheme, hostname, f'/vi_webp/{video_id}/{name}.webp', '', '', ''))
urls.extend((jpg, webp))
headers = ('Scheme', 'Hostname', 'Path', 'Filename')
rows = []
for url in urls:
parsed = urlparse(url)
rows.append((
parsed.scheme,
parsed.hostname,
parsed.path,
PurePosixPath(parsed.path).name,
))
fmt = output_format.lower().strip()
if fmt not in FORMATTER_MAP:
raise ValueError(f'Unsupported format "{output_format}".')
return FORMATTER_MAP[fmt](urls, headers, rows)
def download_thumbnails_pycurl(self) -> Path | None:
import pycurl
def download_thumbnails_parallel(video_id: str, max_connections: int = 2) -> dict[str, io.BytesIO]:
"""
Executes high-performance parallel downloads via pycurl completely in memory.
Returns a dictionary mapping filenames to populated BytesIO buffers.
"""
url_targets = make_youtube_thumbnail_urls(video_id=video_id, output_format='dicts')
downloaded_buffers = {}
# This is a quick and dirty attempt to support proxies.
env_proxy = getenv('https_proxy') or getenv('http_proxy') or getenv('all_proxy')
no_proxy_setting = getenv('no_proxy')
with pycurl.CurlMulti() as multi:
multi.setopt(pycurl.M_PIPELINING, pycurl.PIPE_NOTHING)
multi.setopt(pycurl.M_MAX_HOST_CONNECTIONS, max_connections)
connections = {}
for target in url_targets:
c = pycurl.Curl()
c.setopt(c.URL, target['url'])
c.setopt(c.FOLLOWLOCATION, True)
c.setopt(c.FAILONERROR, True)
buffer = io.BytesIO()
c.setopt(c.WRITEDATA, buffer)
if env_proxy:
c.setopt(pycurl.PROXY, env_proxy)
if no_proxy_setting:
c.setopt(pycurl.NOPROXY, no_proxy_setting)
multi.add_handle(c)
connections[c] = {'filename': target['filename'], 'buffer': buffer}
c = buffer = None
num_handles = 1
while 0 < num_handles:
ret, num_handles = multi.perform()
if ret != pycurl.E_CALL_MULTI_PERFORM:
multi.select(1.0)
num_q = 1
while 0 < num_q:
# err_list never used
# ruff: ignore[RUF059]
num_q, ok_list, err_list = multi.info_read()
for curl in ok_list:
info = connections[curl]
buf = info['buffer']
buf.seek(0, io.SEEK_END)
if 0 < buf.tell():
buf.seek(0, io.SEEK_SET)
downloaded_buffers[info['filename']] = buf
for curl in connections:
curl.close()
log.debug(f'Parallel download pass completed. Successfully stored {len(downloaded_buffers)} valid buffers for: {video_id}')
return downloaded_buffers
downloaded_data = download_thumbnails_parallel(self.key)
if not downloaded_data:
return
chosen_filename = PurePosixPath(self.thumbnail).name
width = getattr(settings, 'MEDIA_THUMBNAIL_WIDTH', 430)
height = getattr(settings, 'MEDIA_THUMBNAIL_HEIGHT', 240)
saved_size = (0, 0)
thumb_path = None
for filename, buffer in downloaded_data.items():
filename_path = Path(filename)
# accept: maxres webp, the filename from self.thumbnail, or any jpg thumbnails
if not (
chosen_filename == filename_path.name or
'.jpg' == filename_path.suffix or
'maxresdefault' == filename_path.stem
):
continue
image_file = io.BytesIO()
with Image.open(buffer) as img:
if img.size < saved_size:
continue
saved_size = img.size
if 'RGB' != img.mode:
img = img.convert('RGB')
if (img.width > width) and (img.height > height):
log.debug(f'Resizing {img.width}x{img.height} thumbnail to '
f'{width}x{height}: {filename_path.name}')
img = resize_image_to_height(img, width, height)
img.save(image_file, 'JPEG', quality=85, optimize=True, progressive=True)
img = None
image_file.seek(0, io.SEEK_SET)
thumbnail_bytes = image_file.read()
image_file = None
if self.thumb_file_exists:
self.thumb.delete(save=False)
retry_django_db(5)(self.thumb.save)(
'thumb',
SimpleUploadedFile(
'thumb',
thumbnail_bytes,
'image/jpeg',
),
save=True,
)
thumbnail_bytes = None
thumb_path = filename_path
if chosen_filename == filename_path.name:
break
copy_thumbnail(self)
if thumb_path is None:
return
return thumb_path
def download_thumbnails(self) -> Path | None:
def export_urls_to_temp_file(video_id: str) -> str:
"""Writes URLs.txt using the 'string' format."""
prefix = f'i.ytimg.com-thumbnails-[{video_id}]-'
with TemporaryDirectory(prefix=prefix, delete=False) as temp_dir:
file_path = Path(temp_dir) / 'URLs.txt'
with open(file_path, 'w') as f:
f.write(make_youtube_thumbnail_urls(video_id=video_id, output_format='string'))
f.write('\n')
return file_path
def download_thumbnails_parallel(video_id: str, max_connections: int = 2) -> tuple[Path,...]:
"""
Executes high-performance parallel downloads via curl.
Attempts all URLs, but strictly skips writing files for any 404 responses.
"""
file_path = Path(export_urls_to_temp_file(video_id))
curl_command = (
'curl',
'--parallel', '--parallel-immediate',
# Debian 13 curl is too old for this option
# '--parallel-max-host', str(max_connections),
'--parallel-max', str(max_connections),
'--remote-time', '--remote-name-all',
'--fail', '--verbose', '--show-error',
'--dump-header', 'curl.header.log.txt',
'--stderr', 'curl.stderr.log.txt',
'--url', f'@{file_path.name}',
)
try:
subprocess.run(
curl_command,
cwd=str(file_path.parent),
check=False,
capture_output=True,
text=True,
)
downloaded_files = tuple(
e_path for e in os.scandir(file_path.parent)
if (e_path := Path(e.path)).suffix in ('.jpg', '.webp') and e.is_file() and 0 < e.stat().st_size
)
log.debug(f'Parallel download pass completed. Successfully stored {len(downloaded_files)} valid files for: {video_id}')
return downloaded_files
except FileNotFoundError:
raise RuntimeError('Missing dependencies: "curl" executable was not found on your system environment PATH.')
paths = download_thumbnails_parallel(self.key)
if not paths:
return
chosen_filename = PurePosixPath(self.thumbnail).name
width = getattr(settings, 'MEDIA_THUMBNAIL_WIDTH', 430)
height = getattr(settings, 'MEDIA_THUMBNAIL_HEIGHT', 240)
saved_size = (0, 0)
thumb_path = None
try:
for e_path in paths:
# accept: maxres webp, the filename from self.thumbnail, or any jpg thumbnails
if not (
chosen_filename == e_path.name or
'.jpg' == e_path.suffix or
'maxresdefault' == e_path.stem
):
continue
image_file = io.BytesIO()
with Image.open(e_path) as img:
if img.size < saved_size:
continue
saved_size = img.size
if 'RGB' != img.mode:
img = img.convert('RGB')
if (img.width > width) and (img.height > height):
log.debug(f'Resizing {img.width}x{img.height} thumbnail to '
f'{width}x{height}: {e_path.name}')
img = resize_image_to_height(img, width, height)
img.save(image_file, 'JPEG', quality=85, optimize=True, progressive=True)
img = None
image_file.seek(0, io.SEEK_SET)
thumbnail_bytes = image_file.read()
image_file = None
if self.thumb_file_exists:
self.thumb.delete(save=False)
retry_django_db(5)(self.thumb.save)(
'thumb',
SimpleUploadedFile(
'thumb',
thumbnail_bytes,
'image/jpeg',
),
save=True,
)
thumbnail_bytes = None
thumb_path = e_path
if chosen_filename == e_path.name:
break
except:
temp_dir = (next(iter(paths))).parent
rmtree(temp_dir, True)
raise
copy_thumbnail(self)
if thumb_path is None:
return
return thumb_path
def failed_format(self, format_str, /, *, cause=None, exc=None):
if not self.has_metadata:
return
@@ -192,10 +506,10 @@ def refresh_formats(self):
# select and save our best thumbnail url
try:
thumbnail = [ thumb.get('url') for thumb in multi_key_sort(
thumbnail = next(thumb.get('url') for thumb in multi_key_sort(
thumbnails,
[('preference', True,)],
) if thumb.get('url', '').endswith('.jpg') ][0]
) if thumb.get('url', '').endswith('.jpg'))
except IndexError:
pass
else:
@@ -232,7 +546,7 @@ def wait_for_premiere(self):
else:
in_hours = hours(self.published - now)
self.manual_skip = True
self.title = _(f'Premieres in {in_hours} hours')
self.title = _('Premieres in {:d} hours').format(in_hours)
return (True, in_hours,)

View File

@@ -649,19 +649,6 @@ def index_source(source_id):
else:
# log the new media instances
log.info(f'Indexed new media: {source} / {media}')
log.info(f'Scheduling tasks to download thumbnail for: {media.key}')
thumbnail_fmt = 'https://i.ytimg.com/vi/{}/{}default.jpg'
for num, prefix in enumerate(reversed(('hq', 'sd', 'maxres',))):
thumbnail_url = thumbnail_fmt.format(
media.key,
prefix,
)
download_media_image(
str(media.pk),
thumbnail_url,
priority=10+(5*num),
delay=max(0, 65-(30*num)),
)
priority = download_media_metadata.settings.get('default_priority', 50)
if source.download_media:
priority += 5
@@ -676,6 +663,13 @@ def index_source(source_id):
vn_fmt=_('Downloading metadata for: "{}": {}'),
vn_args=(media.key, media.name,),
)
TaskHistory.schedule(
download_media_thumbnails,
str(media.pk),
remove_duplicates=True,
vn_fmt=_('Downloading thumbnails for: "{}": {}'),
vn_args=(media.key, media.name,),
)
# Reset task.verbose_name to the saved value
update_task_status(task, None)
# Update any remaining items in the batches
@@ -979,7 +973,7 @@ def download_media_metadata(media_id):
metadata_lock.acquired = False
@db_task(delay=10, priority=90, retries=15, backoff_class=DjangoBackgroundTasksBackoff, task_base=AttemptsTask, queue=Val(TaskQueue.NET))
@db_task(delay=10, priority=80, retries=15, backoff_class=DjangoBackgroundTasksBackoff, task_base=AttemptsTask, queue=Val(TaskQueue.NET))
def download_media_image(media_id, url):
'''
Downloads an image from a URL and save it as a local thumbnail attached to a
@@ -1016,6 +1010,8 @@ def download_media_image(media_id, url):
image_file.seek(0)
thumbnail_bytes = image_file.read()
i = image_file = None
if media.thumb_file_exists:
media.thumb.delete(save=False)
retry_django_db(3)(media.thumb.save)(
'thumb',
SimpleUploadedFile(
@@ -1043,6 +1039,7 @@ def on_complete_download_media_image(signal_name, task_obj, exception_obj=None,
if result is False or result is True:
huey.result(preserve=False, id=task_obj.id)
@db_task(delay=60, priority=70, timeout=max(0, settings.MAX_RUN_TIME-600), context=True, queue=Val(TaskQueue.LIMIT))
def download_media_file(media_id, override=False, *, task=None):
'''
@@ -1150,6 +1147,28 @@ def download_media_file(media_id, override=False, *, task=None):
)
@db_task(priority=90, retries=5, backoff_class=DjangoBackgroundTasksBackoff, task_base=AttemptsTask, queue=Val(TaskQueue.NET))
def download_media_thumbnails(media_id):
try:
media = Media.objects.get(pk=media_id)
except Media.DoesNotExist as e:
# Task triggered but the media no longer exists, do nothing
raise CancelExecution(_('no such media'), retry=False) from e
if media.thumb_file_exists:
raise CancelExecution(_('thumbnail exists already'), retry=False)
selected_thumbnail = media.download_thumbnails()
if selected_thumbnail is not None:
selected_thumbnail = Path(selected_thumbnail)
log.info(f'Selected thumbnail file {selected_thumbnail.name} for: {media.key}')
try:
temp_dir = selected_thumbnail.resolve(strict=True).parent
except FileNotFoundError:
pass
else:
if '-thumbnails-' in temp_dir.name:
rmtree(temp_dir, True)
@db_task(delay=30, expires=210, priority=100, queue=Val(TaskQueue.NET))
def rescan_media_server(mediaserver_id):
'''