Compare commits

...

9 Commits

Author SHA1 Message Date
Alex Shnitman 50250f8374 Merge PR #1037: bump actions/setup-node from 6 to 7
Dependabot github-actions group update. Single-line bump in main.yml,
consistent with the existing actions/checkout@v7.

Co-authored-by: dependabot[bot] <support@github.com>
2026-07-24 12:09:53 +03:00
Alex Shnitman 926d392926 Merge PR #1038: surface add-time failures as failed done entries
Adds __record_add_failure so a URL that fails before a real download starts
(unsupported/unextractable URL, SSRF-rejected, extraction error) appears in the
Completed list as a red-cross entry with retry and error-detail, instead of only
a transient toast and a server log line. Keyed by info.url like any errored
download. Includes _short_title_for_failed_url for a readable hostname title.

The frontend hunk removing the 'Click for details' hint from every error row was
dropped from this merge; that affordance (added in fd3aaea, #143) is kept.

Co-authored-by: streamer1122 <streamer1122@users.noreply.github.com>
2026-07-24 12:04:08 +03:00
Alex Shnitman 7f13784445 Merge PR #1031: conservative music metadata enrichment for audio downloads
Adds MusicMetadataPreProcessor (app/music_metadata.py), a pre_process
postprocessor that enriches the info dict using only extractor-owned fields:
track-number/total resolution, album-title fallback, and square-thumbnail
preference. Track numbers and album flow into files through yt-dlp's existing
FFmpegMetadata embedding, so no per-format tag-writing code and no new
dependency. The earlier mutagen writer and its download-failing
PostProcessingError path were dropped per review.

Co-authored-by: jahruz67 <jahruz67@users.noreply.github.com>
2026-07-24 12:00:38 +03:00
Alex Shnitman 4aa20890d4 Merge PR #1035: release per-download status_queue proxy on close to fix fd leak (#485)
Wraps Download.close() in try/finally and nulls self.status_queue so the
per-download manager.Queue() proxy is released once the completed Download is
retained in the done list. Previously every finished download permanently
pinned one Manager-process connection, accumulating file descriptors until
the instance hit 'too many open files' and self-terminated (#485, #980).

Co-authored-by: Matt Van Horn <mvanhorn@users.noreply.github.com>
2026-07-24 11:44:47 +03:00
James Tew 4e27600329 Added handling for unsupported URL 2026-07-20 22:20:18 +01:00
dependabot[bot] 1c7261ab59 build(deps): bump actions/setup-node in the github-actions group
Bumps the github-actions group with 1 update: [actions/setup-node](https://github.com/actions/setup-node).


Updates `actions/setup-node` from 6 to 7
- [Release notes](https://github.com/actions/setup-node/releases)
- [Commits](https://github.com/actions/setup-node/compare/v6...v7)

---
updated-dependencies:
- dependency-name: actions/setup-node
  dependency-version: '7'
  dependency-type: direct:production
  update-type: version-update:semver-major
  dependency-group: github-actions
...

Signed-off-by: dependabot[bot] <support@github.com>
2026-07-19 16:12:37 +00:00
Matt Van Horn 4cf2b1b0d0 fix: release per-download status_queue proxy on close to stop FD leak 2026-07-19 01:13:16 -07:00
Your GitHub Name f3d670e288 refactor: simplify music metadata processing by removing unused code and improving album signal detection 2026-07-17 10:43:39 -07:00
Your GitHub Name edf101faa0 feat: add music metadata processing and writing functionality 2026-07-16 19:31:57 -07:00
6 changed files with 445 additions and 6 deletions
+1 -1
View File
@@ -14,7 +14,7 @@ jobs:
- name: Checkout - name: Checkout
uses: actions/checkout@v7 uses: actions/checkout@v7
- name: Set up Node.js - name: Set up Node.js
uses: actions/setup-node@v6 uses: actions/setup-node@v7
with: with:
node-version: lts/* node-version: lts/*
- name: Enable pnpm - name: Enable pnpm
+124
View File
@@ -0,0 +1,124 @@
"""Conservative music metadata enrichment for audio downloads.
This module only consumes fields already supplied by yt-dlp or retained on
MeTube's queued playlist entry. It intentionally performs no external lookup
or site-specific album detection.
"""
from __future__ import annotations
from typing import Any, Optional
from yt_dlp.postprocessor.common import PostProcessor
def _has_value(value: Any) -> bool:
if isinstance(value, str):
return bool(value.strip())
if isinstance(value, (list, tuple)):
return any(_has_value(item) for item in value)
return value is not None
def _positive_int(value: Any) -> Optional[int]:
if isinstance(value, bool):
return None
try:
number = int(value)
except (TypeError, ValueError):
return None
return number if number > 0 else None
def _track_position(value: Any) -> tuple[Optional[int], Optional[int]]:
"""Return a track number and optional total from a scalar or ``n/total``."""
if isinstance(value, str) and '/' in value:
number, total = value.split('/', 1)
return _positive_int(number.strip()), _positive_int(total.strip())
return _positive_int(value), None
def _first_positive_int(*values: Any) -> Optional[int]:
return next((number for value in values if (number := _positive_int(value))), None)
def _has_album_signal(info: dict[str, Any], source_entry: dict[str, Any]) -> bool:
"""Use only extractor-owned fields to identify album-level metadata."""
return any(
_has_value(entry.get(key))
for entry in (info, source_entry)
for key in ('album', 'track_number')
)
def _is_music_audio(info: dict[str, Any], source_entry: dict[str, Any]) -> bool:
return _has_album_signal(info, source_entry) or any(
_has_value(entry.get(key))
for entry in (info, source_entry)
for key in ('track', 'artists')
)
def prefer_square_thumbnail(info: dict[str, Any]) -> None:
"""Move the largest known square thumbnail to yt-dlp's preferred slot."""
thumbnails = info.get('thumbnails')
if not isinstance(thumbnails, list) or len(thumbnails) < 2:
return
candidates: list[tuple[int, int]] = []
for index, thumbnail in enumerate(thumbnails):
if not isinstance(thumbnail, dict):
continue
width = _positive_int(thumbnail.get('width'))
height = _positive_int(thumbnail.get('height'))
if width is not None and width == height:
candidates.append((width * height, index))
if not candidates:
return
_, selected_index = max(candidates)
selected = thumbnails.pop(selected_index)
thumbnails.append(selected)
if selected.get('url'):
info['thumbnail'] = selected['url']
class MusicMetadataPreProcessor(PostProcessor):
"""Enrich extracted audio metadata using extractor-owned album signals."""
def __init__(self, downloader=None, *, source_entry=None):
super().__init__(downloader)
self._source_entry = source_entry if isinstance(source_entry, dict) else {}
def run(self, info):
if _has_album_signal(info, self._source_entry):
number, inline_total = _track_position(info.get('track_number'))
if number is None:
number, source_inline_total = _track_position(
self._source_entry.get('track_number')
)
inline_total = inline_total or source_inline_total
if number is None:
number = _positive_int(self._source_entry.get('playlist_index'))
total = inline_total or _first_positive_int(
info.get('track_count'),
info.get('track_total'),
self._source_entry.get('track_count'),
self._source_entry.get('track_total'),
self._source_entry.get('playlist_count'),
self._source_entry.get('n_entries'),
)
if number is not None:
info['track_number'] = f'{number}/{total}' if total is not None else number
if not _has_value(info.get('album')):
album = self._source_entry.get('album') or self._source_entry.get(
'playlist_title'
)
if isinstance(album, str) and album.strip():
info['album'] = album.strip()
if _is_music_audio(info, self._source_entry):
prefer_square_thumbnail(info)
return [], info
+82
View File
@@ -115,6 +115,54 @@ async def test_add_single_video_goes_to_pending_when_auto_start_false(dq_env):
assert dq.pending.exists("https://example.com/watch?v=1") assert dq.pending.exists("https://example.com/watch?v=1")
@pytest.mark.asyncio
async def test_add_unsupported_url_recorded_as_failed_entry(dq_env):
"""An unsupported/unextractable URL must show up as a red-cross entry in the
done list, not just a transient toast and a server log line."""
import ytdl
notifier = AsyncMock()
url = "https://example.com/not-a-video"
def boom(self, url, ytdl_options_presets=None, ytdl_options_overrides=None):
raise ytdl.yt_dlp.utils.YoutubeDLError(f'Unsupported URL: {url}')
dq = DownloadQueue(dq_env, notifier)
with patch.object(DownloadQueue, "_DownloadQueue__extract_info", boom):
result = await dq.add(
url, "video", "auto", "any", "best", "", "", 0, auto_start=True,
)
assert result["status"] == "error"
assert dq.done.exists(url)
failed = dq.done.get(url)
assert failed.info.status == "error"
assert failed.info.error == result["msg"]
assert failed.info.url == url
# The full URL stays in .url/.error for the detail panel; the display
# title is shortened to the hostname so the Completed row stays readable.
assert failed.info.title == "example.com"
notifier.completed.assert_awaited()
@pytest.mark.asyncio
async def test_add_ssrf_rejected_url_recorded_as_failed_entry(dq_env):
"""A URL rejected by the SSRF guard (before yt-dlp ever runs) must also
surface as a failed entry, not just an error status returned to the caller."""
notifier = AsyncMock()
url = "file:///etc/passwd"
dq = DownloadQueue(dq_env, notifier)
result = await dq.add(
url, "video", "auto", "any", "best", "", "", 0, auto_start=True,
)
assert result["status"] == "error"
assert dq.done.exists(url)
failed = dq.done.get(url)
assert failed.info.status == "error"
assert failed.info.error == result["msg"]
notifier.completed.assert_awaited()
@pytest.mark.asyncio @pytest.mark.asyncio
async def test_cancel_removes_from_pending(dq_env): async def test_cancel_removes_from_pending(dq_env):
notifier = AsyncMock() notifier = AsyncMock()
@@ -898,6 +946,40 @@ def _make_download(dq_env, *, download_type="video", status="downloading", filen
) )
def test_download_close_releases_status_queue(dq_env):
download = _make_download(dq_env)
status_queue = MagicMock()
proc = MagicMock()
download.status_queue = status_queue
download.proc = proc
download.close()
proc.close.assert_called_once()
assert download.status_queue is None
def test_download_close_releases_status_queue_without_process(dq_env):
download = _make_download(dq_env)
download.status_queue = MagicMock()
download.close()
assert download.status_queue is None
def test_download_close_releases_status_queue_when_process_close_fails(dq_env):
download = _make_download(dq_env)
download.status_queue = MagicMock()
download.proc = MagicMock()
download.proc.close.side_effect = RuntimeError('close failed')
with pytest.raises(RuntimeError, match='close failed'):
download.close()
assert download.status_queue is None
@pytest.mark.asyncio @pytest.mark.asyncio
async def test_post_download_cleanup_clears_filename_on_error(dq_env): async def test_post_download_cleanup_clears_filename_on_error(dq_env):
notifier = AsyncMock() notifier = AsyncMock()
+119
View File
@@ -0,0 +1,119 @@
"""Tests for conservative audio metadata enrichment."""
from __future__ import annotations
from music_metadata import MusicMetadataPreProcessor
def _preprocess(source_entry, info):
processor = MusicMetadataPreProcessor(source_entry=source_entry)
_, result = processor.run(info)
return result
def test_album_uses_existing_order_and_total_when_track_number_is_missing():
result = _preprocess(
{
'playlist_index': '03',
'playlist_count': 12,
'playlist_title': 'Example Album',
},
{'title': 'Track', 'album': 'Example Album'},
)
assert result['track_number'] == '3/12'
assert result['album'] == 'Example Album'
def test_official_track_number_wins_over_album_order():
result = _preprocess(
{'playlist_index': 3, 'playlist_count': 12},
{'track_number': 7, 'album': 'Official Album'},
)
assert result['track_number'] == '7/12'
assert result['album'] == 'Official Album'
def test_inline_official_track_total_is_preserved():
result = _preprocess(
{'playlist_count': 12},
{'track_number': '4/10', 'album': 'Official Album'},
)
assert result['track_number'] == '4/10'
def test_source_track_number_and_total_are_retained_from_flat_extraction():
result = _preprocess(
{'track_number': 2, 'track_count': 9, 'playlist_index': 4},
{'title': 'Track'},
)
assert result['track_number'] == '2/9'
def test_album_title_falls_back_to_source_playlist_title():
result = _preprocess(
{'playlist_title': 'Example Album'},
{'track_number': 4},
)
assert result['album'] == 'Example Album'
assert result['track_number'] == 4
def test_playlist_without_extractor_album_signals_is_not_changed():
result = _preprocess(
{
'playlist_index': 3,
'playlist_count': 12,
'playlist_title': 'Example Playlist',
},
{'title': 'Track'},
)
assert 'album' not in result
assert 'track_number' not in result
def test_regular_video_artwork_is_not_changed():
thumbnails = [
{'url': 'square.jpg', 'width': 500, 'height': 500},
{'url': 'landscape.jpg', 'width': 1280, 'height': 720},
]
result = _preprocess({}, {'title': 'Regular Video', 'thumbnails': thumbnails.copy()})
assert result['thumbnails'] == thumbnails
assert 'thumbnail' not in result
def test_music_audio_prefers_largest_existing_square_thumbnail():
result = _preprocess(
{},
{
'track': 'Track',
'thumbnails': [
{'url': 'small-square.jpg', 'width': 200, 'height': 200},
{'url': 'large-square.jpg', 'width': 1000, 'height': 1000},
{'url': 'landscape.jpg', 'width': 1280, 'height': 720},
],
},
)
assert result['thumbnails'][-1]['url'] == 'large-square.jpg'
assert result['thumbnail'] == 'large-square.jpg'
def test_landscape_only_music_artwork_keeps_existing_order():
thumbnails = [
{'url': 'small.jpg', 'width': 640, 'height': 360},
{'url': 'large.jpg', 'width': 1280, 'height': 720},
]
result = _preprocess(
{},
{'track': 'Track', 'thumbnails': thumbnails.copy()},
)
assert result['thumbnails'] == thumbnails
assert 'thumbnail' not in result
+28 -2
View File
@@ -73,12 +73,14 @@ import ytdl
from ytdl import ( from ytdl import (
Download, Download,
DownloadInfo, DownloadInfo,
MusicMetadataPreProcessor,
_compact_persisted_entry, _compact_persisted_entry,
_convert_srt_to_txt_file, _convert_srt_to_txt_file,
_AlbumArtistPostProcessor, _AlbumArtistPostProcessor,
_resolve_outtmpl_fields, _resolve_outtmpl_fields,
_sanitize_entry_for_pickle, _sanitize_entry_for_pickle,
_sanitize_path_component, _sanitize_path_component,
_short_title_for_failed_url,
) )
# Detect whether the real yt-dlp is loaded (as opposed to the minimal fake # Detect whether the real yt-dlp is loaded (as opposed to the minimal fake
@@ -201,9 +203,15 @@ class AlbumArtistRegistrationTests(unittest.TestCase):
result = download._make_youtube_dl({'quiet': True}) result = download._make_youtube_dl({'quiet': True})
self.assertIs(result, fake_ydl) self.assertIs(result, fake_ydl)
postprocessor, = fake_ydl.add_post_processor.call_args.args album_artist_call = fake_ydl.add_post_processor.call_args_list[0]
postprocessor, = album_artist_call.args
self.assertIsInstance(postprocessor, _AlbumArtistPostProcessor) self.assertIsInstance(postprocessor, _AlbumArtistPostProcessor)
self.assertEqual(fake_ydl.add_post_processor.call_args.kwargs, {'when': 'pre_process'}) self.assertEqual(album_artist_call.kwargs, {'when': 'pre_process'})
metadata_pre_call = fake_ydl.add_post_processor.call_args_list[1]
metadata_preprocessor, = metadata_pre_call.args
self.assertIsInstance(metadata_preprocessor, MusicMetadataPreProcessor)
self.assertEqual(metadata_pre_call.kwargs, {'when': 'pre_process'})
self.assertEqual(fake_ydl.add_post_processor.call_count, 2)
def test_video_download_does_not_register_postprocessor(self): def test_video_download_does_not_register_postprocessor(self):
download = _make_test_download() download = _make_test_download()
@@ -801,5 +809,23 @@ class CompactPersistedEntryTests(unittest.TestCase):
self.assertIsNone(_compact_persisted_entry({"id": "x", "title": "y"})) self.assertIsNone(_compact_persisted_entry({"id": "x", "title": "y"}))
class ShortTitleForFailedUrlTests(unittest.TestCase):
def test_uses_hostname_for_a_normal_url(self):
self.assertEqual(
_short_title_for_failed_url("https://example.com/watch?v=1"),
"example.com",
)
def test_falls_back_to_raw_value_when_there_is_no_hostname(self):
# file:// URIs and bare search terms/video IDs have no netloc to extract.
self.assertEqual(_short_title_for_failed_url("file:///etc/passwd"), "file:///etc/passwd")
self.assertEqual(_short_title_for_failed_url("ytsearch:some query"), "ytsearch:some query")
def test_falls_back_to_raw_value_on_unparseable_input(self):
# A malformed IPv6-looking host raises ValueError in urlsplit().hostname.
malformed = "https://[::1/watch"
self.assertEqual(_short_title_for_failed_url(malformed), malformed)
if __name__ == "__main__": if __name__ == "__main__":
unittest.main() unittest.main()
+91 -3
View File
@@ -23,10 +23,12 @@ from yt_dlp.postprocessor.common import PostProcessor
from yt_dlp.utils import STR_FORMAT_RE_TMPL, STR_FORMAT_TYPES from yt_dlp.utils import STR_FORMAT_RE_TMPL, STR_FORMAT_TYPES
import bg_tasks import bg_tasks
from dl_formats import get_format, get_opts, AUDIO_FORMATS, merge_ytdl_option_layers from dl_formats import get_format, get_opts, AUDIO_FORMATS, merge_ytdl_option_layers
from music_metadata import MusicMetadataPreProcessor
from datetime import datetime from datetime import datetime
from state_store import AtomicJsonStore, from_json_compatible, read_legacy_shelf, to_json_compatible from state_store import AtomicJsonStore, from_json_compatible, read_legacy_shelf, to_json_compatible
from subscriptions import _entry_id from subscriptions import _entry_id
from url_guard import validate_url, install_socket_guard from url_guard import validate_url, install_socket_guard
from urllib.parse import urlsplit
log = logging.getLogger('ytdl') log = logging.getLogger('ytdl')
@@ -497,6 +499,17 @@ _PERSISTED_DOWNLOAD_FIELDS = (
) )
def _short_title_for_failed_url(url: str) -> str:
"""A concise display title for a URL that failed before yt-dlp could extract a
real title (unsupported URL, SSRF-rejected, extraction error). The full URL
remains available in DownloadInfo.url and the error-detail panel."""
try:
hostname = urlsplit(url).hostname
except ValueError:
hostname = None
return hostname or url
_COMPACT_ENTRY_EXTRA_KEYS = frozenset(("n_entries", "__last_playlist_index")) _COMPACT_ENTRY_EXTRA_KEYS = frozenset(("n_entries", "__last_playlist_index"))
@@ -625,6 +638,13 @@ class Download:
) )
if getattr(self.info, 'download_type', '') == 'audio': if getattr(self.info, 'download_type', '') == 'audio':
ydl.add_post_processor(_AlbumArtistPostProcessor(ydl), when='pre_process') ydl.add_post_processor(_AlbumArtistPostProcessor(ydl), when='pre_process')
ydl.add_post_processor(
MusicMetadataPreProcessor(
ydl,
source_entry=getattr(self.info, 'entry', None),
),
when='pre_process',
)
return ydl return ydl
def _download(self): def _download(self):
@@ -787,8 +807,11 @@ class Download:
def close(self): def close(self):
log.info(f"Closing download process for: {self.info.title}") log.info(f"Closing download process for: {self.info.title}")
if self.started(): try:
self.proc.close() if self.started():
self.proc.close()
finally:
self.status_queue = None
def running(self): def running(self):
try: try:
@@ -1517,6 +1540,58 @@ class DownloadQueue:
return {'status': 'ok'} return {'status': 'ok'}
return {'status': 'error', 'msg': f'Unsupported resource "{etype}"'} return {'status': 'error', 'msg': f'Unsupported resource "{etype}"'}
async def __record_add_failure(
self,
url,
msg,
download_type,
codec,
format,
quality,
folder,
custom_name_prefix,
playlist_item_limit,
split_by_chapters,
chapter_template,
subtitle_language,
subtitle_mode,
ytdl_options_presets,
ytdl_options_overrides,
clip_start,
clip_end,
):
"""Surface a URL that failed before a DownloadInfo could be created (unsupported
URL, SSRF-rejected, extraction error) as a failed entry in the done list, so the
frontend shows it with the same red-cross/retry/error-detail treatment as a
download that failed mid-stream, instead of only a toast and a server log line."""
info = DownloadInfo(
id=url,
title=_short_title_for_failed_url(url),
url=url,
quality=quality,
download_type=download_type,
codec=codec,
format=format,
folder=folder,
custom_name_prefix=custom_name_prefix,
error=msg,
entry=None,
playlist_item_limit=playlist_item_limit,
split_by_chapters=split_by_chapters,
chapter_template=chapter_template,
subtitle_language=subtitle_language,
subtitle_mode=subtitle_mode,
ytdl_options_presets=ytdl_options_presets,
ytdl_options_overrides=ytdl_options_overrides,
clip_start=clip_start,
clip_end=clip_end,
)
info.status = 'error'
info.msg = msg
download = Download(None, None, None, None, quality, format, {}, info)
self.done.put(download)
await self.notifier.completed(info)
async def add( async def add(
self, self,
url, url,
@@ -1562,6 +1637,12 @@ class DownloadQueue:
None, partial(validate_url, url, allow_private=self.config.ALLOW_PRIVATE_ADDRESSES)) None, partial(validate_url, url, allow_private=self.config.ALLOW_PRIVATE_ADDRESSES))
if url_error is not None: if url_error is not None:
log.warning('Rejected URL "%s": %s', url, url_error) log.warning('Rejected URL "%s": %s', url, url_error)
await self.__record_add_failure(
url, url_error, download_type, codec, format, quality, folder,
custom_name_prefix, playlist_item_limit, split_by_chapters, chapter_template,
subtitle_language, subtitle_mode, ytdl_options_presets, ytdl_options_overrides,
clip_start, clip_end,
)
return {'status': 'error', 'msg': url_error} return {'status': 'error', 'msg': url_error}
try: try:
entry = await asyncio.get_running_loop().run_in_executor( entry = await asyncio.get_running_loop().run_in_executor(
@@ -1569,7 +1650,14 @@ class DownloadQueue:
partial(self.__extract_info, url, ytdl_options_presets, ytdl_options_overrides), partial(self.__extract_info, url, ytdl_options_presets, ytdl_options_overrides),
) )
except yt_dlp.utils.YoutubeDLError as exc: except yt_dlp.utils.YoutubeDLError as exc:
return {'status': 'error', 'msg': str(exc)} msg = str(exc)
await self.__record_add_failure(
url, msg, download_type, codec, format, quality, folder,
custom_name_prefix, playlist_item_limit, split_by_chapters, chapter_template,
subtitle_language, subtitle_mode, ytdl_options_presets, ytdl_options_overrides,
clip_start, clip_end,
)
return {'status': 'error', 'msg': msg}
return await self.__add_entry( return await self.__add_entry(
entry, entry,
download_type, download_type,