diff --git a/app/music_metadata.py b/app/music_metadata.py new file mode 100644 index 0000000..a3b1bef --- /dev/null +++ b/app/music_metadata.py @@ -0,0 +1,124 @@ +"""Conservative music metadata enrichment for audio downloads. + +This module only consumes fields already supplied by yt-dlp or retained on +MeTube's queued playlist entry. It intentionally performs no external lookup +or site-specific album detection. +""" + +from __future__ import annotations + +from typing import Any, Optional + +from yt_dlp.postprocessor.common import PostProcessor + + +def _has_value(value: Any) -> bool: + if isinstance(value, str): + return bool(value.strip()) + if isinstance(value, (list, tuple)): + return any(_has_value(item) for item in value) + return value is not None + + +def _positive_int(value: Any) -> Optional[int]: + if isinstance(value, bool): + return None + try: + number = int(value) + except (TypeError, ValueError): + return None + return number if number > 0 else None + + +def _track_position(value: Any) -> tuple[Optional[int], Optional[int]]: + """Return a track number and optional total from a scalar or ``n/total``.""" + if isinstance(value, str) and '/' in value: + number, total = value.split('/', 1) + return _positive_int(number.strip()), _positive_int(total.strip()) + return _positive_int(value), None + + +def _first_positive_int(*values: Any) -> Optional[int]: + return next((number for value in values if (number := _positive_int(value))), None) + + +def _has_album_signal(info: dict[str, Any], source_entry: dict[str, Any]) -> bool: + """Use only extractor-owned fields to identify album-level metadata.""" + return any( + _has_value(entry.get(key)) + for entry in (info, source_entry) + for key in ('album', 'track_number') + ) + + +def _is_music_audio(info: dict[str, Any], source_entry: dict[str, Any]) -> bool: + return _has_album_signal(info, source_entry) or any( + _has_value(entry.get(key)) + for entry in (info, source_entry) + for key in ('track', 'artists') + ) + + +def prefer_square_thumbnail(info: dict[str, Any]) -> None: + """Move the largest known square thumbnail to yt-dlp's preferred slot.""" + thumbnails = info.get('thumbnails') + if not isinstance(thumbnails, list) or len(thumbnails) < 2: + return + + candidates: list[tuple[int, int]] = [] + for index, thumbnail in enumerate(thumbnails): + if not isinstance(thumbnail, dict): + continue + width = _positive_int(thumbnail.get('width')) + height = _positive_int(thumbnail.get('height')) + if width is not None and width == height: + candidates.append((width * height, index)) + if not candidates: + return + + _, selected_index = max(candidates) + selected = thumbnails.pop(selected_index) + thumbnails.append(selected) + if selected.get('url'): + info['thumbnail'] = selected['url'] + + +class MusicMetadataPreProcessor(PostProcessor): + """Enrich extracted audio metadata using extractor-owned album signals.""" + + def __init__(self, downloader=None, *, source_entry=None): + super().__init__(downloader) + self._source_entry = source_entry if isinstance(source_entry, dict) else {} + + def run(self, info): + if _has_album_signal(info, self._source_entry): + number, inline_total = _track_position(info.get('track_number')) + if number is None: + number, source_inline_total = _track_position( + self._source_entry.get('track_number') + ) + inline_total = inline_total or source_inline_total + if number is None: + number = _positive_int(self._source_entry.get('playlist_index')) + + total = inline_total or _first_positive_int( + info.get('track_count'), + info.get('track_total'), + self._source_entry.get('track_count'), + self._source_entry.get('track_total'), + self._source_entry.get('playlist_count'), + self._source_entry.get('n_entries'), + ) + if number is not None: + info['track_number'] = f'{number}/{total}' if total is not None else number + + if not _has_value(info.get('album')): + album = self._source_entry.get('album') or self._source_entry.get( + 'playlist_title' + ) + if isinstance(album, str) and album.strip(): + info['album'] = album.strip() + + if _is_music_audio(info, self._source_entry): + prefer_square_thumbnail(info) + return [], info diff --git a/app/tests/test_music_metadata.py b/app/tests/test_music_metadata.py new file mode 100644 index 0000000..e3493a5 --- /dev/null +++ b/app/tests/test_music_metadata.py @@ -0,0 +1,119 @@ +"""Tests for conservative audio metadata enrichment.""" + +from __future__ import annotations + +from music_metadata import MusicMetadataPreProcessor + + +def _preprocess(source_entry, info): + processor = MusicMetadataPreProcessor(source_entry=source_entry) + _, result = processor.run(info) + return result + + +def test_album_uses_existing_order_and_total_when_track_number_is_missing(): + result = _preprocess( + { + 'playlist_index': '03', + 'playlist_count': 12, + 'playlist_title': 'Example Album', + }, + {'title': 'Track', 'album': 'Example Album'}, + ) + + assert result['track_number'] == '3/12' + assert result['album'] == 'Example Album' + + +def test_official_track_number_wins_over_album_order(): + result = _preprocess( + {'playlist_index': 3, 'playlist_count': 12}, + {'track_number': 7, 'album': 'Official Album'}, + ) + + assert result['track_number'] == '7/12' + assert result['album'] == 'Official Album' + + +def test_inline_official_track_total_is_preserved(): + result = _preprocess( + {'playlist_count': 12}, + {'track_number': '4/10', 'album': 'Official Album'}, + ) + + assert result['track_number'] == '4/10' + + +def test_source_track_number_and_total_are_retained_from_flat_extraction(): + result = _preprocess( + {'track_number': 2, 'track_count': 9, 'playlist_index': 4}, + {'title': 'Track'}, + ) + + assert result['track_number'] == '2/9' + + +def test_album_title_falls_back_to_source_playlist_title(): + result = _preprocess( + {'playlist_title': 'Example Album'}, + {'track_number': 4}, + ) + + assert result['album'] == 'Example Album' + assert result['track_number'] == 4 + + +def test_playlist_without_extractor_album_signals_is_not_changed(): + result = _preprocess( + { + 'playlist_index': 3, + 'playlist_count': 12, + 'playlist_title': 'Example Playlist', + }, + {'title': 'Track'}, + ) + + assert 'album' not in result + assert 'track_number' not in result + + +def test_regular_video_artwork_is_not_changed(): + thumbnails = [ + {'url': 'square.jpg', 'width': 500, 'height': 500}, + {'url': 'landscape.jpg', 'width': 1280, 'height': 720}, + ] + result = _preprocess({}, {'title': 'Regular Video', 'thumbnails': thumbnails.copy()}) + + assert result['thumbnails'] == thumbnails + assert 'thumbnail' not in result + + +def test_music_audio_prefers_largest_existing_square_thumbnail(): + result = _preprocess( + {}, + { + 'track': 'Track', + 'thumbnails': [ + {'url': 'small-square.jpg', 'width': 200, 'height': 200}, + {'url': 'large-square.jpg', 'width': 1000, 'height': 1000}, + {'url': 'landscape.jpg', 'width': 1280, 'height': 720}, + ], + }, + ) + + assert result['thumbnails'][-1]['url'] == 'large-square.jpg' + assert result['thumbnail'] == 'large-square.jpg' + + +def test_landscape_only_music_artwork_keeps_existing_order(): + thumbnails = [ + {'url': 'small.jpg', 'width': 640, 'height': 360}, + {'url': 'large.jpg', 'width': 1280, 'height': 720}, + ] + result = _preprocess( + {}, + {'track': 'Track', 'thumbnails': thumbnails.copy()}, + ) + + assert result['thumbnails'] == thumbnails + assert 'thumbnail' not in result diff --git a/app/tests/test_ytdl_utils.py b/app/tests/test_ytdl_utils.py index de13a7a..692bad3 100644 --- a/app/tests/test_ytdl_utils.py +++ b/app/tests/test_ytdl_utils.py @@ -73,6 +73,7 @@ import ytdl from ytdl import ( Download, DownloadInfo, + MusicMetadataPreProcessor, _compact_persisted_entry, _convert_srt_to_txt_file, _AlbumArtistPostProcessor, @@ -201,9 +202,15 @@ class AlbumArtistRegistrationTests(unittest.TestCase): result = download._make_youtube_dl({'quiet': True}) self.assertIs(result, fake_ydl) - postprocessor, = fake_ydl.add_post_processor.call_args.args + album_artist_call = fake_ydl.add_post_processor.call_args_list[0] + postprocessor, = album_artist_call.args self.assertIsInstance(postprocessor, _AlbumArtistPostProcessor) - self.assertEqual(fake_ydl.add_post_processor.call_args.kwargs, {'when': 'pre_process'}) + self.assertEqual(album_artist_call.kwargs, {'when': 'pre_process'}) + metadata_pre_call = fake_ydl.add_post_processor.call_args_list[1] + metadata_preprocessor, = metadata_pre_call.args + self.assertIsInstance(metadata_preprocessor, MusicMetadataPreProcessor) + self.assertEqual(metadata_pre_call.kwargs, {'when': 'pre_process'}) + self.assertEqual(fake_ydl.add_post_processor.call_count, 2) def test_video_download_does_not_register_postprocessor(self): download = _make_test_download() diff --git a/app/ytdl.py b/app/ytdl.py index be36bf9..97c43a1 100644 --- a/app/ytdl.py +++ b/app/ytdl.py @@ -23,6 +23,7 @@ from yt_dlp.postprocessor.common import PostProcessor from yt_dlp.utils import STR_FORMAT_RE_TMPL, STR_FORMAT_TYPES import bg_tasks from dl_formats import get_format, get_opts, AUDIO_FORMATS, merge_ytdl_option_layers +from music_metadata import MusicMetadataPreProcessor from datetime import datetime from state_store import AtomicJsonStore, from_json_compatible, read_legacy_shelf, to_json_compatible from subscriptions import _entry_id @@ -625,6 +626,13 @@ class Download: ) if getattr(self.info, 'download_type', '') == 'audio': ydl.add_post_processor(_AlbumArtistPostProcessor(ydl), when='pre_process') + ydl.add_post_processor( + MusicMetadataPreProcessor( + ydl, + source_entry=getattr(self.info, 'entry', None), + ), + when='pre_process', + ) return ydl def _download(self):