diff --git a/python-backend/config.yml.example b/python-backend/config.yml.example index 695014e..13bda14 100644 --- a/python-backend/config.yml.example +++ b/python-backend/config.yml.example @@ -61,6 +61,19 @@ general: # a half-finished .tmp - is ignored. audio_extensions: [".mp3", ".ogg", ".oga", ".opus", ".flac", ".wav", ".m4a", ".aac"] + # How many episodes of each podcast show to keep, newest first. A show that has + # published for years grows without bound: GEOlino Spezial alone is 358 episodes and + # 5.8 GB, which will not sit next to the rest of a library on a Pi's SD card. + # + # The same number caps what gets downloaded, which is what makes the folder settle. + # Prune to the newest N but fetch everything the feed offers, and every poll would + # re-download the episodes the last one deleted. + # + # Lowering this DELETES the episodes that fall outside the window on the next poll, + # and an episode that has aged out of its feed cannot be fetched again. Use `null` to + # keep every episode and mind the free space yourself. + podcast_episode_limit: 50 + # The web front-end. Omit the whole section to run without it. # # There is no authentication: this is a device on a home network. The settings panel diff --git a/python-backend/musicmouse/__main__.py b/python-backend/musicmouse/__main__.py index 91a0965..3dc8b79 100644 --- a/python-backend/musicmouse/__main__.py +++ b/python-backend/musicmouse/__main__.py @@ -378,6 +378,7 @@ def _build_services( on_change=lambda: app.rescan_library( broadcast=web_service.hub.broadcast_library if web_service else None ), + episode_limit=app.config.general.podcast_episode_limit, ) ) diff --git a/python-backend/musicmouse/config.py b/python-backend/musicmouse/config.py index c9315f2..4812a50 100644 --- a/python-backend/musicmouse/config.py +++ b/python-backend/musicmouse/config.py @@ -26,6 +26,7 @@ from ruamel.yaml.error import YAMLError from musicmouse.color import ColorRGBW, parse_color from musicmouse.hardware import NO_FIGURE_TAG, RFID_TAG_LENGTH +from musicmouse.library.podcast_feeds import DEFAULT_EPISODE_LIMIT _log = logging.getLogger(__name__) @@ -288,6 +289,14 @@ class GeneralConfig(_Strict): audio_extensions: tuple[str, ...] = DEFAULT_AUDIO_EXTENSIONS + #: How many episodes of each podcast show to keep, newest first. A show that has + #: published for years grows without bound and will eventually fill the device's SD + #: card. ``null`` keeps every episode, and minding the free space is then on you. + #: + #: Lowering this *deletes* the episodes that fall outside the window on the next + #: poll, and an episode that has aged out of its feed cannot be fetched again. + podcast_episode_limit: int | None = Field(default=DEFAULT_EPISODE_LIMIT, ge=1) + @property def serial_simulated(self) -> bool: return self.serial_port == SIMULATE diff --git a/python-backend/musicmouse/library/podcast_feeds.py b/python-backend/musicmouse/library/podcast_feeds.py index ec3f6f3..a2b3ec7 100644 --- a/python-backend/musicmouse/library/podcast_feeds.py +++ b/python-backend/musicmouse/library/podcast_feeds.py @@ -12,6 +12,12 @@ a feed offers real per-episode artwork, it's saved alongside as a same-named sid image, which ``scanner.py``'s ``_cover_for_episode`` picks up automatically. A video-only enclosure (some shows publish no audio feed at all) is transcoded to audio via the ``ffmpeg`` binary, which must be on ``PATH`` for those shows to sync. +A show folder is also kept to a fixed number of the newest episodes +(:data:`DEFAULT_EPISODE_LIMIT`), so a long-running feed cannot fill the device's disk. +The same limit caps what is downloaded, which is what stops the two halves fighting: +prune what is older than the newest N, download only the newest N, and the folder +settles instead of re-fetching every episode it just deleted. + Nothing here raises on bad input - an unreachable feed or a broken enclosure is logged and skipped, not a crash, mirroring ``scanner.py``'s own rule. """ @@ -37,6 +43,7 @@ from musicmouse.library.sections import SECTIONS _log = logging.getLogger(__name__) __all__ = [ + "DEFAULT_EPISODE_LIMIT", "FEED_MARKER_NAME", "AudioExtractionError", "Episode", @@ -46,7 +53,9 @@ __all__ = [ "episode_filename", "find_feed_shows", "missing_episodes", + "newest_episodes", "parse_feed", + "prune_show", "resolve_episode_cover", "sync_all_shows", "sync_show", @@ -54,6 +63,15 @@ __all__ = [ FEED_MARKER_NAME: Final = "feed.txt" +#: How many episodes of one show to keep, newest first. A show that has published for +#: years is otherwise unbounded: GEOlino Spezial alone is 358 episodes and 5.8 GB, which +#: does not fit next to the rest of the library on a Pi's SD card. +#: +#: This caps downloads as well as deletions, and it has to be one number for both. Prune +#: to the newest N but download everything the feed offers, and every poll would +#: re-fetch the episodes the last one deleted, forever. +DEFAULT_EPISODE_LIMIT: Final = 50 + #: Enclosure content-type -> file extension, for a URL whose own suffix is missing or #: not a real extension (tracking-redirect URLs are common in the wild). _EXTENSION_BY_TYPE: Final[dict[str, str]] = { @@ -287,6 +305,80 @@ def missing_episodes(folder: Path, episodes: list[Episode]) -> list[tuple[Episod return out +def newest_episodes(episodes: list[Episode], keep: int | None) -> list[Episode]: + """The ``keep`` most recently published episodes, newest first. + + Applied to the *feed* before anything is downloaded. Without it, capping the folder + would be pointless: the next poll would see every pruned episode as missing again. + """ + ordered = sorted(episodes, key=lambda episode: episode.published, reverse=True) + return ordered if keep is None else ordered[:keep] + + +def _episode_date(name: str) -> datetime | None: + """The date out of a ``YYYYMMDD - Title.ext`` filename, or ``None`` if it has none. + + Deliberately strict. A file that does not follow the convention is one this module + did not write - something dropped in by hand under another name - and it is left + alone rather than guessed at, because the alternative is deleting somebody's file + on a parse that happened to fail. + """ + stem = Path(name).stem + if len(stem) < 8 or not stem[:8].isdigit(): + return None + try: + return datetime.strptime(stem[:8], "%Y%m%d").replace(tzinfo=UTC) + except ValueError: + return None + + +def prune_show(folder: Path, keep: int | None, extensions: frozenset[str] | None = None) -> int: + """Delete all but the ``keep`` newest episodes in ``folder``. Returns how many went. + + Only touches files named the way :func:`episode_filename` names them, so a + ``feed.txt``, a ``folder.jpg``, the failed-download record and any hand-named file + are all safe. An episode's sidecar cover image goes with it - it shares the stem, + and leaving it behind would strand art for a track that no longer exists. + + ``keep`` of ``None`` disables pruning entirely. + """ + if keep is None: + return 0 + audio_extensions = extensions if extensions is not None else _KNOWN_EXTENSIONS + + dated: list[tuple[datetime, Path]] = [] + for path in folder.iterdir(): + if not path.is_file() or path.suffix.lower() not in audio_extensions: + continue + published = _episode_date(path.name) + if published is not None: + dated.append((published, path)) + + if len(dated) <= keep: + return 0 + + dated.sort(key=lambda item: (item[0], item[1].name), reverse=True) + removed = 0 + for _published, path in dated[keep:]: + try: + path.unlink() + except OSError as exc: + _log.warning("Could not delete old episode %s: %s", path, exc) + continue + removed += 1 + for image_extension in _IMAGE_EXTENSIONS: + sidecar = path.with_suffix(image_extension) + try: + sidecar.unlink(missing_ok=True) + except OSError as exc: + _log.debug("Could not delete cover %s: %s", sidecar, exc) + if removed: + _log.info( + "Pruned %d old episode(s) from %s, keeping the newest %d", removed, folder.name, keep + ) + return removed + + def _load_failed_downloads(folder: Path) -> dict[str, datetime]: """Filename -> when it last failed to download, for episodes ``sync_show`` should leave alone until :data:`_RETRY_BACKOFF` has passed. @@ -437,14 +529,28 @@ async def resolve_episode_cover(client: httpx2.AsyncClient, episode: Episode) -> return _extract_og_image(response.text) -async def sync_show(client: httpx2.AsyncClient, folder: Path, feed_url: str) -> bool: - """Download every episode in ``feed_url`` that ``folder`` doesn't have yet. +async def sync_show( + client: httpx2.AsyncClient, + folder: Path, + feed_url: str, + *, + keep: int | None = DEFAULT_EPISODE_LIMIT, +) -> bool: + """Bring ``folder`` up to date with the newest ``keep`` episodes of ``feed_url``. - Returns whether anything changed. Errors - an unreachable feed, a malformed one, a - single broken enclosure - are logged and swallowed here so one bad show never stops - the others or takes down the poll loop. An episode that fails is remembered and left - alone for :data:`_RETRY_BACKOFF` before it's attempted again, so a permanently dead - enclosure doesn't get hammered on every poll. + Downloads what is missing from that window and deletes what has fallen out of it. + Both halves use the same ``keep``, which is what makes the folder settle - see + :data:`DEFAULT_EPISODE_LIMIT`. + + Returns whether anything changed, deletions included: a pruned folder needs a + rescan just as much as a downloaded episode does, or the library keeps offering + tracks whose files are gone. + + Errors - an unreachable feed, a malformed one, a single broken enclosure - are + logged and swallowed here so one bad show never stops the others or takes down the + poll loop. An episode that fails is remembered and left alone for + :data:`_RETRY_BACKOFF` before it's attempted again, so a permanently dead enclosure + doesn't get hammered on every poll. """ try: response = await client.get(feed_url, timeout=_HTTP_TIMEOUT, follow_redirects=True) @@ -453,7 +559,7 @@ async def sync_show(client: httpx2.AsyncClient, folder: Path, feed_url: str) -> _log.warning("Could not fetch podcast feed %s for %s: %s", feed_url, folder.name, exc) return False - pending = missing_episodes(folder, parse_feed(response.content)) + pending = missing_episodes(folder, newest_episodes(parse_feed(response.content), keep)) failed = _load_failed_downloads(folder) now = datetime.now(UTC) changed = False @@ -490,13 +596,20 @@ async def sync_show(client: httpx2.AsyncClient, folder: Path, feed_url: str) -> pending_filenames = {filename for _episode, filename in pending} failed = {filename: when for filename, when in failed.items() if filename in pending_filenames} _save_failed_downloads(folder, failed) + + # After downloading, not before: an episode that just arrived is one of the newest + # and must be counted when deciding what falls off the end. + if prune_show(folder, keep): + changed = True return changed -async def sync_all_shows(client: httpx2.AsyncClient, root: Path) -> bool: +async def sync_all_shows( + client: httpx2.AsyncClient, root: Path, *, keep: int | None = DEFAULT_EPISODE_LIMIT +) -> bool: """Poll every show with a feed marker under ``root``. Returns whether any changed.""" changed = False for _section_name, folder, feed_url in find_feed_shows(root): - if await sync_show(client, folder, feed_url): + if await sync_show(client, folder, feed_url, keep=keep): changed = True return changed diff --git a/python-backend/musicmouse/services/podcasts.py b/python-backend/musicmouse/services/podcasts.py index f177052..2586fa3 100644 --- a/python-backend/musicmouse/services/podcasts.py +++ b/python-backend/musicmouse/services/podcasts.py @@ -13,7 +13,7 @@ from typing import Final import httpx2 from musicmouse.library import MusicLibrary -from musicmouse.library.podcast_feeds import sync_all_shows +from musicmouse.library.podcast_feeds import DEFAULT_EPISODE_LIMIT, sync_all_shows _log = logging.getLogger(__name__) @@ -34,11 +34,13 @@ class PodcastFeedService: client: httpx2.AsyncClient, on_change: Callable[[], Awaitable[None]], interval: float = _CHECK_INTERVAL_SECONDS, + episode_limit: int | None = DEFAULT_EPISODE_LIMIT, ) -> None: self.library = library self.client = client self.on_change = on_change self.interval = interval + self.episode_limit = episode_limit async def run(self) -> None: """Checks immediately at startup, then every `interval` seconds. @@ -51,7 +53,8 @@ class PodcastFeedService: try: while True: try: - if await sync_all_shows(self.client, self.library.root): + keep = self.episode_limit + if await sync_all_shows(self.client, self.library.root, keep=keep): await self.on_change() except Exception: _log.exception("Podcast feed check failed; will retry next interval") diff --git a/python-backend/tests/test_podcast_feeds.py b/python-backend/tests/test_podcast_feeds.py index 9618594..193d889 100644 --- a/python-backend/tests/test_podcast_feeds.py +++ b/python-backend/tests/test_podcast_feeds.py @@ -19,7 +19,9 @@ from musicmouse.library.podcast_feeds import ( episode_filename, find_feed_shows, missing_episodes, + newest_episodes, parse_feed, + prune_show, resolve_episode_cover, sync_all_shows, sync_show, @@ -630,3 +632,148 @@ async def test_sync_show_survives_a_corrupt_failure_record(tmp_path: Path) -> No assert changed is True assert (folder / "20250101 - Episode One.mp3").exists() + + +# --------------------------------------------------------------- episode retention + + +def _episode(day: int, *, title: str | None = None) -> Episode: + return Episode( + title=title or f"Episode {day}", + published=datetime(2025, 1, day, tzinfo=UTC), + enclosure_url=f"http://example.com/ep{day}.mp3", + enclosure_type="audio/mpeg", + ) + + +def _write_episode(folder: Path, name: str) -> Path: + path = folder / name + path.write_bytes(b"audio") + return path + + +def test_newest_episodes_takes_the_most_recent_and_orders_them_newest_first() -> None: + episodes = [_episode(1), _episode(5), _episode(3)] + assert [e.published.day for e in newest_episodes(episodes, 2)] == [5, 3] + + +def test_newest_episodes_keeps_everything_when_the_limit_is_none() -> None: + episodes = [_episode(1), _episode(5), _episode(3)] + assert len(newest_episodes(episodes, None)) == 3 + + +def test_prune_show_deletes_only_the_oldest_beyond_the_limit(tmp_path: Path) -> None: + folder = tmp_path / "show" + folder.mkdir() + for day in range(1, 6): + _write_episode(folder, f"2025010{day} - Episode {day}.mp3") + + assert prune_show(folder, 2) == 3 + + remaining = sorted(path.name for path in folder.iterdir()) + assert remaining == ["20250104 - Episode 4.mp3", "20250105 - Episode 5.mp3"] + + +def test_prune_show_is_a_noop_below_the_limit(tmp_path: Path) -> None: + folder = tmp_path / "show" + folder.mkdir() + _write_episode(folder, "20250101 - Episode 1.mp3") + assert prune_show(folder, 50) == 0 + assert prune_show(folder, None) == 0 + assert (folder / "20250101 - Episode 1.mp3").exists() + + +def test_prune_show_takes_the_sidecar_cover_with_the_episode(tmp_path: Path) -> None: + folder = tmp_path / "show" + folder.mkdir() + _write_episode(folder, "20250101 - Old.mp3") + (folder / "20250101 - Old.jpg").write_bytes(b"art") + _write_episode(folder, "20250102 - New.mp3") + (folder / "20250102 - New.jpg").write_bytes(b"art") + + assert prune_show(folder, 1) == 1 + + assert not (folder / "20250101 - Old.jpg").exists() + assert (folder / "20250102 - New.jpg").exists() + + +def test_prune_show_leaves_everything_it_did_not_name(tmp_path: Path) -> None: + """A folder holds more than episodes, and a hand-placed file is not ours to delete.""" + folder = tmp_path / "show" + folder.mkdir() + for day in range(1, 4): + _write_episode(folder, f"2025010{day} - Episode {day}.mp3") + (folder / "feed.txt").write_text("http://example.com/feed.xml\n") + (folder / "folder.jpg").write_bytes(b"art") + (folder / ".failed-downloads.json").write_text("{}") + _write_episode(folder, "Grandpa's tape.mp3") + + prune_show(folder, 1) + + assert (folder / "feed.txt").exists() + assert (folder / "folder.jpg").exists() + assert (folder / ".failed-downloads.json").exists() + assert (folder / "Grandpa's tape.mp3").exists() + + +@pytest.mark.asyncio +async def test_sync_show_prunes_to_the_limit_after_downloading(tmp_path: Path) -> None: + folder = tmp_path / "show" + folder.mkdir() + + def handler(request: httpx2.Request) -> httpx2.Response: + if str(request.url) == "http://example.com/feed.xml": + return httpx2.Response(200, content=_FEED) + return httpx2.Response(200, content=b"audio-bytes") + + async with httpx2.AsyncClient(transport=httpx2.MockTransport(handler)) as client: + changed = await sync_show(client, folder, "http://example.com/feed.xml", keep=1) + + assert changed is True + episodes = sorted(path.name for path in folder.iterdir() if path.suffix == ".mp3") + assert episodes == ["20250103 - Episode Two_ Special_Chars_.mp3"] + + +@pytest.mark.asyncio +async def test_sync_show_does_not_redownload_what_it_pruned(tmp_path: Path) -> None: + """The whole point of capping downloads as well as deletions. + + Prune to the newest N but fetch everything the feed offers, and each poll would + re-download the episodes the previous one deleted - forever, at full size. + """ + folder = tmp_path / "show" + folder.mkdir() + downloads: list[str] = [] + + def handler(request: httpx2.Request) -> httpx2.Response: + if str(request.url) == "http://example.com/feed.xml": + return httpx2.Response(200, content=_FEED) + downloads.append(str(request.url)) + return httpx2.Response(200, content=b"audio-bytes") + + async with httpx2.AsyncClient(transport=httpx2.MockTransport(handler)) as client: + await sync_show(client, folder, "http://example.com/feed.xml", keep=1) + first_pass = len(downloads) + changed_again = await sync_show(client, folder, "http://example.com/feed.xml", keep=1) + + assert changed_again is False + assert len(downloads) == first_pass + + +@pytest.mark.asyncio +async def test_sync_show_reports_a_change_when_it_only_pruned(tmp_path: Path) -> None: + """A deletion needs a rescan as much as a download does, or the library goes on + offering tracks whose files are gone.""" + folder = tmp_path / "show" + folder.mkdir() + _write_episode(folder, "20250101 - Episode One.mp3") + _write_episode(folder, "20250103 - Episode Two_ Special_Chars_.mp3") + + def handler(request: httpx2.Request) -> httpx2.Response: + return httpx2.Response(200, content=_FEED) + + async with httpx2.AsyncClient(transport=httpx2.MockTransport(handler)) as client: + changed = await sync_show(client, folder, "http://example.com/feed.xml", keep=1) + + assert changed is True + assert not (folder / "20250101 - Episode One.mp3").exists() diff --git a/python-backend/tests/test_podcast_service.py b/python-backend/tests/test_podcast_service.py index dc6d89c..1cf758e 100644 --- a/python-backend/tests/test_podcast_service.py +++ b/python-backend/tests/test_podcast_service.py @@ -32,7 +32,9 @@ async def test_run_checks_immediately_and_calls_on_change_only_when_something_ch ) -> None: calls = 0 - async def fake_sync_all_shows(client: httpx2.AsyncClient, root: Path) -> bool: + async def fake_sync_all_shows( + client: httpx2.AsyncClient, root: Path, *, keep: int | None = None + ) -> bool: nonlocal calls calls += 1 return calls == 1 # only the very first pass finds something new @@ -63,7 +65,9 @@ async def test_run_survives_a_failing_pass( ) -> None: calls = 0 - async def fake_sync_all_shows(client: httpx2.AsyncClient, root: Path) -> bool: + async def fake_sync_all_shows( + client: httpx2.AsyncClient, root: Path, *, keep: int | None = None + ) -> bool: nonlocal calls calls += 1 if calls == 1: diff --git a/scripts/sync-library.sh b/scripts/sync-library.sh index 184581d..c81e7b3 100755 --- a/scripts/sync-library.sh +++ b/scripts/sync-library.sh @@ -206,6 +206,9 @@ if [[ $COPY_CACHE -eq 1 ]]; then for part in covers analysis; do [[ -d "${CACHE}/${part}" ]] || continue echo "--- ${part}" + # rsync creates the last path component but not a missing one above it, and on a + # fresh device neither .musicmouse-cache nor its subdirectory exists yet. + [[ $DRY_RUN -eq 1 ]] || ssh "$SSH_TARGET" "mkdir -p '${DEST}/.musicmouse-cache/${part}'" # No --delete: the device may have analysed tracks this machine never saw. rsync "${RSYNC_OPTS[@]}" \ "${CACHE}/${part}/" "${SSH_TARGET}:${DEST}/.musicmouse-cache/${part}/"