Add per-episode podcast cover art and fix nested album grouping

- podcast_feeds.py: fetch per-episode cover art from a feed's itunes:image
  (when it's genuinely distinct from the channel image) or, failing that,
  from the og:image on the episode's own linked page; extract audio from
  video-only enclosures via ffmpeg; match podcast-dl's filename convention
  (illegal characters become "_" instead of being dropped) so enabling
  feed.txt on an already-downloaded show doesn't re-download its back catalog
- scanner.py: _cover_for_episode now checks for a same-stem sidecar cover
  image before falling back to embedded ID3 art and the shared folder cover
- scanner.py/__init__.py: fixed a bug where an album folder nested one level
  deeper than usual (an age-range grouping folder, say) was mistaken for an
  empty album and skipped; added periodic progress logging for long scans
  and analysis passes

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
2026-09-11 09:54:23 +02:00
parent 747c390303
commit 69119bb72a
5 changed files with 635 additions and 46 deletions

View File

@@ -11,6 +11,7 @@ import asyncio
import contextlib
import logging
import os
import time
from collections.abc import Awaitable, Callable, Collection, Mapping
from dataclasses import replace
from pathlib import Path
@@ -62,6 +63,11 @@ _BUSY_POLL_SECONDS: Final = 2.0
#: browser tabs rather than only after the whole thing finishes.
_PUBLISH_BATCH_SIZE: Final = 25
#: How often a long analysis pass reports where it is, so a first-time run over a large
#: library - minutes of DSP per track - doesn't sit silent with nothing on the console
#: to say it is still going.
_PROGRESS_INTERVAL_SECONDS: Final = 5.0
def _analyze_one(
analyzer: Analyzer, path: Path
@@ -273,6 +279,7 @@ class MusicLibrary:
return 0
done = 0
last_report = time.monotonic()
for album in self.albums:
if album.kind not in kinds:
continue
@@ -283,6 +290,10 @@ class MusicLibrary:
cached = self.cache.load_analysis(key)
if cached is not None and cached.version >= analyzer.version:
continue
now = time.monotonic()
if now - last_report >= _PROGRESS_INTERVAL_SECONDS:
_log.info("Analyzing library: %d tracks done so far, now on %s", done, track.path)
last_report = now
try:
analysis, grid, curve = await asyncio.to_thread(
_analyze_one, analyzer, track.path

View File

@@ -7,15 +7,21 @@ its shape entirely off the folder tree (see :mod:`musicmouse.library.sections`).
Downloaded episodes land in the show folder using the exact ``YYYYMMDD - Title.ext``
convention ``sections.py`` already documents for ``Kinderpodcasts``, so a freshly
downloaded episode sorts and scans exactly like one a person dropped in by hand.
downloaded episode sorts and scans exactly like one a person dropped in by hand. When
a feed offers real per-episode artwork, it's saved alongside as a same-named sidecar
image, which ``scanner.py``'s ``_cover_for_episode`` picks up automatically. A
video-only enclosure (some shows publish no audio feed at all) is transcoded to audio
via the ``ffmpeg`` binary, which must be on ``PATH`` for those shows to sync.
Nothing here raises on bad input - an unreachable feed or a broken enclosure is logged
and skipped, not a crash, mirroring ``scanner.py``'s own rule.
"""
from __future__ import annotations
import asyncio
import logging
import re
import shutil
from dataclasses import dataclass
from datetime import UTC, datetime
from pathlib import Path
@@ -31,12 +37,16 @@ _log = logging.getLogger(__name__)
__all__ = [
"FEED_MARKER_NAME",
"AudioExtractionError",
"Episode",
"download_episode",
"download_episode_cover",
"episode_cover_filename",
"episode_filename",
"find_feed_shows",
"missing_episodes",
"parse_feed",
"resolve_episode_cover",
"sync_all_shows",
"sync_show",
]
@@ -59,11 +69,34 @@ _EXTENSION_BY_TYPE: Final[dict[str, str]] = {
_DEFAULT_EXTENSION: Final = ".mp3"
_KNOWN_EXTENSIONS: Final = frozenset({".mp3", ".m4a", ".aac", ".ogg", ".opus", ".wav", ".flac"})
#: Cover URL suffix -> file extension, for the per-episode sidecar image.
_IMAGE_EXTENSIONS: Final[frozenset[str]] = frozenset({".jpg", ".jpeg", ".png", ".webp"})
_DEFAULT_IMAGE_EXTENSION: Final = ".jpg"
#: Characters illegal (or awkward) in a filename, plus the path separators themselves.
#: Replaced with "_" rather than dropped, matching the convention every episode
#: already downloaded by hand (via the ``podcast-dl`` CLI) was named with - a
#: mismatch here would make every one of them look "new" to `missing_episodes`.
_ILLEGAL_FILENAME_CHARS: Final = re.compile(r'[\\/:*?"<>|]')
_MAX_TITLE_LENGTH: Final = 120
_HTTP_TIMEOUT: Final = 30.0
#: `og:image` scraping is a fallback for a page we don't control; kept short so one
#: slow or hanging host can't stall a whole sync pass.
_OG_IMAGE_TIMEOUT: Final = 15.0
_OG_IMAGE_RE: Final = re.compile(
r'<meta[^>]+property=["\']og:image["\'][^>]*content=["\']([^"\']+)["\']'
r'|<meta[^>]+content=["\']([^"\']+)["\'][^>]*property=["\']og:image["\']',
re.IGNORECASE,
)
#: How long a video enclosure is given to download-and-transcode before it's treated
#: as failed - generous, since this runs on a Pi and a long video can take a while.
_FFMPEG_TIMEOUT: Final = 600.0
class AudioExtractionError(Exception):
"""Raised when a video enclosure could not be turned into an audio file."""
@dataclass(frozen=True, slots=True)
@@ -72,6 +105,15 @@ class Episode:
published: datetime
enclosure_url: str
enclosure_type: str
#: The feed's own per-episode artwork, when it has one distinct from the show's
#: overall cover. ``None`` when the feed has no image at all, or (as with GEOlino)
#: every item merely repeats the channel's own image.
cover_url: str | None = None
#: The episode's own page, when the feed links to one distinct from the show's
#: general page - the fallback route to real per-episode art for a feed (like
#: Wissen macht Ah) that has no per-item image of its own, via that page's
#: `og:image`. ``None`` when the feed has no such per-episode page.
link: str | None = None
def find_feed_shows(root: Path) -> list[tuple[str, Path, str]]:
@@ -110,12 +152,19 @@ def _read_feed_url(marker: Path) -> str | None:
def parse_feed(content: bytes) -> list[Episode]:
"""Every episode in a feed that has both a publish date and an audio enclosure.
"""Every episode in a feed that has both a publish date and a playable enclosure.
Anything else is skipped and logged rather than raising - a malformed or unusual
entry in one feed must never stop the episodes around it from being picked up.
"""
parsed = feedparser.parse(content)
# feedparser exposes an `itunes:image` identically at the channel and item level,
# under the same "image" key an ordinary RSS `<image>` uses - there is no separate
# `itunes_image` field. Read the channel's own image/link once so each entry can
# tell whether it has something genuinely its own, or is just repeating them
# (as GEOlino's per-item `itunes:image` does).
channel_image = parsed.feed.get("image", {}).get("href")
channel_link = parsed.feed.get("link")
episodes: list[Episode] = []
for entry in parsed.entries:
published = _entry_published(entry)
@@ -124,15 +173,26 @@ def parse_feed(content: bytes) -> list[Episode]:
continue
enclosure = _entry_enclosure(entry)
if enclosure is None:
_log.debug("Feed entry %r has no audio enclosure; skipping", entry.get("title"))
_log.debug(
"Feed entry %r has no audio or video enclosure; skipping", entry.get("title")
)
continue
url, enclosure_type = enclosure
item_image = entry.get("image", {}).get("href")
cover_url = item_image if item_image and item_image != channel_image else None
item_link = entry.get("link")
link = item_link if item_link and item_link != channel_link else None
episodes.append(
Episode(
title=entry.get("title") or url,
published=published,
enclosure_url=url,
enclosure_type=enclosure_type,
cover_url=cover_url,
link=link,
)
)
return episodes
@@ -147,12 +207,32 @@ def _entry_published(entry: Any) -> datetime | None:
def _entry_enclosure(entry: Any) -> tuple[str, str] | None:
"""The enclosure to download for one entry, preferring audio when both are offered.
A video enclosure (some shows, like Wissen macht Ah, publish no audio version at
all) is still returned rather than dropped - :func:`download_episode` turns it
into audio via ffmpeg. It's just the least preferred of the three: an audio
enclosure, or one with no declared type at all (assumed audio), both win outright.
"""
video: tuple[str, str] | None = None
for enclosure in entry.get("enclosures", []):
url = enclosure.get("href") or enclosure.get("url")
enclosure_type = enclosure.get("type") or ""
if url and (enclosure_type.startswith("audio/") or not enclosure_type):
if not url:
continue
if enclosure_type.startswith("audio/") or not enclosure_type:
return str(url), str(enclosure_type)
return None
if video is None and enclosure_type.startswith("video/"):
video = (str(url), str(enclosure_type))
return video
def _episode_stem(published: datetime, title: str) -> str:
"""``YYYYMMDD - Title``, with no extension yet - shared by the audio filename and
its sidecar cover image, so the two always line up."""
sanitized = _ILLEGAL_FILENAME_CHARS.sub("_", title).strip().strip(".")
sanitized = " ".join(sanitized.split())[:_MAX_TITLE_LENGTH] or "Episode"
return f"{published:%Y%m%d} - {sanitized}"
def episode_filename(
@@ -160,9 +240,13 @@ def episode_filename(
) -> str:
"""``YYYYMMDD - Title.ext``, matching the convention every hand-placed episode
already follows (see ``sections.py``'s module comment on ``Kinderpodcasts``)."""
sanitized = _ILLEGAL_FILENAME_CHARS.sub("", title).strip().strip(".")
sanitized = " ".join(sanitized.split())[:_MAX_TITLE_LENGTH] or "Episode"
return f"{published:%Y%m%d} - {sanitized}{_extension_for(enclosure_type, enclosure_url)}"
return f"{_episode_stem(published, title)}{_extension_for(enclosure_type, enclosure_url)}"
def episode_cover_filename(published: datetime, title: str, cover_url: str) -> str:
"""The sidecar image filename for an episode's cover, sharing its stem so the
scanner (``scanner.py``'s ``_cover_for_episode``) can find it next to the audio."""
return f"{_episode_stem(published, title)}{_image_extension_for(cover_url)}"
def _extension_for(enclosure_type: str, url: str) -> str:
@@ -172,6 +256,11 @@ def _extension_for(enclosure_type: str, url: str) -> str:
return _EXTENSION_BY_TYPE.get(enclosure_type, _DEFAULT_EXTENSION)
def _image_extension_for(url: str) -> str:
suffix = Path(urlsplit(url).path).suffix.lower()
return suffix if suffix in _IMAGE_EXTENSIONS else _DEFAULT_IMAGE_EXTENSION
def missing_episodes(folder: Path, episodes: list[Episode]) -> list[tuple[Episode, str]]:
"""Episodes whose target filename isn't already on disk, paired with that filename.
@@ -192,26 +281,117 @@ def missing_episodes(folder: Path, episodes: list[Episode]) -> list[tuple[Episod
async def download_episode(
client: httpx2.AsyncClient, folder: Path, episode: Episode, filename: str
) -> None:
"""Stream an episode to ``folder / filename`` via a dotfile temp path.
"""Get an episode to ``folder / filename`` via a dotfile temp path.
A half-written file must never look like a track: the scanner already skips
dotfiles for exactly this reason (see ``scanner.py``'s ``_audio_files``), so the
rename to the real name only happens once the download is complete.
rename to the real name only happens once the download is complete. A video
enclosure is routed through ffmpeg instead of being streamed as-is - see
:func:`_extract_audio`.
"""
temp_path = folder / f".downloading-{filename}.tmp"
try:
async with client.stream(
"GET", episode.enclosure_url, timeout=_HTTP_TIMEOUT
) as response:
response.raise_for_status()
with temp_path.open("wb") as handle:
async for chunk in response.aiter_bytes():
handle.write(chunk)
if episode.enclosure_type.startswith("video/"):
await _extract_audio(episode.enclosure_url, temp_path)
else:
async with client.stream(
"GET", episode.enclosure_url, timeout=_HTTP_TIMEOUT
) as response:
response.raise_for_status()
with temp_path.open("wb") as handle:
async for chunk in response.aiter_bytes():
handle.write(chunk)
temp_path.replace(folder / filename)
finally:
temp_path.unlink(missing_ok=True)
async def _extract_audio(source_url: str, dest: Path) -> None:
"""Pull ``source_url`` (a video enclosure) through ffmpeg, writing just its audio
track to ``dest``. ffmpeg fetches the URL itself, so the video is never stored.
Raises :class:`AudioExtractionError` rather than a bare ``OSError`` or timeout, so
callers can tell a missing/failing ffmpeg apart from an ordinary network error -
but either way this is meant to be logged and skipped, not fatal.
"""
if shutil.which("ffmpeg") is None:
raise AudioExtractionError(
"ffmpeg is not installed; cannot extract audio from a video enclosure"
)
process = await asyncio.create_subprocess_exec(
"ffmpeg",
"-y",
"-i",
source_url,
"-vn",
"-acodec",
"libmp3lame",
"-q:a",
"2",
str(dest),
stdout=asyncio.subprocess.DEVNULL,
stderr=asyncio.subprocess.PIPE,
)
try:
_, stderr = await asyncio.wait_for(process.communicate(), timeout=_FFMPEG_TIMEOUT)
except TimeoutError:
process.kill()
await process.wait()
raise AudioExtractionError(f"ffmpeg timed out extracting audio from {source_url}") from None
if process.returncode != 0:
raise AudioExtractionError(
f"ffmpeg exited {process.returncode} extracting audio from {source_url}: "
f"{stderr.decode(errors='replace')[-500:]}"
)
async def download_episode_cover(
client: httpx2.AsyncClient, folder: Path, cover_url: str, filename: str
) -> None:
"""Best-effort sidecar cover image fetch, atomic like :func:`download_episode`.
Cover art is small enough that streaming it in chunks isn't worth the extra code.
"""
temp_path = folder / f".downloading-{filename}.tmp"
try:
response = await client.get(cover_url, timeout=_HTTP_TIMEOUT, follow_redirects=True)
response.raise_for_status()
temp_path.write_bytes(response.content)
temp_path.replace(folder / filename)
finally:
temp_path.unlink(missing_ok=True)
def _extract_og_image(html: str) -> str | None:
match = _OG_IMAGE_RE.search(html)
if match is None:
return None
return match.group(1) or match.group(2)
async def resolve_episode_cover(client: httpx2.AsyncClient, episode: Episode) -> str | None:
"""The best cover URL available for ``episode``, or ``None`` when there isn't one.
Prefers the feed's own per-episode image. Failing that, when the feed links to a
page of the episode's own (as Wissen macht Ah does, despite having no per-item
image tag), falls back to that page's ``og:image`` - one targeted fetch of a URL
the feed itself provided, not a scrape of a list page. Never raises: an
unreachable page, a redirect loop, or a page with no such tag all just mean no
cover was found here.
"""
if episode.cover_url:
return episode.cover_url
if not episode.link:
return None
try:
response = await client.get(episode.link, timeout=_OG_IMAGE_TIMEOUT, follow_redirects=True)
response.raise_for_status()
except httpx2.HTTPError as exc:
_log.debug("Could not fetch %s for its og:image: %s", episode.link, exc)
return None
return _extract_og_image(response.text)
async def sync_show(client: httpx2.AsyncClient, folder: Path, feed_url: str) -> bool:
"""Download every episode in ``feed_url`` that ``folder`` doesn't have yet.
@@ -231,13 +411,27 @@ async def sync_show(client: httpx2.AsyncClient, folder: Path, feed_url: str) ->
for episode, filename in pending:
try:
await download_episode(client, folder, episode, filename)
except (httpx2.HTTPError, OSError) as exc:
except (httpx2.HTTPError, OSError, AudioExtractionError) as exc:
_log.warning(
"Could not download episode %r for %s: %s", episode.title, folder.name, exc
)
continue
_log.info("Downloaded new episode %r for %s", episode.title, folder.name)
changed = True
cover_url = await resolve_episode_cover(client, episode)
if cover_url:
cover_filename = episode_cover_filename(episode.published, episode.title, cover_url)
if not (folder / cover_filename).exists():
try:
await download_episode_cover(client, folder, cover_url, cover_filename)
except (httpx2.HTTPError, OSError) as exc:
_log.warning(
"Could not download cover for episode %r in %s: %s",
episode.title,
folder.name,
exc,
)
return changed

View File

@@ -9,8 +9,9 @@ input - an unreadable file loses its metadata, not the boot.
from __future__ import annotations
import logging
import time
from collections import Counter
from collections.abc import Mapping
from collections.abc import Iterator, Mapping
from pathlib import Path
from musicmouse.library.cache import Fingerprint, LibraryCache
@@ -25,6 +26,33 @@ __all__ = ["scan_library"]
#: Checked in order for a cover sitting next to the audio.
_COVER_NAMES = ("cover.jpg", "cover.jpeg", "cover.png", "folder.jpg")
#: Extensions tried for an episode's own same-stem sidecar cover image (see
#: :func:`_cover_for_episode`), matching what `podcast_feeds.py`'s
#: ``episode_cover_filename`` can produce.
_SIDECAR_COVER_EXTENSIONS = (".jpg", ".jpeg", ".png", ".webp")
#: How often a long scan reports where it is, so a library of thousands of albums
#: doesn't sit silent for minutes with nothing on the console to say it is still going.
_PROGRESS_INTERVAL_SECONDS = 5.0
def _album_folders(folder: Path, extensions: frozenset[str]) -> Iterator[Path]:
"""Every leaf album folder under ``folder``, however deep it is nested.
A folder that holds audio files directly *is* an album - an artist who groups
their books under an extra "ab 3" / "ab 5" age-range folder, or a series folder
that groups its episodes one directory further down than usual, still bottoms out
here without needing to be special-cased. Only descended into when a folder holds
no audio of its own, so an ordinary album folder is never mistaken for a grouping
folder just because it also happens to contain subfolders.
"""
if _audio_files(folder, extensions):
yield folder
return
for sub in sorted(folder.iterdir(), key=lambda path: path.name):
if sub.is_dir() and not sub.name.startswith("."):
yield from _album_folders(sub, extensions)
def _audio_files(folder: Path, extensions: frozenset[str]) -> list[Path]:
"""The playable files in ``folder``, alphabetically.
@@ -106,9 +134,16 @@ def _cover_for(
def _cover_for_episode(
folder: Path, path: Path, identifier: str, cache: LibraryCache
) -> tuple[Path | None, bytes | None]:
"""The reverse priority from :func:`_cover_for`: with one album per episode, the
episode's own embedded art is the more specific and more correct source. The
folder's shared cover is the fallback for an episode whose file carries none."""
"""The reverse priority from :func:`_cover_for`: with one album per episode, art
specific to this one episode wins over the folder's shared cover. A same-stem
sidecar image (what ``podcast_feeds.py`` saves when a feed has real per-episode
art) is tried first - it's a plain file stat, cheaper than reading tags - then the
episode's own embedded art, then the folder's shared cover as the last resort for
an episode with neither."""
for extension in _SIDECAR_COVER_EXTENSIONS:
candidate = path.with_suffix(extension)
if candidate.is_file():
return candidate, candidate.read_bytes()
art = _embedded_art(path)
if art is not None:
return cache.store_cover(identifier, art), art
@@ -262,6 +297,7 @@ def scan_library(
cache.prepare()
known = known or {}
out: dict[str, tuple[Album, Fingerprint]] = {}
last_report = time.monotonic()
for section_name, section in SECTIONS.items():
section_root = root / section_name
@@ -269,16 +305,17 @@ def scan_library(
_log.warning("Library section %r has no folder at %s", section_name, section_root)
continue
for folder in sorted(section_root.iterdir(), key=lambda path: path.name):
if not folder.is_dir() or folder.name.startswith("."):
for top in sorted(section_root.iterdir(), key=lambda path: path.name):
if not top.is_dir() or top.name.startswith("."):
continue
if section.album_unit == "episode":
# Fingerprinted per episode inside scan_episodes; the whole-folder
# shortcut below does not apply since one folder yields many albums.
# shortcut below does not apply since one folder yields many albums,
# and an episode show is never nested deeper than this either.
out.update(
scan_episodes(
folder,
top,
root=root,
section_name=section_name,
section=section,
@@ -289,24 +326,34 @@ def scan_library(
)
continue
identifier = album_id(root, folder)
cached = known.get(identifier)
if cached is not None:
paths = _audio_files(folder, extensions)
if paths and Fingerprint.of(paths) == cached[1]:
out[identifier] = cached
continue
scanned = scan_album(
folder,
root=root,
section_name=section_name,
section=section,
extensions=extensions,
cache=cache,
figure_kinds=figure_kinds,
)
if scanned is not None:
out[identifier] = scanned
for folder in _album_folders(top, extensions):
now = time.monotonic()
if now - last_report >= _PROGRESS_INTERVAL_SECONDS:
_log.info(
"Scanning library: %d albums found so far, now in %s",
len(out),
folder.relative_to(root),
)
last_report = now
identifier = album_id(root, folder)
cached = known.get(identifier)
if cached is not None:
paths = _audio_files(folder, extensions)
if paths and Fingerprint.of(paths) == cached[1]:
out[identifier] = cached
continue
scanned = scan_album(
folder,
root=root,
section_name=section_name,
section=section,
extensions=extensions,
cache=cache,
figure_kinds=figure_kinds,
)
if scanned is not None:
out[identifier] = scanned
_log.info("Library: %d albums under %s", len(out), root)
return out

View File

@@ -237,6 +237,22 @@ async def test_a_cover_file_is_found_and_used(config_dir: Path) -> None:
assert album.colors[0] != colors_from_id(album.id)[0]
async def test_an_episodes_sidecar_cover_wins_over_the_shows_shared_one(config_dir: Path) -> None:
from PIL import Image
# Downloaded by `podcast_feeds.py` when a feed has real per-episode art (see its
# `episode_cover_filename`) - same stem as the episode's audio file.
podcast = config_dir / "music" / "Kinderpodcasts" / "Wissen macht Ah"
Image.new("RGB", (32, 32), (10, 20, 30)).save(podcast / "cover.jpg")
Image.new("RGB", (32, 32), (200, 40, 30)).save(podcast / "20260101 - Neu.jpg")
library = await build(config_dir)
assert album_named(library, "Neu").cover == podcast / "20260101 - Neu.jpg"
# No sidecar for this one, so it still falls back to the show's shared cover.
assert album_named(library, "Alt").cover == podcast / "cover.jpg"
# --------------------------------------------------------------------------- cache
@@ -359,6 +375,28 @@ def test_album_ids_are_stable_and_path_derived(tmp_path: Path) -> None:
assert first != other
async def test_an_album_nested_under_an_extra_grouping_folder_is_still_found(
config_dir: Path,
) -> None:
"""The bug this fixed: an artist folder that groups its books one level deeper than
usual (an age-range folder, say) has no audio directly in it, so the scanner used to
treat it as an empty album and skip the whole artist rather than looking further
down for the actual album folders."""
for index in range(2):
write_track(
config_dir / "music" / "Hörbücher" / "Petzi" / "ab 5" / "Petzi und der Wal"
/ f"{index:02d} - teil.mp3",
title=f"Teil {index}",
album="Petzi und der Wal",
albumartist="Petzi",
)
library = await build(config_dir)
album = album_named(library, "Petzi und der Wal")
assert album.kind == "book"
assert len(album.tracks) == 2
# -------------------------------------------------------------------- figure kinds

View File

@@ -7,13 +7,18 @@ from pathlib import Path
import httpx2
import pytest
from musicmouse.library import podcast_feeds
from musicmouse.library.podcast_feeds import (
AudioExtractionError,
Episode,
_entry_enclosure,
download_episode,
episode_cover_filename,
episode_filename,
find_feed_shows,
missing_episodes,
parse_feed,
resolve_episode_cover,
sync_all_shows,
sync_show,
)
@@ -43,6 +48,46 @@ _FEED = b"""<?xml version="1.0"?>
</rss>
"""
_FEED_WITH_COVERS = b"""<?xml version="1.0"?>
<rss version="2.0" xmlns:itunes="http://www.itunes.com/dtds/podcast-1.0.dtd">
<channel>
<title>Cover Show</title>
<link>http://example.com/show</link>
<itunes:image href="http://example.com/channel.jpg" />
<item>
<title>Unique Cover</title>
<link>http://example.com/episodes/unique</link>
<pubDate>Wed, 01 Jan 2025 08:00:00 GMT</pubDate>
<itunes:image href="http://example.com/unique.jpg" />
<enclosure url="http://example.com/ep1.mp3" length="100" type="audio/mpeg" />
<guid>ep1</guid>
</item>
<item>
<title>Same As Channel</title>
<link>http://example.com/show</link>
<pubDate>Thu, 02 Jan 2025 08:00:00 GMT</pubDate>
<itunes:image href="http://example.com/channel.jpg" />
<enclosure url="http://example.com/ep2.mp3" length="100" type="audio/mpeg" />
<guid>ep2</guid>
</item>
</channel>
</rss>
"""
_VIDEO_FEED = b"""<?xml version="1.0"?>
<rss version="2.0">
<channel>
<title>Video Show</title>
<item>
<title>Video Only</title>
<pubDate>Wed, 01 Jan 2025 08:00:00 GMT</pubDate>
<enclosure url="http://example.com/ep1.mp4" length="100" type="video/mpeg" />
<guid>ep1</guid>
</item>
</channel>
</rss>
"""
def _root_with_show(
tmp_path: Path, *, feed_url: str | None = "http://example.com/feed.xml"
@@ -99,7 +144,19 @@ def test_episode_filename_sanitizes_illegal_characters_and_falls_back_on_type()
name = episode_filename(
episode.published, episode.title, episode.enclosure_type, episode.enclosure_url
)
assert name == "20250103 - Episode Two SpecialChars.mp3"
# Illegal characters become "_" rather than being dropped, matching the
# convention every episode already on disk (downloaded by hand via the
# `podcast-dl` CLI) was named with.
assert name == "20250103 - Episode Two_ Special_Chars_.mp3"
def test_episode_filename_matches_the_podcast_dl_convention_on_a_real_title() -> None:
# Pinned against a real file already in the library: `podcast-dl` replaced the
# "?" with "_" rather than dropping it, so ours must too or the whole back
# catalog would look "new" the moment a feed marker is added for that show.
published = parse_feed(_FEED)[0].published
name = episode_filename(published, "Wie funktioniert Bogenschießen?", "audio/mpeg", "x.mp3")
assert name == "20250101 - Wie funktioniert Bogenschießen_.mp3"
def test_missing_episodes_skips_files_already_on_disk(tmp_path: Path) -> None:
@@ -151,7 +208,7 @@ async def test_sync_show_downloads_new_episodes_and_is_a_noop_on_rerun(tmp_path:
assert changed_first is True
assert changed_second is False
assert (folder / "20250101 - Episode One.mp3").exists()
assert (folder / "20250103 - Episode Two SpecialChars.mp3").exists()
assert (folder / "20250103 - Episode Two_ Special_Chars_.mp3").exists()
@pytest.mark.asyncio
@@ -211,3 +268,245 @@ async def test_sync_all_shows_keeps_going_past_one_broken_show(tmp_path: Path) -
assert changed is True
assert list(good.glob("*.mp3"))
assert list(bad.glob("*.mp3")) == []
# ------------------------------------------------------------------- per-episode art
def test_parse_feed_extracts_cover_and_link_only_when_distinct_from_the_channel() -> None:
unique, same = parse_feed(_FEED_WITH_COVERS)
assert unique.cover_url == "http://example.com/unique.jpg"
assert unique.link == "http://example.com/episodes/unique"
# This item's image/link both just repeat the channel's own - not real per-episode
# art, so both come out None rather than a false positive.
assert same.cover_url is None
assert same.link is None
def test_episode_cover_filename_shares_the_audio_stem_and_takes_the_url_extension() -> None:
episode = parse_feed(_FEED_WITH_COVERS)[0]
name = episode_cover_filename(episode.published, episode.title, episode.cover_url)
assert name == "20250101 - Unique Cover.jpg"
@pytest.mark.asyncio
async def test_resolve_episode_cover_prefers_the_feed_image_without_a_request() -> None:
episode = parse_feed(_FEED_WITH_COVERS)[0]
def handler(request: httpx2.Request) -> httpx2.Response:
raise AssertionError("must not fetch the episode page when cover_url is already set")
async with httpx2.AsyncClient(transport=httpx2.MockTransport(handler)) as client:
cover = await resolve_episode_cover(client, episode)
assert cover == "http://example.com/unique.jpg"
@pytest.mark.asyncio
async def test_resolve_episode_cover_falls_back_to_og_image_on_the_linked_page() -> None:
episode = Episode(
title="No Feed Image",
published=parse_feed(_FEED)[0].published,
enclosure_url="http://example.com/ep.mp3",
enclosure_type="audio/mpeg",
link="http://example.com/episode-page",
)
def handler(request: httpx2.Request) -> httpx2.Response:
return httpx2.Response(
200,
content=b'<html><head>'
b'<meta property="og:image" content="http://example.com/scraped.jpg"/>'
b"</head></html>",
)
async with httpx2.AsyncClient(transport=httpx2.MockTransport(handler)) as client:
cover = await resolve_episode_cover(client, episode)
assert cover == "http://example.com/scraped.jpg"
@pytest.mark.asyncio
async def test_resolve_episode_cover_returns_none_when_the_link_is_unreachable() -> None:
episode = Episode(
title="Broken Link",
published=parse_feed(_FEED)[0].published,
enclosure_url="http://example.com/ep.mp3",
enclosure_type="audio/mpeg",
link="http://example.com/missing",
)
async with httpx2.AsyncClient(
transport=httpx2.MockTransport(lambda _: httpx2.Response(404))
) as client:
cover = await resolve_episode_cover(client, episode)
assert cover is None
@pytest.mark.asyncio
async def test_resolve_episode_cover_returns_none_without_a_cover_or_link() -> None:
episode = Episode(
title="Nothing",
published=parse_feed(_FEED)[0].published,
enclosure_url="http://example.com/ep.mp3",
enclosure_type="audio/mpeg",
)
async def handler(request: httpx2.Request) -> httpx2.Response: # pragma: no cover
raise AssertionError("must not make a request with neither a cover nor a link")
async with httpx2.AsyncClient(transport=httpx2.MockTransport(handler)) as client:
cover = await resolve_episode_cover(client, episode)
assert cover is None
@pytest.mark.asyncio
async def test_sync_show_downloads_the_episode_cover_alongside_the_audio(tmp_path: Path) -> None:
folder = tmp_path / "show"
folder.mkdir()
def handler(request: httpx2.Request) -> httpx2.Response:
url = str(request.url)
if url == "http://example.com/feed.xml":
return httpx2.Response(200, content=_FEED_WITH_COVERS)
if url == "http://example.com/unique.jpg":
return httpx2.Response(200, content=b"cover-bytes")
return httpx2.Response(200, content=b"audio-bytes")
async with httpx2.AsyncClient(transport=httpx2.MockTransport(handler)) as client:
changed = await sync_show(client, folder, "http://example.com/feed.xml")
assert changed is True
assert (folder / "20250101 - Unique Cover.jpg").read_bytes() == b"cover-bytes"
# No cover file for the episode whose image just repeats the channel's.
assert not (folder / "20250102 - Same As Channel.jpg").exists()
assert list(folder.glob(".downloading-*.tmp")) == []
@pytest.mark.asyncio
async def test_sync_show_survives_a_failed_cover_download(tmp_path: Path) -> None:
folder = tmp_path / "show"
folder.mkdir()
def handler(request: httpx2.Request) -> httpx2.Response:
url = str(request.url)
if url == "http://example.com/feed.xml":
return httpx2.Response(200, content=_FEED_WITH_COVERS)
if url == "http://example.com/unique.jpg":
return httpx2.Response(404)
return httpx2.Response(200, content=b"audio-bytes")
async with httpx2.AsyncClient(transport=httpx2.MockTransport(handler)) as client:
changed = await sync_show(client, folder, "http://example.com/feed.xml")
# The episode itself still downloaded fine - only its cover failed.
assert changed is True
assert (folder / "20250101 - Unique Cover.mp3").exists()
assert not (folder / "20250101 - Unique Cover.jpg").exists()
# ---------------------------------------------------------------------- video shows
def test_entry_enclosure_accepts_video_when_no_audio_is_offered() -> None:
entry = {"enclosures": [{"href": "http://example.com/ep.mp4", "type": "video/mpeg"}]}
assert _entry_enclosure(entry) == ("http://example.com/ep.mp4", "video/mpeg")
def test_entry_enclosure_prefers_audio_over_video_when_both_are_offered() -> None:
entry = {
"enclosures": [
{"href": "http://example.com/ep.mp4", "type": "video/mpeg"},
{"href": "http://example.com/ep.mp3", "type": "audio/mpeg"},
]
}
assert _entry_enclosure(entry) == ("http://example.com/ep.mp3", "audio/mpeg")
def test_parse_feed_keeps_a_video_only_episode() -> None:
episodes = parse_feed(_VIDEO_FEED)
assert episodes[0].enclosure_url == "http://example.com/ep1.mp4"
assert episodes[0].enclosure_type == "video/mpeg"
@pytest.mark.asyncio
async def test_download_episode_raises_when_ffmpeg_is_not_installed(
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
) -> None:
monkeypatch.setattr(podcast_feeds.shutil, "which", lambda _: None)
folder = tmp_path / "show"
folder.mkdir()
episode = parse_feed(_VIDEO_FEED)[0]
async with httpx2.AsyncClient(
transport=httpx2.MockTransport(lambda _: httpx2.Response(200))
) as client:
with pytest.raises(AudioExtractionError):
await download_episode(client, folder, episode, "20250101 - Video Only.mp3")
assert list(folder.glob(".downloading-*.tmp")) == []
@pytest.mark.asyncio
async def test_download_episode_raises_when_ffmpeg_exits_nonzero(
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
) -> None:
monkeypatch.setattr(podcast_feeds.shutil, "which", lambda _: "/usr/bin/ffmpeg")
class _FailingProcess:
returncode = 1
async def communicate(self) -> tuple[bytes, bytes]:
return b"", b"boom"
async def fake_exec(*args: object, **kwargs: object) -> _FailingProcess:
return _FailingProcess()
monkeypatch.setattr(podcast_feeds.asyncio, "create_subprocess_exec", fake_exec)
folder = tmp_path / "show"
folder.mkdir()
episode = parse_feed(_VIDEO_FEED)[0]
async with httpx2.AsyncClient(
transport=httpx2.MockTransport(lambda _: httpx2.Response(200))
) as client:
with pytest.raises(AudioExtractionError, match="boom"):
await download_episode(client, folder, episode, "20250101 - Video Only.mp3")
assert list(folder.glob(".downloading-*.tmp")) == []
@pytest.mark.asyncio
async def test_download_episode_extracts_audio_via_ffmpeg(
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
) -> None:
monkeypatch.setattr(podcast_feeds.shutil, "which", lambda _: "/usr/bin/ffmpeg")
class _SucceedingProcess:
returncode = 0
async def communicate(self) -> tuple[bytes, bytes]:
return b"", b""
async def fake_exec(*args: object, **kwargs: object) -> _SucceedingProcess:
# ffmpeg's destination path is the last positional argument; write to it here
# to stand in for what the real binary would have produced.
Path(str(args[-1])).write_bytes(b"extracted-audio")
return _SucceedingProcess()
monkeypatch.setattr(podcast_feeds.asyncio, "create_subprocess_exec", fake_exec)
folder = tmp_path / "show"
folder.mkdir()
episode = parse_feed(_VIDEO_FEED)[0]
async with httpx2.AsyncClient(
transport=httpx2.MockTransport(lambda _: httpx2.Response(200))
) as client:
await download_episode(client, folder, episode, "20250101 - Video Only.mp3")
assert (folder / "20250101 - Video Only.mp3").read_bytes() == b"extracted-audio"
assert list(folder.glob(".downloading-*.tmp")) == []