Add per-episode podcast cover art and fix nested album grouping
- podcast_feeds.py: fetch per-episode cover art from a feed's itunes:image (when it's genuinely distinct from the channel image) or, failing that, from the og:image on the episode's own linked page; extract audio from video-only enclosures via ffmpeg; match podcast-dl's filename convention (illegal characters become "_" instead of being dropped) so enabling feed.txt on an already-downloaded show doesn't re-download its back catalog - scanner.py: _cover_for_episode now checks for a same-stem sidecar cover image before falling back to embedded ID3 art and the shared folder cover - scanner.py/__init__.py: fixed a bug where an album folder nested one level deeper than usual (an age-range grouping folder, say) was mistaken for an empty album and skipped; added periodic progress logging for long scans and analysis passes Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
@@ -11,6 +11,7 @@ import asyncio
|
||||
import contextlib
|
||||
import logging
|
||||
import os
|
||||
import time
|
||||
from collections.abc import Awaitable, Callable, Collection, Mapping
|
||||
from dataclasses import replace
|
||||
from pathlib import Path
|
||||
@@ -62,6 +63,11 @@ _BUSY_POLL_SECONDS: Final = 2.0
|
||||
#: browser tabs rather than only after the whole thing finishes.
|
||||
_PUBLISH_BATCH_SIZE: Final = 25
|
||||
|
||||
#: How often a long analysis pass reports where it is, so a first-time run over a large
|
||||
#: library - minutes of DSP per track - doesn't sit silent with nothing on the console
|
||||
#: to say it is still going.
|
||||
_PROGRESS_INTERVAL_SECONDS: Final = 5.0
|
||||
|
||||
|
||||
def _analyze_one(
|
||||
analyzer: Analyzer, path: Path
|
||||
@@ -273,6 +279,7 @@ class MusicLibrary:
|
||||
return 0
|
||||
|
||||
done = 0
|
||||
last_report = time.monotonic()
|
||||
for album in self.albums:
|
||||
if album.kind not in kinds:
|
||||
continue
|
||||
@@ -283,6 +290,10 @@ class MusicLibrary:
|
||||
cached = self.cache.load_analysis(key)
|
||||
if cached is not None and cached.version >= analyzer.version:
|
||||
continue
|
||||
now = time.monotonic()
|
||||
if now - last_report >= _PROGRESS_INTERVAL_SECONDS:
|
||||
_log.info("Analyzing library: %d tracks done so far, now on %s", done, track.path)
|
||||
last_report = now
|
||||
try:
|
||||
analysis, grid, curve = await asyncio.to_thread(
|
||||
_analyze_one, analyzer, track.path
|
||||
|
||||
@@ -7,15 +7,21 @@ its shape entirely off the folder tree (see :mod:`musicmouse.library.sections`).
|
||||
|
||||
Downloaded episodes land in the show folder using the exact ``YYYYMMDD - Title.ext``
|
||||
convention ``sections.py`` already documents for ``Kinderpodcasts``, so a freshly
|
||||
downloaded episode sorts and scans exactly like one a person dropped in by hand.
|
||||
downloaded episode sorts and scans exactly like one a person dropped in by hand. When
|
||||
a feed offers real per-episode artwork, it's saved alongside as a same-named sidecar
|
||||
image, which ``scanner.py``'s ``_cover_for_episode`` picks up automatically. A
|
||||
video-only enclosure (some shows publish no audio feed at all) is transcoded to audio
|
||||
via the ``ffmpeg`` binary, which must be on ``PATH`` for those shows to sync.
|
||||
Nothing here raises on bad input - an unreachable feed or a broken enclosure is logged
|
||||
and skipped, not a crash, mirroring ``scanner.py``'s own rule.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
import re
|
||||
import shutil
|
||||
from dataclasses import dataclass
|
||||
from datetime import UTC, datetime
|
||||
from pathlib import Path
|
||||
@@ -31,12 +37,16 @@ _log = logging.getLogger(__name__)
|
||||
|
||||
__all__ = [
|
||||
"FEED_MARKER_NAME",
|
||||
"AudioExtractionError",
|
||||
"Episode",
|
||||
"download_episode",
|
||||
"download_episode_cover",
|
||||
"episode_cover_filename",
|
||||
"episode_filename",
|
||||
"find_feed_shows",
|
||||
"missing_episodes",
|
||||
"parse_feed",
|
||||
"resolve_episode_cover",
|
||||
"sync_all_shows",
|
||||
"sync_show",
|
||||
]
|
||||
@@ -59,11 +69,34 @@ _EXTENSION_BY_TYPE: Final[dict[str, str]] = {
|
||||
_DEFAULT_EXTENSION: Final = ".mp3"
|
||||
_KNOWN_EXTENSIONS: Final = frozenset({".mp3", ".m4a", ".aac", ".ogg", ".opus", ".wav", ".flac"})
|
||||
|
||||
#: Cover URL suffix -> file extension, for the per-episode sidecar image.
|
||||
_IMAGE_EXTENSIONS: Final[frozenset[str]] = frozenset({".jpg", ".jpeg", ".png", ".webp"})
|
||||
_DEFAULT_IMAGE_EXTENSION: Final = ".jpg"
|
||||
|
||||
#: Characters illegal (or awkward) in a filename, plus the path separators themselves.
|
||||
#: Replaced with "_" rather than dropped, matching the convention every episode
|
||||
#: already downloaded by hand (via the ``podcast-dl`` CLI) was named with - a
|
||||
#: mismatch here would make every one of them look "new" to `missing_episodes`.
|
||||
_ILLEGAL_FILENAME_CHARS: Final = re.compile(r'[\\/:*?"<>|]')
|
||||
_MAX_TITLE_LENGTH: Final = 120
|
||||
|
||||
_HTTP_TIMEOUT: Final = 30.0
|
||||
#: `og:image` scraping is a fallback for a page we don't control; kept short so one
|
||||
#: slow or hanging host can't stall a whole sync pass.
|
||||
_OG_IMAGE_TIMEOUT: Final = 15.0
|
||||
_OG_IMAGE_RE: Final = re.compile(
|
||||
r'<meta[^>]+property=["\']og:image["\'][^>]*content=["\']([^"\']+)["\']'
|
||||
r'|<meta[^>]+content=["\']([^"\']+)["\'][^>]*property=["\']og:image["\']',
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
#: How long a video enclosure is given to download-and-transcode before it's treated
|
||||
#: as failed - generous, since this runs on a Pi and a long video can take a while.
|
||||
_FFMPEG_TIMEOUT: Final = 600.0
|
||||
|
||||
|
||||
class AudioExtractionError(Exception):
|
||||
"""Raised when a video enclosure could not be turned into an audio file."""
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
@@ -72,6 +105,15 @@ class Episode:
|
||||
published: datetime
|
||||
enclosure_url: str
|
||||
enclosure_type: str
|
||||
#: The feed's own per-episode artwork, when it has one distinct from the show's
|
||||
#: overall cover. ``None`` when the feed has no image at all, or (as with GEOlino)
|
||||
#: every item merely repeats the channel's own image.
|
||||
cover_url: str | None = None
|
||||
#: The episode's own page, when the feed links to one distinct from the show's
|
||||
#: general page - the fallback route to real per-episode art for a feed (like
|
||||
#: Wissen macht Ah) that has no per-item image of its own, via that page's
|
||||
#: `og:image`. ``None`` when the feed has no such per-episode page.
|
||||
link: str | None = None
|
||||
|
||||
|
||||
def find_feed_shows(root: Path) -> list[tuple[str, Path, str]]:
|
||||
@@ -110,12 +152,19 @@ def _read_feed_url(marker: Path) -> str | None:
|
||||
|
||||
|
||||
def parse_feed(content: bytes) -> list[Episode]:
|
||||
"""Every episode in a feed that has both a publish date and an audio enclosure.
|
||||
"""Every episode in a feed that has both a publish date and a playable enclosure.
|
||||
|
||||
Anything else is skipped and logged rather than raising - a malformed or unusual
|
||||
entry in one feed must never stop the episodes around it from being picked up.
|
||||
"""
|
||||
parsed = feedparser.parse(content)
|
||||
# feedparser exposes an `itunes:image` identically at the channel and item level,
|
||||
# under the same "image" key an ordinary RSS `<image>` uses - there is no separate
|
||||
# `itunes_image` field. Read the channel's own image/link once so each entry can
|
||||
# tell whether it has something genuinely its own, or is just repeating them
|
||||
# (as GEOlino's per-item `itunes:image` does).
|
||||
channel_image = parsed.feed.get("image", {}).get("href")
|
||||
channel_link = parsed.feed.get("link")
|
||||
episodes: list[Episode] = []
|
||||
for entry in parsed.entries:
|
||||
published = _entry_published(entry)
|
||||
@@ -124,15 +173,26 @@ def parse_feed(content: bytes) -> list[Episode]:
|
||||
continue
|
||||
enclosure = _entry_enclosure(entry)
|
||||
if enclosure is None:
|
||||
_log.debug("Feed entry %r has no audio enclosure; skipping", entry.get("title"))
|
||||
_log.debug(
|
||||
"Feed entry %r has no audio or video enclosure; skipping", entry.get("title")
|
||||
)
|
||||
continue
|
||||
url, enclosure_type = enclosure
|
||||
|
||||
item_image = entry.get("image", {}).get("href")
|
||||
cover_url = item_image if item_image and item_image != channel_image else None
|
||||
|
||||
item_link = entry.get("link")
|
||||
link = item_link if item_link and item_link != channel_link else None
|
||||
|
||||
episodes.append(
|
||||
Episode(
|
||||
title=entry.get("title") or url,
|
||||
published=published,
|
||||
enclosure_url=url,
|
||||
enclosure_type=enclosure_type,
|
||||
cover_url=cover_url,
|
||||
link=link,
|
||||
)
|
||||
)
|
||||
return episodes
|
||||
@@ -147,12 +207,32 @@ def _entry_published(entry: Any) -> datetime | None:
|
||||
|
||||
|
||||
def _entry_enclosure(entry: Any) -> tuple[str, str] | None:
|
||||
"""The enclosure to download for one entry, preferring audio when both are offered.
|
||||
|
||||
A video enclosure (some shows, like Wissen macht Ah, publish no audio version at
|
||||
all) is still returned rather than dropped - :func:`download_episode` turns it
|
||||
into audio via ffmpeg. It's just the least preferred of the three: an audio
|
||||
enclosure, or one with no declared type at all (assumed audio), both win outright.
|
||||
"""
|
||||
video: tuple[str, str] | None = None
|
||||
for enclosure in entry.get("enclosures", []):
|
||||
url = enclosure.get("href") or enclosure.get("url")
|
||||
enclosure_type = enclosure.get("type") or ""
|
||||
if url and (enclosure_type.startswith("audio/") or not enclosure_type):
|
||||
if not url:
|
||||
continue
|
||||
if enclosure_type.startswith("audio/") or not enclosure_type:
|
||||
return str(url), str(enclosure_type)
|
||||
return None
|
||||
if video is None and enclosure_type.startswith("video/"):
|
||||
video = (str(url), str(enclosure_type))
|
||||
return video
|
||||
|
||||
|
||||
def _episode_stem(published: datetime, title: str) -> str:
|
||||
"""``YYYYMMDD - Title``, with no extension yet - shared by the audio filename and
|
||||
its sidecar cover image, so the two always line up."""
|
||||
sanitized = _ILLEGAL_FILENAME_CHARS.sub("_", title).strip().strip(".")
|
||||
sanitized = " ".join(sanitized.split())[:_MAX_TITLE_LENGTH] or "Episode"
|
||||
return f"{published:%Y%m%d} - {sanitized}"
|
||||
|
||||
|
||||
def episode_filename(
|
||||
@@ -160,9 +240,13 @@ def episode_filename(
|
||||
) -> str:
|
||||
"""``YYYYMMDD - Title.ext``, matching the convention every hand-placed episode
|
||||
already follows (see ``sections.py``'s module comment on ``Kinderpodcasts``)."""
|
||||
sanitized = _ILLEGAL_FILENAME_CHARS.sub("", title).strip().strip(".")
|
||||
sanitized = " ".join(sanitized.split())[:_MAX_TITLE_LENGTH] or "Episode"
|
||||
return f"{published:%Y%m%d} - {sanitized}{_extension_for(enclosure_type, enclosure_url)}"
|
||||
return f"{_episode_stem(published, title)}{_extension_for(enclosure_type, enclosure_url)}"
|
||||
|
||||
|
||||
def episode_cover_filename(published: datetime, title: str, cover_url: str) -> str:
|
||||
"""The sidecar image filename for an episode's cover, sharing its stem so the
|
||||
scanner (``scanner.py``'s ``_cover_for_episode``) can find it next to the audio."""
|
||||
return f"{_episode_stem(published, title)}{_image_extension_for(cover_url)}"
|
||||
|
||||
|
||||
def _extension_for(enclosure_type: str, url: str) -> str:
|
||||
@@ -172,6 +256,11 @@ def _extension_for(enclosure_type: str, url: str) -> str:
|
||||
return _EXTENSION_BY_TYPE.get(enclosure_type, _DEFAULT_EXTENSION)
|
||||
|
||||
|
||||
def _image_extension_for(url: str) -> str:
|
||||
suffix = Path(urlsplit(url).path).suffix.lower()
|
||||
return suffix if suffix in _IMAGE_EXTENSIONS else _DEFAULT_IMAGE_EXTENSION
|
||||
|
||||
|
||||
def missing_episodes(folder: Path, episodes: list[Episode]) -> list[tuple[Episode, str]]:
|
||||
"""Episodes whose target filename isn't already on disk, paired with that filename.
|
||||
|
||||
@@ -192,26 +281,117 @@ def missing_episodes(folder: Path, episodes: list[Episode]) -> list[tuple[Episod
|
||||
async def download_episode(
|
||||
client: httpx2.AsyncClient, folder: Path, episode: Episode, filename: str
|
||||
) -> None:
|
||||
"""Stream an episode to ``folder / filename`` via a dotfile temp path.
|
||||
"""Get an episode to ``folder / filename`` via a dotfile temp path.
|
||||
|
||||
A half-written file must never look like a track: the scanner already skips
|
||||
dotfiles for exactly this reason (see ``scanner.py``'s ``_audio_files``), so the
|
||||
rename to the real name only happens once the download is complete.
|
||||
rename to the real name only happens once the download is complete. A video
|
||||
enclosure is routed through ffmpeg instead of being streamed as-is - see
|
||||
:func:`_extract_audio`.
|
||||
"""
|
||||
temp_path = folder / f".downloading-{filename}.tmp"
|
||||
try:
|
||||
async with client.stream(
|
||||
"GET", episode.enclosure_url, timeout=_HTTP_TIMEOUT
|
||||
) as response:
|
||||
response.raise_for_status()
|
||||
with temp_path.open("wb") as handle:
|
||||
async for chunk in response.aiter_bytes():
|
||||
handle.write(chunk)
|
||||
if episode.enclosure_type.startswith("video/"):
|
||||
await _extract_audio(episode.enclosure_url, temp_path)
|
||||
else:
|
||||
async with client.stream(
|
||||
"GET", episode.enclosure_url, timeout=_HTTP_TIMEOUT
|
||||
) as response:
|
||||
response.raise_for_status()
|
||||
with temp_path.open("wb") as handle:
|
||||
async for chunk in response.aiter_bytes():
|
||||
handle.write(chunk)
|
||||
temp_path.replace(folder / filename)
|
||||
finally:
|
||||
temp_path.unlink(missing_ok=True)
|
||||
|
||||
|
||||
async def _extract_audio(source_url: str, dest: Path) -> None:
|
||||
"""Pull ``source_url`` (a video enclosure) through ffmpeg, writing just its audio
|
||||
track to ``dest``. ffmpeg fetches the URL itself, so the video is never stored.
|
||||
|
||||
Raises :class:`AudioExtractionError` rather than a bare ``OSError`` or timeout, so
|
||||
callers can tell a missing/failing ffmpeg apart from an ordinary network error -
|
||||
but either way this is meant to be logged and skipped, not fatal.
|
||||
"""
|
||||
if shutil.which("ffmpeg") is None:
|
||||
raise AudioExtractionError(
|
||||
"ffmpeg is not installed; cannot extract audio from a video enclosure"
|
||||
)
|
||||
process = await asyncio.create_subprocess_exec(
|
||||
"ffmpeg",
|
||||
"-y",
|
||||
"-i",
|
||||
source_url,
|
||||
"-vn",
|
||||
"-acodec",
|
||||
"libmp3lame",
|
||||
"-q:a",
|
||||
"2",
|
||||
str(dest),
|
||||
stdout=asyncio.subprocess.DEVNULL,
|
||||
stderr=asyncio.subprocess.PIPE,
|
||||
)
|
||||
try:
|
||||
_, stderr = await asyncio.wait_for(process.communicate(), timeout=_FFMPEG_TIMEOUT)
|
||||
except TimeoutError:
|
||||
process.kill()
|
||||
await process.wait()
|
||||
raise AudioExtractionError(f"ffmpeg timed out extracting audio from {source_url}") from None
|
||||
if process.returncode != 0:
|
||||
raise AudioExtractionError(
|
||||
f"ffmpeg exited {process.returncode} extracting audio from {source_url}: "
|
||||
f"{stderr.decode(errors='replace')[-500:]}"
|
||||
)
|
||||
|
||||
|
||||
async def download_episode_cover(
|
||||
client: httpx2.AsyncClient, folder: Path, cover_url: str, filename: str
|
||||
) -> None:
|
||||
"""Best-effort sidecar cover image fetch, atomic like :func:`download_episode`.
|
||||
|
||||
Cover art is small enough that streaming it in chunks isn't worth the extra code.
|
||||
"""
|
||||
temp_path = folder / f".downloading-{filename}.tmp"
|
||||
try:
|
||||
response = await client.get(cover_url, timeout=_HTTP_TIMEOUT, follow_redirects=True)
|
||||
response.raise_for_status()
|
||||
temp_path.write_bytes(response.content)
|
||||
temp_path.replace(folder / filename)
|
||||
finally:
|
||||
temp_path.unlink(missing_ok=True)
|
||||
|
||||
|
||||
def _extract_og_image(html: str) -> str | None:
|
||||
match = _OG_IMAGE_RE.search(html)
|
||||
if match is None:
|
||||
return None
|
||||
return match.group(1) or match.group(2)
|
||||
|
||||
|
||||
async def resolve_episode_cover(client: httpx2.AsyncClient, episode: Episode) -> str | None:
|
||||
"""The best cover URL available for ``episode``, or ``None`` when there isn't one.
|
||||
|
||||
Prefers the feed's own per-episode image. Failing that, when the feed links to a
|
||||
page of the episode's own (as Wissen macht Ah does, despite having no per-item
|
||||
image tag), falls back to that page's ``og:image`` - one targeted fetch of a URL
|
||||
the feed itself provided, not a scrape of a list page. Never raises: an
|
||||
unreachable page, a redirect loop, or a page with no such tag all just mean no
|
||||
cover was found here.
|
||||
"""
|
||||
if episode.cover_url:
|
||||
return episode.cover_url
|
||||
if not episode.link:
|
||||
return None
|
||||
try:
|
||||
response = await client.get(episode.link, timeout=_OG_IMAGE_TIMEOUT, follow_redirects=True)
|
||||
response.raise_for_status()
|
||||
except httpx2.HTTPError as exc:
|
||||
_log.debug("Could not fetch %s for its og:image: %s", episode.link, exc)
|
||||
return None
|
||||
return _extract_og_image(response.text)
|
||||
|
||||
|
||||
async def sync_show(client: httpx2.AsyncClient, folder: Path, feed_url: str) -> bool:
|
||||
"""Download every episode in ``feed_url`` that ``folder`` doesn't have yet.
|
||||
|
||||
@@ -231,13 +411,27 @@ async def sync_show(client: httpx2.AsyncClient, folder: Path, feed_url: str) ->
|
||||
for episode, filename in pending:
|
||||
try:
|
||||
await download_episode(client, folder, episode, filename)
|
||||
except (httpx2.HTTPError, OSError) as exc:
|
||||
except (httpx2.HTTPError, OSError, AudioExtractionError) as exc:
|
||||
_log.warning(
|
||||
"Could not download episode %r for %s: %s", episode.title, folder.name, exc
|
||||
)
|
||||
continue
|
||||
_log.info("Downloaded new episode %r for %s", episode.title, folder.name)
|
||||
changed = True
|
||||
|
||||
cover_url = await resolve_episode_cover(client, episode)
|
||||
if cover_url:
|
||||
cover_filename = episode_cover_filename(episode.published, episode.title, cover_url)
|
||||
if not (folder / cover_filename).exists():
|
||||
try:
|
||||
await download_episode_cover(client, folder, cover_url, cover_filename)
|
||||
except (httpx2.HTTPError, OSError) as exc:
|
||||
_log.warning(
|
||||
"Could not download cover for episode %r in %s: %s",
|
||||
episode.title,
|
||||
folder.name,
|
||||
exc,
|
||||
)
|
||||
return changed
|
||||
|
||||
|
||||
|
||||
@@ -9,8 +9,9 @@ input - an unreadable file loses its metadata, not the boot.
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import time
|
||||
from collections import Counter
|
||||
from collections.abc import Mapping
|
||||
from collections.abc import Iterator, Mapping
|
||||
from pathlib import Path
|
||||
|
||||
from musicmouse.library.cache import Fingerprint, LibraryCache
|
||||
@@ -25,6 +26,33 @@ __all__ = ["scan_library"]
|
||||
#: Checked in order for a cover sitting next to the audio.
|
||||
_COVER_NAMES = ("cover.jpg", "cover.jpeg", "cover.png", "folder.jpg")
|
||||
|
||||
#: Extensions tried for an episode's own same-stem sidecar cover image (see
|
||||
#: :func:`_cover_for_episode`), matching what `podcast_feeds.py`'s
|
||||
#: ``episode_cover_filename`` can produce.
|
||||
_SIDECAR_COVER_EXTENSIONS = (".jpg", ".jpeg", ".png", ".webp")
|
||||
|
||||
#: How often a long scan reports where it is, so a library of thousands of albums
|
||||
#: doesn't sit silent for minutes with nothing on the console to say it is still going.
|
||||
_PROGRESS_INTERVAL_SECONDS = 5.0
|
||||
|
||||
|
||||
def _album_folders(folder: Path, extensions: frozenset[str]) -> Iterator[Path]:
|
||||
"""Every leaf album folder under ``folder``, however deep it is nested.
|
||||
|
||||
A folder that holds audio files directly *is* an album - an artist who groups
|
||||
their books under an extra "ab 3" / "ab 5" age-range folder, or a series folder
|
||||
that groups its episodes one directory further down than usual, still bottoms out
|
||||
here without needing to be special-cased. Only descended into when a folder holds
|
||||
no audio of its own, so an ordinary album folder is never mistaken for a grouping
|
||||
folder just because it also happens to contain subfolders.
|
||||
"""
|
||||
if _audio_files(folder, extensions):
|
||||
yield folder
|
||||
return
|
||||
for sub in sorted(folder.iterdir(), key=lambda path: path.name):
|
||||
if sub.is_dir() and not sub.name.startswith("."):
|
||||
yield from _album_folders(sub, extensions)
|
||||
|
||||
|
||||
def _audio_files(folder: Path, extensions: frozenset[str]) -> list[Path]:
|
||||
"""The playable files in ``folder``, alphabetically.
|
||||
@@ -106,9 +134,16 @@ def _cover_for(
|
||||
def _cover_for_episode(
|
||||
folder: Path, path: Path, identifier: str, cache: LibraryCache
|
||||
) -> tuple[Path | None, bytes | None]:
|
||||
"""The reverse priority from :func:`_cover_for`: with one album per episode, the
|
||||
episode's own embedded art is the more specific and more correct source. The
|
||||
folder's shared cover is the fallback for an episode whose file carries none."""
|
||||
"""The reverse priority from :func:`_cover_for`: with one album per episode, art
|
||||
specific to this one episode wins over the folder's shared cover. A same-stem
|
||||
sidecar image (what ``podcast_feeds.py`` saves when a feed has real per-episode
|
||||
art) is tried first - it's a plain file stat, cheaper than reading tags - then the
|
||||
episode's own embedded art, then the folder's shared cover as the last resort for
|
||||
an episode with neither."""
|
||||
for extension in _SIDECAR_COVER_EXTENSIONS:
|
||||
candidate = path.with_suffix(extension)
|
||||
if candidate.is_file():
|
||||
return candidate, candidate.read_bytes()
|
||||
art = _embedded_art(path)
|
||||
if art is not None:
|
||||
return cache.store_cover(identifier, art), art
|
||||
@@ -262,6 +297,7 @@ def scan_library(
|
||||
cache.prepare()
|
||||
known = known or {}
|
||||
out: dict[str, tuple[Album, Fingerprint]] = {}
|
||||
last_report = time.monotonic()
|
||||
|
||||
for section_name, section in SECTIONS.items():
|
||||
section_root = root / section_name
|
||||
@@ -269,16 +305,17 @@ def scan_library(
|
||||
_log.warning("Library section %r has no folder at %s", section_name, section_root)
|
||||
continue
|
||||
|
||||
for folder in sorted(section_root.iterdir(), key=lambda path: path.name):
|
||||
if not folder.is_dir() or folder.name.startswith("."):
|
||||
for top in sorted(section_root.iterdir(), key=lambda path: path.name):
|
||||
if not top.is_dir() or top.name.startswith("."):
|
||||
continue
|
||||
|
||||
if section.album_unit == "episode":
|
||||
# Fingerprinted per episode inside scan_episodes; the whole-folder
|
||||
# shortcut below does not apply since one folder yields many albums.
|
||||
# shortcut below does not apply since one folder yields many albums,
|
||||
# and an episode show is never nested deeper than this either.
|
||||
out.update(
|
||||
scan_episodes(
|
||||
folder,
|
||||
top,
|
||||
root=root,
|
||||
section_name=section_name,
|
||||
section=section,
|
||||
@@ -289,24 +326,34 @@ def scan_library(
|
||||
)
|
||||
continue
|
||||
|
||||
identifier = album_id(root, folder)
|
||||
cached = known.get(identifier)
|
||||
if cached is not None:
|
||||
paths = _audio_files(folder, extensions)
|
||||
if paths and Fingerprint.of(paths) == cached[1]:
|
||||
out[identifier] = cached
|
||||
continue
|
||||
scanned = scan_album(
|
||||
folder,
|
||||
root=root,
|
||||
section_name=section_name,
|
||||
section=section,
|
||||
extensions=extensions,
|
||||
cache=cache,
|
||||
figure_kinds=figure_kinds,
|
||||
)
|
||||
if scanned is not None:
|
||||
out[identifier] = scanned
|
||||
for folder in _album_folders(top, extensions):
|
||||
now = time.monotonic()
|
||||
if now - last_report >= _PROGRESS_INTERVAL_SECONDS:
|
||||
_log.info(
|
||||
"Scanning library: %d albums found so far, now in %s",
|
||||
len(out),
|
||||
folder.relative_to(root),
|
||||
)
|
||||
last_report = now
|
||||
|
||||
identifier = album_id(root, folder)
|
||||
cached = known.get(identifier)
|
||||
if cached is not None:
|
||||
paths = _audio_files(folder, extensions)
|
||||
if paths and Fingerprint.of(paths) == cached[1]:
|
||||
out[identifier] = cached
|
||||
continue
|
||||
scanned = scan_album(
|
||||
folder,
|
||||
root=root,
|
||||
section_name=section_name,
|
||||
section=section,
|
||||
extensions=extensions,
|
||||
cache=cache,
|
||||
figure_kinds=figure_kinds,
|
||||
)
|
||||
if scanned is not None:
|
||||
out[identifier] = scanned
|
||||
|
||||
_log.info("Library: %d albums under %s", len(out), root)
|
||||
return out
|
||||
|
||||
@@ -237,6 +237,22 @@ async def test_a_cover_file_is_found_and_used(config_dir: Path) -> None:
|
||||
assert album.colors[0] != colors_from_id(album.id)[0]
|
||||
|
||||
|
||||
async def test_an_episodes_sidecar_cover_wins_over_the_shows_shared_one(config_dir: Path) -> None:
|
||||
from PIL import Image
|
||||
|
||||
# Downloaded by `podcast_feeds.py` when a feed has real per-episode art (see its
|
||||
# `episode_cover_filename`) - same stem as the episode's audio file.
|
||||
podcast = config_dir / "music" / "Kinderpodcasts" / "Wissen macht Ah"
|
||||
Image.new("RGB", (32, 32), (10, 20, 30)).save(podcast / "cover.jpg")
|
||||
Image.new("RGB", (32, 32), (200, 40, 30)).save(podcast / "20260101 - Neu.jpg")
|
||||
|
||||
library = await build(config_dir)
|
||||
|
||||
assert album_named(library, "Neu").cover == podcast / "20260101 - Neu.jpg"
|
||||
# No sidecar for this one, so it still falls back to the show's shared cover.
|
||||
assert album_named(library, "Alt").cover == podcast / "cover.jpg"
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- cache
|
||||
|
||||
|
||||
@@ -359,6 +375,28 @@ def test_album_ids_are_stable_and_path_derived(tmp_path: Path) -> None:
|
||||
assert first != other
|
||||
|
||||
|
||||
async def test_an_album_nested_under_an_extra_grouping_folder_is_still_found(
|
||||
config_dir: Path,
|
||||
) -> None:
|
||||
"""The bug this fixed: an artist folder that groups its books one level deeper than
|
||||
usual (an age-range folder, say) has no audio directly in it, so the scanner used to
|
||||
treat it as an empty album and skip the whole artist rather than looking further
|
||||
down for the actual album folders."""
|
||||
for index in range(2):
|
||||
write_track(
|
||||
config_dir / "music" / "Hörbücher" / "Petzi" / "ab 5" / "Petzi und der Wal"
|
||||
/ f"{index:02d} - teil.mp3",
|
||||
title=f"Teil {index}",
|
||||
album="Petzi und der Wal",
|
||||
albumartist="Petzi",
|
||||
)
|
||||
library = await build(config_dir)
|
||||
|
||||
album = album_named(library, "Petzi und der Wal")
|
||||
assert album.kind == "book"
|
||||
assert len(album.tracks) == 2
|
||||
|
||||
|
||||
# -------------------------------------------------------------------- figure kinds
|
||||
|
||||
|
||||
|
||||
@@ -7,13 +7,18 @@ from pathlib import Path
|
||||
import httpx2
|
||||
import pytest
|
||||
|
||||
from musicmouse.library import podcast_feeds
|
||||
from musicmouse.library.podcast_feeds import (
|
||||
AudioExtractionError,
|
||||
Episode,
|
||||
_entry_enclosure,
|
||||
download_episode,
|
||||
episode_cover_filename,
|
||||
episode_filename,
|
||||
find_feed_shows,
|
||||
missing_episodes,
|
||||
parse_feed,
|
||||
resolve_episode_cover,
|
||||
sync_all_shows,
|
||||
sync_show,
|
||||
)
|
||||
@@ -43,6 +48,46 @@ _FEED = b"""<?xml version="1.0"?>
|
||||
</rss>
|
||||
"""
|
||||
|
||||
_FEED_WITH_COVERS = b"""<?xml version="1.0"?>
|
||||
<rss version="2.0" xmlns:itunes="http://www.itunes.com/dtds/podcast-1.0.dtd">
|
||||
<channel>
|
||||
<title>Cover Show</title>
|
||||
<link>http://example.com/show</link>
|
||||
<itunes:image href="http://example.com/channel.jpg" />
|
||||
<item>
|
||||
<title>Unique Cover</title>
|
||||
<link>http://example.com/episodes/unique</link>
|
||||
<pubDate>Wed, 01 Jan 2025 08:00:00 GMT</pubDate>
|
||||
<itunes:image href="http://example.com/unique.jpg" />
|
||||
<enclosure url="http://example.com/ep1.mp3" length="100" type="audio/mpeg" />
|
||||
<guid>ep1</guid>
|
||||
</item>
|
||||
<item>
|
||||
<title>Same As Channel</title>
|
||||
<link>http://example.com/show</link>
|
||||
<pubDate>Thu, 02 Jan 2025 08:00:00 GMT</pubDate>
|
||||
<itunes:image href="http://example.com/channel.jpg" />
|
||||
<enclosure url="http://example.com/ep2.mp3" length="100" type="audio/mpeg" />
|
||||
<guid>ep2</guid>
|
||||
</item>
|
||||
</channel>
|
||||
</rss>
|
||||
"""
|
||||
|
||||
_VIDEO_FEED = b"""<?xml version="1.0"?>
|
||||
<rss version="2.0">
|
||||
<channel>
|
||||
<title>Video Show</title>
|
||||
<item>
|
||||
<title>Video Only</title>
|
||||
<pubDate>Wed, 01 Jan 2025 08:00:00 GMT</pubDate>
|
||||
<enclosure url="http://example.com/ep1.mp4" length="100" type="video/mpeg" />
|
||||
<guid>ep1</guid>
|
||||
</item>
|
||||
</channel>
|
||||
</rss>
|
||||
"""
|
||||
|
||||
|
||||
def _root_with_show(
|
||||
tmp_path: Path, *, feed_url: str | None = "http://example.com/feed.xml"
|
||||
@@ -99,7 +144,19 @@ def test_episode_filename_sanitizes_illegal_characters_and_falls_back_on_type()
|
||||
name = episode_filename(
|
||||
episode.published, episode.title, episode.enclosure_type, episode.enclosure_url
|
||||
)
|
||||
assert name == "20250103 - Episode Two SpecialChars.mp3"
|
||||
# Illegal characters become "_" rather than being dropped, matching the
|
||||
# convention every episode already on disk (downloaded by hand via the
|
||||
# `podcast-dl` CLI) was named with.
|
||||
assert name == "20250103 - Episode Two_ Special_Chars_.mp3"
|
||||
|
||||
|
||||
def test_episode_filename_matches_the_podcast_dl_convention_on_a_real_title() -> None:
|
||||
# Pinned against a real file already in the library: `podcast-dl` replaced the
|
||||
# "?" with "_" rather than dropping it, so ours must too or the whole back
|
||||
# catalog would look "new" the moment a feed marker is added for that show.
|
||||
published = parse_feed(_FEED)[0].published
|
||||
name = episode_filename(published, "Wie funktioniert Bogenschießen?", "audio/mpeg", "x.mp3")
|
||||
assert name == "20250101 - Wie funktioniert Bogenschießen_.mp3"
|
||||
|
||||
|
||||
def test_missing_episodes_skips_files_already_on_disk(tmp_path: Path) -> None:
|
||||
@@ -151,7 +208,7 @@ async def test_sync_show_downloads_new_episodes_and_is_a_noop_on_rerun(tmp_path:
|
||||
assert changed_first is True
|
||||
assert changed_second is False
|
||||
assert (folder / "20250101 - Episode One.mp3").exists()
|
||||
assert (folder / "20250103 - Episode Two SpecialChars.mp3").exists()
|
||||
assert (folder / "20250103 - Episode Two_ Special_Chars_.mp3").exists()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@@ -211,3 +268,245 @@ async def test_sync_all_shows_keeps_going_past_one_broken_show(tmp_path: Path) -
|
||||
assert changed is True
|
||||
assert list(good.glob("*.mp3"))
|
||||
assert list(bad.glob("*.mp3")) == []
|
||||
|
||||
|
||||
# ------------------------------------------------------------------- per-episode art
|
||||
|
||||
|
||||
def test_parse_feed_extracts_cover_and_link_only_when_distinct_from_the_channel() -> None:
|
||||
unique, same = parse_feed(_FEED_WITH_COVERS)
|
||||
assert unique.cover_url == "http://example.com/unique.jpg"
|
||||
assert unique.link == "http://example.com/episodes/unique"
|
||||
# This item's image/link both just repeat the channel's own - not real per-episode
|
||||
# art, so both come out None rather than a false positive.
|
||||
assert same.cover_url is None
|
||||
assert same.link is None
|
||||
|
||||
|
||||
def test_episode_cover_filename_shares_the_audio_stem_and_takes_the_url_extension() -> None:
|
||||
episode = parse_feed(_FEED_WITH_COVERS)[0]
|
||||
name = episode_cover_filename(episode.published, episode.title, episode.cover_url)
|
||||
assert name == "20250101 - Unique Cover.jpg"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_resolve_episode_cover_prefers_the_feed_image_without_a_request() -> None:
|
||||
episode = parse_feed(_FEED_WITH_COVERS)[0]
|
||||
|
||||
def handler(request: httpx2.Request) -> httpx2.Response:
|
||||
raise AssertionError("must not fetch the episode page when cover_url is already set")
|
||||
|
||||
async with httpx2.AsyncClient(transport=httpx2.MockTransport(handler)) as client:
|
||||
cover = await resolve_episode_cover(client, episode)
|
||||
|
||||
assert cover == "http://example.com/unique.jpg"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_resolve_episode_cover_falls_back_to_og_image_on_the_linked_page() -> None:
|
||||
episode = Episode(
|
||||
title="No Feed Image",
|
||||
published=parse_feed(_FEED)[0].published,
|
||||
enclosure_url="http://example.com/ep.mp3",
|
||||
enclosure_type="audio/mpeg",
|
||||
link="http://example.com/episode-page",
|
||||
)
|
||||
|
||||
def handler(request: httpx2.Request) -> httpx2.Response:
|
||||
return httpx2.Response(
|
||||
200,
|
||||
content=b'<html><head>'
|
||||
b'<meta property="og:image" content="http://example.com/scraped.jpg"/>'
|
||||
b"</head></html>",
|
||||
)
|
||||
|
||||
async with httpx2.AsyncClient(transport=httpx2.MockTransport(handler)) as client:
|
||||
cover = await resolve_episode_cover(client, episode)
|
||||
|
||||
assert cover == "http://example.com/scraped.jpg"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_resolve_episode_cover_returns_none_when_the_link_is_unreachable() -> None:
|
||||
episode = Episode(
|
||||
title="Broken Link",
|
||||
published=parse_feed(_FEED)[0].published,
|
||||
enclosure_url="http://example.com/ep.mp3",
|
||||
enclosure_type="audio/mpeg",
|
||||
link="http://example.com/missing",
|
||||
)
|
||||
|
||||
async with httpx2.AsyncClient(
|
||||
transport=httpx2.MockTransport(lambda _: httpx2.Response(404))
|
||||
) as client:
|
||||
cover = await resolve_episode_cover(client, episode)
|
||||
|
||||
assert cover is None
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_resolve_episode_cover_returns_none_without_a_cover_or_link() -> None:
|
||||
episode = Episode(
|
||||
title="Nothing",
|
||||
published=parse_feed(_FEED)[0].published,
|
||||
enclosure_url="http://example.com/ep.mp3",
|
||||
enclosure_type="audio/mpeg",
|
||||
)
|
||||
|
||||
async def handler(request: httpx2.Request) -> httpx2.Response: # pragma: no cover
|
||||
raise AssertionError("must not make a request with neither a cover nor a link")
|
||||
|
||||
async with httpx2.AsyncClient(transport=httpx2.MockTransport(handler)) as client:
|
||||
cover = await resolve_episode_cover(client, episode)
|
||||
|
||||
assert cover is None
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_sync_show_downloads_the_episode_cover_alongside_the_audio(tmp_path: Path) -> None:
|
||||
folder = tmp_path / "show"
|
||||
folder.mkdir()
|
||||
|
||||
def handler(request: httpx2.Request) -> httpx2.Response:
|
||||
url = str(request.url)
|
||||
if url == "http://example.com/feed.xml":
|
||||
return httpx2.Response(200, content=_FEED_WITH_COVERS)
|
||||
if url == "http://example.com/unique.jpg":
|
||||
return httpx2.Response(200, content=b"cover-bytes")
|
||||
return httpx2.Response(200, content=b"audio-bytes")
|
||||
|
||||
async with httpx2.AsyncClient(transport=httpx2.MockTransport(handler)) as client:
|
||||
changed = await sync_show(client, folder, "http://example.com/feed.xml")
|
||||
|
||||
assert changed is True
|
||||
assert (folder / "20250101 - Unique Cover.jpg").read_bytes() == b"cover-bytes"
|
||||
# No cover file for the episode whose image just repeats the channel's.
|
||||
assert not (folder / "20250102 - Same As Channel.jpg").exists()
|
||||
assert list(folder.glob(".downloading-*.tmp")) == []
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_sync_show_survives_a_failed_cover_download(tmp_path: Path) -> None:
|
||||
folder = tmp_path / "show"
|
||||
folder.mkdir()
|
||||
|
||||
def handler(request: httpx2.Request) -> httpx2.Response:
|
||||
url = str(request.url)
|
||||
if url == "http://example.com/feed.xml":
|
||||
return httpx2.Response(200, content=_FEED_WITH_COVERS)
|
||||
if url == "http://example.com/unique.jpg":
|
||||
return httpx2.Response(404)
|
||||
return httpx2.Response(200, content=b"audio-bytes")
|
||||
|
||||
async with httpx2.AsyncClient(transport=httpx2.MockTransport(handler)) as client:
|
||||
changed = await sync_show(client, folder, "http://example.com/feed.xml")
|
||||
|
||||
# The episode itself still downloaded fine - only its cover failed.
|
||||
assert changed is True
|
||||
assert (folder / "20250101 - Unique Cover.mp3").exists()
|
||||
assert not (folder / "20250101 - Unique Cover.jpg").exists()
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------- video shows
|
||||
|
||||
|
||||
def test_entry_enclosure_accepts_video_when_no_audio_is_offered() -> None:
|
||||
entry = {"enclosures": [{"href": "http://example.com/ep.mp4", "type": "video/mpeg"}]}
|
||||
assert _entry_enclosure(entry) == ("http://example.com/ep.mp4", "video/mpeg")
|
||||
|
||||
|
||||
def test_entry_enclosure_prefers_audio_over_video_when_both_are_offered() -> None:
|
||||
entry = {
|
||||
"enclosures": [
|
||||
{"href": "http://example.com/ep.mp4", "type": "video/mpeg"},
|
||||
{"href": "http://example.com/ep.mp3", "type": "audio/mpeg"},
|
||||
]
|
||||
}
|
||||
assert _entry_enclosure(entry) == ("http://example.com/ep.mp3", "audio/mpeg")
|
||||
|
||||
|
||||
def test_parse_feed_keeps_a_video_only_episode() -> None:
|
||||
episodes = parse_feed(_VIDEO_FEED)
|
||||
assert episodes[0].enclosure_url == "http://example.com/ep1.mp4"
|
||||
assert episodes[0].enclosure_type == "video/mpeg"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_download_episode_raises_when_ffmpeg_is_not_installed(
|
||||
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
|
||||
) -> None:
|
||||
monkeypatch.setattr(podcast_feeds.shutil, "which", lambda _: None)
|
||||
folder = tmp_path / "show"
|
||||
folder.mkdir()
|
||||
episode = parse_feed(_VIDEO_FEED)[0]
|
||||
|
||||
async with httpx2.AsyncClient(
|
||||
transport=httpx2.MockTransport(lambda _: httpx2.Response(200))
|
||||
) as client:
|
||||
with pytest.raises(AudioExtractionError):
|
||||
await download_episode(client, folder, episode, "20250101 - Video Only.mp3")
|
||||
|
||||
assert list(folder.glob(".downloading-*.tmp")) == []
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_download_episode_raises_when_ffmpeg_exits_nonzero(
|
||||
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
|
||||
) -> None:
|
||||
monkeypatch.setattr(podcast_feeds.shutil, "which", lambda _: "/usr/bin/ffmpeg")
|
||||
|
||||
class _FailingProcess:
|
||||
returncode = 1
|
||||
|
||||
async def communicate(self) -> tuple[bytes, bytes]:
|
||||
return b"", b"boom"
|
||||
|
||||
async def fake_exec(*args: object, **kwargs: object) -> _FailingProcess:
|
||||
return _FailingProcess()
|
||||
|
||||
monkeypatch.setattr(podcast_feeds.asyncio, "create_subprocess_exec", fake_exec)
|
||||
|
||||
folder = tmp_path / "show"
|
||||
folder.mkdir()
|
||||
episode = parse_feed(_VIDEO_FEED)[0]
|
||||
|
||||
async with httpx2.AsyncClient(
|
||||
transport=httpx2.MockTransport(lambda _: httpx2.Response(200))
|
||||
) as client:
|
||||
with pytest.raises(AudioExtractionError, match="boom"):
|
||||
await download_episode(client, folder, episode, "20250101 - Video Only.mp3")
|
||||
|
||||
assert list(folder.glob(".downloading-*.tmp")) == []
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_download_episode_extracts_audio_via_ffmpeg(
|
||||
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
|
||||
) -> None:
|
||||
monkeypatch.setattr(podcast_feeds.shutil, "which", lambda _: "/usr/bin/ffmpeg")
|
||||
|
||||
class _SucceedingProcess:
|
||||
returncode = 0
|
||||
|
||||
async def communicate(self) -> tuple[bytes, bytes]:
|
||||
return b"", b""
|
||||
|
||||
async def fake_exec(*args: object, **kwargs: object) -> _SucceedingProcess:
|
||||
# ffmpeg's destination path is the last positional argument; write to it here
|
||||
# to stand in for what the real binary would have produced.
|
||||
Path(str(args[-1])).write_bytes(b"extracted-audio")
|
||||
return _SucceedingProcess()
|
||||
|
||||
monkeypatch.setattr(podcast_feeds.asyncio, "create_subprocess_exec", fake_exec)
|
||||
|
||||
folder = tmp_path / "show"
|
||||
folder.mkdir()
|
||||
episode = parse_feed(_VIDEO_FEED)[0]
|
||||
|
||||
async with httpx2.AsyncClient(
|
||||
transport=httpx2.MockTransport(lambda _: httpx2.Response(200))
|
||||
) as client:
|
||||
await download_episode(client, folder, episode, "20250101 - Video Only.mp3")
|
||||
|
||||
assert (folder / "20250101 - Video Only.mp3").read_bytes() == b"extracted-audio"
|
||||
assert list(folder.glob(".downloading-*.tmp")) == []
|
||||
|
||||
Reference in New Issue
Block a user