diff --git a/app/api/deps.py b/app/api/deps.py index 6c8608d..51ae72d 100644 --- a/app/api/deps.py +++ b/app/api/deps.py @@ -18,6 +18,7 @@ from sqlalchemy.ext.asyncio import AsyncSession from app.application.auth_service import AuthService from app.application.download_service import DownloadService +from app.application.lyrics_service import LyricsService from app.application.metadata_service import MetadataEnrichmentService from app.application.remote_library_service import RemoteLibraryService from app.application.streaming_service import StreamingService @@ -38,6 +39,7 @@ from app.infrastructure.db.repositories import ( SqlAlchemyDownloadJobRepository, SqlAlchemyHistoryRepository, SqlAlchemyLikeRepository, + SqlAlchemyLyricsRepository, SqlAlchemyPlaylistRepository, SqlAlchemyRefreshTokenRepository, SqlAlchemyTrackRepository, @@ -46,6 +48,7 @@ from app.infrastructure.db.repositories import ( ) from app.infrastructure.metadata.acoustid import AcoustIdHttpClient from app.infrastructure.metadata.fingerprint import FpcalcFingerprinter +from app.infrastructure.metadata.lrclib import LrclibHttpClient from app.infrastructure.metadata.tags import MutagenTagReader from app.infrastructure.sources.registry import SourceRegistry, build_source_registry from app.infrastructure.storage.provider import get_file_storage @@ -175,6 +178,19 @@ def get_metadata_service(session: SessionDep, storage: FileStorageDep) -> Metada ) +def get_lyrics_service(session: SessionDep) -> LyricsService: + """Wires the LRCLIB lyrics provider + cache repo (plan §6.7). LRCLIB is + keyless, so this is always available; failures degrade to ``not_found``.""" + settings = get_settings() + return LyricsService( + lyrics=SqlAlchemyLyricsRepository(session), + tracks=SqlAlchemyTrackRepository(session), + artists=SqlAlchemyArtistRepository(session), + albums=SqlAlchemyAlbumRepository(session), + provider=LrclibHttpClient(user_agent=settings.musicbrainz_user_agent), + ) + + def get_download_service(session: SessionDep, storage: FileStorageDep) -> DownloadService: return DownloadService( jobs=SqlAlchemyDownloadJobRepository(session), @@ -198,6 +214,7 @@ def get_remote_library_service(session: SessionDep) -> RemoteLibraryService: UploadServiceDep = Annotated[UploadService, Depends(get_upload_service)] StreamingServiceDep = Annotated[StreamingService, Depends(get_streaming_service)] MetadataServiceDep = Annotated[MetadataEnrichmentService, Depends(get_metadata_service)] +LyricsServiceDep = Annotated[LyricsService, Depends(get_lyrics_service)] DownloadServiceDep = Annotated[DownloadService, Depends(get_download_service)] RemoteLibraryServiceDep = Annotated[RemoteLibraryService, Depends(get_remote_library_service)] diff --git a/app/api/schemas/lyrics.py b/app/api/schemas/lyrics.py new file mode 100644 index 0000000..16b95d5 --- /dev/null +++ b/app/api/schemas/lyrics.py @@ -0,0 +1,33 @@ +"""Lyrics response schema (§6.7 / Now Playing lyrics panel). + +Returns the raw LRC (``synced``) and/or ``plain`` text; the client parses LRC +timestamps for synced highlighting. A miss is a normal 200 with +``status="not_found"`` and null text — not an error — so the panel can render a +"no lyrics" state. +""" + +import uuid + +from pydantic import BaseModel + +from app.domain.entities.lyrics import Lyrics + + +class LyricsOut(BaseModel): + track_id: uuid.UUID + status: str + source: str | None + synced: str | None + plain: str | None + synced_available: bool + + @classmethod + def from_entity(cls, lyrics: Lyrics) -> LyricsOut: + return cls( + track_id=lyrics.track_id, + status=lyrics.status, + source=lyrics.source, + synced=lyrics.synced, + plain=lyrics.plain, + synced_available=lyrics.synced is not None, + ) diff --git a/app/api/v1/tracks.py b/app/api/v1/tracks.py index 996af8f..1db3b2c 100644 --- a/app/api/v1/tracks.py +++ b/app/api/v1/tracks.py @@ -12,12 +12,14 @@ from app.api.deps import ( ArtistRepoDep, CurrentUser, FileStorageDep, + LyricsServiceDep, MetadataServiceDep, RemoteLibraryServiceDep, StreamUser, TrackRepoDep, ) from app.api.schemas.download import DownloadJobOut +from app.api.schemas.lyrics import LyricsOut from app.api.schemas.pagination import PagedResponse from app.api.schemas.track import ( MaterializeResponse, @@ -224,6 +226,24 @@ async def get_similar_tracks(track_id: uuid.UUID, _: CurrentUser) -> Any: ... async def optimize_track(track_id: uuid.UUID, _: CurrentUser) -> Any: ... +@router.get("/{track_id}/lyrics") +async def get_track_lyrics( + track_id: uuid.UUID, lyrics: LyricsServiceDep, _: CurrentUser +) -> LyricsOut: + """Cached lyrics for the Now Playing panel (§6.7). A miss is a normal 200 + with ``status="not_found"`` — the provider (LRCLIB) is queried at most once, + then the outcome is cached.""" + return LyricsOut.from_entity(await lyrics.get_lyrics(track_id)) + + +@router.post("/{track_id}/lyrics/refetch") +async def refetch_track_lyrics( + track_id: uuid.UUID, lyrics: LyricsServiceDep, _: CurrentUser +) -> LyricsOut: + """Force a fresh provider lookup, bypassing the cache (user-triggered).""" + return LyricsOut.from_entity(await lyrics.get_lyrics(track_id, force=True)) + + @router.get("/{track_id}/cover") async def get_track_cover( track_id: uuid.UUID, diff --git a/app/application/lyrics_service.py b/app/application/lyrics_service.py new file mode 100644 index 0000000..3473446 --- /dev/null +++ b/app/application/lyrics_service.py @@ -0,0 +1,97 @@ +"""Lyrics service (plan §6.7). + +Get-or-fetch with caching: a track's lyrics are served from the DB when present; +on a miss (or an expired ``not_found``) we ask the provider (LRCLIB) once, then +cache the outcome. ``not_found`` is cached with a TTL so tracks that genuinely +have no lyrics aren't looked up on every play, but can eventually be retried. + +Degrades gracefully: if the provider is unreachable the lookup just yields a +``not_found`` — the endpoint still returns 200 with empty lyrics, never an error. +""" + +import datetime as dt +import uuid + +from app.domain.entities.lyrics import Lyrics +from app.domain.errors import NotFoundError +from app.domain.ports import ( + AlbumRepository, + ArtistRepository, + LyricsProvider, + LyricsRepository, + TrackRepository, +) + +# Re-lookup a cached "not_found" only after this long — long enough not to spam +# the provider, short enough that lyrics added upstream eventually surface. +_NOT_FOUND_TTL = dt.timedelta(days=7) + +_STATUS_FOUND = "found" +_STATUS_NOT_FOUND = "not_found" + + +class LyricsService: + def __init__( + self, + *, + lyrics: LyricsRepository, + tracks: TrackRepository, + artists: ArtistRepository, + albums: AlbumRepository, + provider: LyricsProvider, + ) -> None: + self._lyrics = lyrics + self._tracks = tracks + self._artists = artists + self._albums = albums + self._provider = provider + + async def get_lyrics(self, track_id: uuid.UUID, *, force: bool = False) -> Lyrics: + """Return cached lyrics, fetching from the provider on a miss/expiry. + ``force`` (the refetch endpoint) bypasses the cache entirely.""" + cached = await self._lyrics.get(track_id) + if not force and cached is not None and self._is_fresh(cached): + return cached + return await self._fetch_and_cache(track_id) + + def _is_fresh(self, cached: Lyrics) -> bool: + if cached.status == _STATUS_FOUND: + return True + if cached.status == _STATUS_NOT_FOUND: + return dt.datetime.now(dt.UTC) - cached.fetched_at < _NOT_FOUND_TTL + # "pending" (never fetched) → not fresh, go fetch. + return False + + async def _fetch_and_cache(self, track_id: uuid.UUID) -> Lyrics: + track = await self._tracks.get_by_id(track_id) + if track is None: + raise NotFoundError(f"Track {track_id} not found.") + + artist = await self._artists.get_by_id(track.artist_id) + album = ( + await self._albums.get_by_id(track.album_id) + if track.album_id is not None + else None + ) + result = await self._provider.fetch( + artist=artist.name if artist else "", + title=track.title, + album=album.title if album else None, + duration_seconds=track.duration_seconds, + ) + + if result is None: + return await self._lyrics.upsert( + track_id=track_id, + synced=None, + plain=None, + source=None, + status=_STATUS_NOT_FOUND, + ) + return await self._lyrics.upsert( + track_id=track_id, + synced=result.synced, + plain=result.plain, + source=result.source, + status=_STATUS_FOUND, + ) diff --git a/app/domain/entities/lyrics.py b/app/domain/entities/lyrics.py new file mode 100644 index 0000000..85a052e --- /dev/null +++ b/app/domain/entities/lyrics.py @@ -0,0 +1,41 @@ +"""Lyrics value objects (plan §6.7). + +``Lyrics`` is the cached row for a track; ``LyricsResult`` is what a provider +(LRCLIB) returns for a lookup. Both cross the domain boundary — no framework +imports. Status values mirror ``LyricsStatus`` in the ORM enum ("found" / +"not_found" / "pending") but are kept as plain strings here so the domain stays +independent of the persistence layer. +""" + +import datetime as dt +import uuid +from dataclasses import dataclass + + +@dataclass(frozen=True, slots=True) +class LyricsResult: + """A provider hit: synced (timestamped LRC) and/or plain text.""" + + synced: str | None + plain: str | None + source: str + + +@dataclass(frozen=True, slots=True) +class Lyrics: + """Cached lyrics for one track. ``status`` is ``found`` / ``not_found`` / + ``pending``; ``not_found`` is cached too (with a TTL in the service) so a + track with no lyrics doesn't hammer the provider on every play.""" + + track_id: uuid.UUID + synced: str | None + plain: str | None + source: str | None + status: str + fetched_at: dt.datetime + + @property + def has_lyrics(self) -> bool: + return self.status == "found" and ( + self.synced is not None or self.plain is not None + ) diff --git a/app/domain/ports.py b/app/domain/ports.py index f30bbcd..ff71630 100644 --- a/app/domain/ports.py +++ b/app/domain/ports.py @@ -29,6 +29,7 @@ from app.domain.entities import ( SubsonicCredentials, User, ) +from app.domain.entities.lyrics import Lyrics, LyricsResult from app.domain.entities.settings import UserSettings from app.domain.entities.track import Artist, Track from app.domain.sources import DownloadResult, RawMetadata, SearchResult, SourceFile, SourceInfo @@ -494,3 +495,33 @@ class CoverArtProvider(Protocol): def is_available(self) -> bool: ... async def fetch_release_group(self, release_group_mbid: str) -> CoverArt | None: ... + + +class LyricsProvider(Protocol): + """Fetches lyrics from an external database (LRCLIB) by artist/title/album/ + duration. Returns a hit or ``None`` (no match / service down), never raising.""" + + async def fetch( + self, + *, + artist: str, + title: str, + album: str | None, + duration_seconds: int | None, + ) -> LyricsResult | None: ... + + +class LyricsRepository(Protocol): + """Cached lyrics, one row per track. ``upsert`` also caches a ``not_found`` + (empty text) so misses aren't re-fetched until the service's TTL lapses.""" + + async def get(self, track_id: uuid.UUID) -> Lyrics | None: ... + async def upsert( + self, + *, + track_id: uuid.UUID, + synced: str | None, + plain: str | None, + source: str | None, + status: str, + ) -> Lyrics: ... diff --git a/app/infrastructure/db/repositories/__init__.py b/app/infrastructure/db/repositories/__init__.py index 205aa2a..2670f3c 100644 --- a/app/infrastructure/db/repositories/__init__.py +++ b/app/infrastructure/db/repositories/__init__.py @@ -7,6 +7,7 @@ from app.infrastructure.db.repositories.download_job_repository import ( ) from app.infrastructure.db.repositories.history_repository import SqlAlchemyHistoryRepository from app.infrastructure.db.repositories.like_repository import SqlAlchemyLikeRepository +from app.infrastructure.db.repositories.lyrics_repository import SqlAlchemyLyricsRepository from app.infrastructure.db.repositories.playlist_repository import SqlAlchemyPlaylistRepository from app.infrastructure.db.repositories.refresh_token_repository import ( SqlAlchemyRefreshTokenRepository, @@ -23,6 +24,7 @@ __all__ = [ "SqlAlchemyDownloadJobRepository", "SqlAlchemyHistoryRepository", "SqlAlchemyLikeRepository", + "SqlAlchemyLyricsRepository", "SqlAlchemyPlaylistRepository", "SqlAlchemyRefreshTokenRepository", "SqlAlchemyTrackRepository", diff --git a/app/infrastructure/db/repositories/lyrics_repository.py b/app/infrastructure/db/repositories/lyrics_repository.py new file mode 100644 index 0000000..d1be512 --- /dev/null +++ b/app/infrastructure/db/repositories/lyrics_repository.py @@ -0,0 +1,72 @@ +"""Lyrics repository — adapter over ``AsyncSession``. + +One cached row per track (``track_id`` unique). ``upsert`` refreshes the row and +bumps ``fetched_at`` so the service's TTL is measured from the last fetch. +""" + +import uuid + +from sqlalchemy import func, select +from sqlalchemy.dialects.postgresql import insert as pg_insert +from sqlalchemy.ext.asyncio import AsyncSession + +from app.domain.entities.lyrics import Lyrics +from app.infrastructure.db.models.lyrics import LyricsModel + + +def _to_entity(row: LyricsModel) -> Lyrics: + return Lyrics( + track_id=row.track_id, + synced=row.synced, + plain=row.plain, + source=row.source, + status=row.status, + fetched_at=row.fetched_at, + ) + + +class SqlAlchemyLyricsRepository: + def __init__(self, session: AsyncSession) -> None: + self._session = session + + async def get(self, track_id: uuid.UUID) -> Lyrics | None: + row = await self._session.scalar( + select(LyricsModel).where(LyricsModel.track_id == track_id) + ) + return _to_entity(row) if row is not None else None + + async def upsert( + self, + *, + track_id: uuid.UUID, + synced: str | None, + plain: str | None, + source: str | None, + status: str, + ) -> Lyrics: + values = { + "track_id": track_id, + "synced": synced, + "plain": plain, + "source": source, + "status": status, + "fetched_at": func.now(), + } + stmt = ( + pg_insert(LyricsModel) + .values(**values) + .on_conflict_do_update( + index_elements=[LyricsModel.track_id], + set_={ + "synced": synced, + "plain": plain, + "source": source, + "status": status, + "fetched_at": func.now(), + }, + ) + .returning(LyricsModel) + ) + row = (await self._session.scalars(stmt)).one() + await self._session.flush() + return _to_entity(row) diff --git a/app/infrastructure/metadata/lrclib.py b/app/infrastructure/metadata/lrclib.py new file mode 100644 index 0000000..1b7d9c2 --- /dev/null +++ b/app/infrastructure/metadata/lrclib.py @@ -0,0 +1,100 @@ +"""LrclibHttpClient — fetches lyrics from LRCLIB (plan §6.7). + +LRCLIB is a free, keyless lyrics database. ``/api/get`` does an exact match on +artist+track+album+duration; if that misses we fall back to ``/api/search`` and +take the best-scoring hit. Graceful degradation: any network/parse error → +``fetch`` returns ``None`` (the service then caches a ``not_found``), never +raising. No API key is needed, so this provider is always "available". +""" + +import httpx + +from app.core.logging import get_logger +from app.domain.entities.lyrics import LyricsResult + +log = get_logger(__name__) + +_BASE_URL = "https://lrclib.net" +_TIMEOUT_SECONDS = 10.0 +_SOURCE = "lrclib" + + +class LrclibHttpClient: + """Implements :class:`app.domain.ports.LyricsProvider`.""" + + def __init__(self, *, user_agent: str, base_url: str = _BASE_URL) -> None: + self._user_agent = user_agent + self._base_url = base_url.rstrip("/") + + async def fetch( + self, + *, + artist: str, + title: str, + album: str | None, + duration_seconds: int | None, + ) -> LyricsResult | None: + try: + async with httpx.AsyncClient( + timeout=_TIMEOUT_SECONDS, + headers={"User-Agent": self._user_agent}, + base_url=self._base_url, + ) as client: + hit = await self._get(client, artist, title, album, duration_seconds) + if hit is None: + hit = await self._search(client, artist, title) + except (httpx.HTTPError, ValueError) as exc: + log.warning("lrclib.fetch_failed", error=str(exc)) + return None + return hit + + async def _get( + self, + client: httpx.AsyncClient, + artist: str, + title: str, + album: str | None, + duration_seconds: int | None, + ) -> LyricsResult | None: + """Exact match via ``/api/get`` (404 when nothing matches exactly).""" + params = {"artist_name": artist, "track_name": title} + if album: + params["album_name"] = album + if duration_seconds is not None: + params["duration"] = str(duration_seconds) + resp = await client.get("/api/get", params=params) + if resp.status_code == httpx.codes.NOT_FOUND: + return None + resp.raise_for_status() + return _to_result(resp.json()) + + async def _search( + self, client: httpx.AsyncClient, artist: str, title: str + ) -> LyricsResult | None: + """Fuzzy fallback via ``/api/search`` — take the first usable hit.""" + resp = await client.get( + "/api/search", params={"artist_name": artist, "track_name": title} + ) + resp.raise_for_status() + results = resp.json() + if not isinstance(results, list): + return None + for item in results: + result = _to_result(item) + if result is not None: + return result + return None + + +def _to_result(payload: object) -> LyricsResult | None: + """Map an LRCLIB record to a ``LyricsResult``. Instrumental tracks and empty + records yield ``None`` (nothing worth caching as "found").""" + if not isinstance(payload, dict): + return None + if payload.get("instrumental"): + return None + synced = payload.get("syncedLyrics") or None + plain = payload.get("plainLyrics") or None + if synced is None and plain is None: + return None + return LyricsResult(synced=synced, plain=plain, source=_SOURCE)