"""Storage analysis and cleanup endpoints.""" from fastapi import APIRouter, Query from app.api.deps import ( AlbumRepoDep, ArtistRepoDep, CurrentUser, FileStorageDep, SuperUser, TrackRepoDep, ) from app.api.schemas.pagination import PagedResponse from app.api.schemas.storage import ( CleanupEnqueuedOut, DiskUsageOut, DuplicateGroupOut, FormatBreakdownOut, GenreCountOut, StorageStatsOut, ) from app.api.schemas.track import TrackOut from app.api.v1.tracks import _build_track_out from app.domain.entities.track import Track from app.workers.queue import enqueue router = APIRouter(prefix="/storage", tags=["storage"]) # How many of the most common genres the dashboard surfaces. _TOP_GENRES = 8 async def _tracks_to_out( tracks: list[Track], artist_repo: ArtistRepoDep, album_repo: AlbumRepoDep ) -> list[TrackOut]: """Hydrate a batch of tracks into ``TrackOut`` (artist/album names + cover flag), resolving each referenced artist/album in a single query.""" artist_ids = list({t.artist_id for t in tracks}) album_ids = list({t.album_id for t in tracks if t.album_id is not None}) artists = {a.id: a for a in await artist_repo.get_many(artist_ids)} albums = {a.id: a for a in await album_repo.get_many(album_ids)} return await _build_track_out(tracks, artists, albums) @router.get("") async def get_storage_stats( track_repo: TrackRepoDep, artist_repo: ArtistRepoDep, album_repo: AlbumRepoDep, storage: FileStorageDep, _: CurrentUser, ) -> StorageStatsOut: """Library + disk statistics for the Storage dashboard (§A6). Aggregates come from the catalogue (cheap GROUP BYs); ``disk`` reflects the real backing volume and is ``None`` for backends without a fixed-capacity disk (e.g. object stores).""" stats = await track_repo.library_stats() total_artists = await artist_repo.count(q=None) total_albums = await album_repo.count(artist_id=None, q=None) genres = await track_repo.genres() disk = await storage.disk_usage() return StorageStatsOut( total_tracks=stats.total_tracks, total_artists=total_artists, total_albums=total_albums, total_size=stats.total_size, total_duration_seconds=stats.total_duration_seconds, largest_track_size=stats.largest_track_size, earliest_added=stats.earliest_added, latest_added=stats.latest_added, by_format=[ FormatBreakdownOut( file_format=f.file_format, track_count=f.track_count, total_size=f.total_size, ) for f in stats.by_format ], by_metadata_status=stats.by_metadata_status, by_source=stats.by_source, top_genres=[ GenreCountOut(genre=genre, track_count=count) for genre, count in genres[:_TOP_GENRES] ], disk=DiskUsageOut(total=disk.total, used=disk.used, free=disk.free) if disk else None, ) @router.get("/duplicates") async def get_duplicates( track_repo: TrackRepoDep, artist_repo: ArtistRepoDep, album_repo: AlbumRepoDep, _: CurrentUser, ) -> list[DuplicateGroupOut]: """Tracks sharing an acoustic fingerprint, grouped — the library's real duplicates (``(source, source_id)`` is already unique). Cheap DB GROUP BY.""" groups = await track_repo.find_duplicate_groups() all_tracks = [track for _, tracks in groups for track in tracks] out = await _tracks_to_out(all_tracks, artist_repo, album_repo) by_id = {item.id: item for item in out} return [ DuplicateGroupOut(fingerprint=fingerprint, tracks=[by_id[t.id] for t in tracks]) for fingerprint, tracks in groups ] @router.get("/broken") async def get_broken_files( track_repo: TrackRepoDep, artist_repo: ArtistRepoDep, album_repo: AlbumRepoDep, _: CurrentUser, limit: int = Query(50, ge=1, le=200), offset: int = Query(0, ge=0), ) -> PagedResponse[TrackOut]: """Tracks whose last enrichment run failed (``metadata_status=failed``) — each carries its ``metadata_error``. A file gone missing on disk is instead reconciled by ``POST /storage/cleanup`` (that needs a filesystem scan).""" tracks = await track_repo.list_by_metadata_status("failed", limit=limit, offset=offset) total = await track_repo.count_by_metadata_status("failed") items = await _tracks_to_out(tracks, artist_repo, album_repo) return PagedResponse(items=items, total=total, limit=limit, offset=offset) @router.get("/missing-metadata") async def get_missing_metadata( track_repo: TrackRepoDep, artist_repo: ArtistRepoDep, album_repo: AlbumRepoDep, _: CurrentUser, limit: int = Query(50, ge=1, le=200), offset: int = Query(0, ge=0), ) -> PagedResponse[TrackOut]: """Tracks still awaiting enrichment (``metadata_status=pending``) — imported but never identified.""" tracks = await track_repo.list_by_metadata_status("pending", limit=limit, offset=offset) total = await track_repo.count_by_metadata_status("pending") items = await _tracks_to_out(tracks, artist_repo, album_repo) return PagedResponse(items=items, total=total, limit=limit, offset=offset) @router.post("/cleanup", status_code=202) async def run_cleanup(_: SuperUser) -> CleanupEnqueuedOut: """Admin: enqueue the storage reconciliation job. It scans the catalogue and removes rows whose backing file has vanished (dangling references). Runs in the worker — the filesystem scan must not block the request cycle.""" job_id = await enqueue("cleanup_storage") return CleanupEnqueuedOut(status="enqueued", job_id=job_id)