"""arq task: reconcile the catalogue against storage. Scans every *local* track and drops rows whose backing file has vanished from storage (a "dangling" reference). Remote placeholders have no local file yet and are skipped by the repository query. The filesystem checks (one ``exists()`` per track) are heavy, so this runs off the request cycle (CLAUDE.md). Guarded against a storage outage: if an implausibly large share of files look missing, it assumes the backend is down and aborts without deleting anything. """ from typing import Any from app.core.logging import get_logger from app.infrastructure.db import session_scope from app.infrastructure.db.repositories import SqlAlchemyTrackRepository from app.infrastructure.storage.provider import get_file_storage log = get_logger("worker.cleanup") # If more than this fraction of tracks look missing, assume the storage backend # is unavailable (not that the library really evaporated) and refuse to delete. _OUTAGE_GUARD_FRACTION = 0.5 async def cleanup_storage(_ctx: dict[str, Any]) -> dict[str, Any]: storage = get_file_storage() async with session_scope() as session: tracks = SqlAlchemyTrackRepository(session) refs = await tracks.all_storage_refs() missing = [track_id for track_id, uri in refs if not await storage.exists(uri)] if refs and len(missing) > len(refs) * _OUTAGE_GUARD_FRACTION: log.warning("cleanup_aborted_outage_guard", scanned=len(refs), missing=len(missing)) return {"scanned": len(refs), "removed": 0, "aborted": True} for track_id in missing: await tracks.delete(track_id) log.info("cleanup_done", scanned=len(refs), removed=len(missing)) return {"scanned": len(refs), "removed": len(missing), "aborted": False}