feat: prune orphaned photo rows + retry-pending action
Adds /api/v1/library/maintenance/{missing-stats,prune-missing} backed
by a new cleanup helper that deletes Photo rows whose files no longer
exist on disk under a *mounted* source root. Skips photos under
unmounted roots so a temporarily-disconnected drive doesn't get
silently nuked.
Settings panel surfaces the orphan count with a destructive Prune
button, plus a "Kick pending" action that re-queues photos stuck in
processing_status='pending' (typically left behind when the scanner
created the row but the worker never picked up the thumbnail task).
Common trigger: PHOTO_DIRS in .env was repointed at a different
library root, leaving every old row dangling.
Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -142,6 +142,12 @@ class RegenerateThumbnailsRequest(BaseModel):
|
||||
default=False,
|
||||
description="If true, only re-queue photos whose processing_status is 'failed'.",
|
||||
)
|
||||
only_pending: bool = Field(
|
||||
default=False,
|
||||
description="If true, only (re-)queue photos whose processing_status is 'pending'. "
|
||||
"Useful for kicking rows that were created by a scan but never had "
|
||||
"their thumbnail task picked up.",
|
||||
)
|
||||
|
||||
|
||||
@router.get("/maintenance/thumbnail-stats")
|
||||
@@ -207,6 +213,8 @@ async def regenerate_thumbnails(
|
||||
query = query.where(Photo.media_type.in_(media_types))
|
||||
if body.only_failed:
|
||||
query = query.where(Photo.processing_status == 'failed')
|
||||
if body.only_pending:
|
||||
query = query.where(Photo.processing_status == 'pending')
|
||||
|
||||
photos = (await db.execute(query)).scalars().all()
|
||||
|
||||
@@ -390,6 +398,29 @@ async def get_worker_status(db: AsyncSession = Depends(get_db)):
|
||||
}
|
||||
|
||||
|
||||
@router.get("/maintenance/missing-stats")
|
||||
async def get_missing_stats():
|
||||
"""Count photos whose files no longer exist on disk under a mounted
|
||||
source root. Surfaced in Settings so the user can see a number before
|
||||
pulling the trigger on prune-missing. Cheap enough to call freely."""
|
||||
from app.services.cleanup import prune_missing_photos
|
||||
return await prune_missing_photos(dry_run=True)
|
||||
|
||||
|
||||
@router.post("/maintenance/prune-missing")
|
||||
async def run_prune_missing():
|
||||
"""Actually delete the orphaned photo rows reported by /missing-stats.
|
||||
Common cause: PHOTO_DIRS in .env was repointed at a different library
|
||||
leaving every old row dangling. Skips any photo whose source root
|
||||
isn't currently mounted (almost always means an unmounted drive)."""
|
||||
from app.services.cleanup import prune_missing_photos
|
||||
try:
|
||||
return {"status": "success", **(await prune_missing_photos(dry_run=False))}
|
||||
except Exception as e:
|
||||
logger.error(f"Prune missing failed: {e}")
|
||||
return {"status": "error", "message": str(e)}
|
||||
|
||||
|
||||
@router.post("/maintenance/cleanup")
|
||||
async def run_data_integrity_cleanup():
|
||||
"""Re-run the source-roots / folders / photos data-integrity cleanup
|
||||
|
||||
@@ -139,6 +139,73 @@ async def _warn_stale_source_roots(session: AsyncSession) -> int:
|
||||
return stale
|
||||
|
||||
|
||||
async def find_missing_photos(session: AsyncSession) -> tuple[list[str], list[str]]:
|
||||
"""Walk every non-discarded photo and check whether its file is still
|
||||
on disk. Returns (deletable_ids, skipped_under_unmounted_roots).
|
||||
|
||||
Skipped rows are photos whose owning source_root path itself doesn't
|
||||
resolve — that's almost always an unmounted drive, and silently
|
||||
deleting those rows would be data loss. The caller can surface the
|
||||
skip count separately so the user knows the cleanup wasn't a no-op
|
||||
by accident.
|
||||
"""
|
||||
sr_rows = (await session.execute(select(SourceRoot))).scalars().all()
|
||||
sr_mounted: dict[str, bool] = {sr.id: os.path.isdir(sr.path) for sr in sr_rows}
|
||||
|
||||
photos = (await session.execute(
|
||||
select(Photo.id, Photo.filepath, Photo.folder_id)
|
||||
.where(Photo.is_discarded.is_(False))
|
||||
)).all()
|
||||
|
||||
# folder -> source_root lookup
|
||||
folders = (await session.execute(select(Folder.id, Folder.source_root_id))).all()
|
||||
folder_to_sr = {fid: srid for fid, srid in folders}
|
||||
|
||||
deletable: list[str] = []
|
||||
skipped: list[str] = []
|
||||
for pid, fp, folder_id in photos:
|
||||
sr_id = folder_to_sr.get(folder_id)
|
||||
if sr_id is None or not sr_mounted.get(sr_id, False):
|
||||
skipped.append(pid)
|
||||
continue
|
||||
if not os.path.exists(fp):
|
||||
deletable.append(pid)
|
||||
return deletable, skipped
|
||||
|
||||
|
||||
async def prune_missing_photos(dry_run: bool = True) -> dict:
|
||||
"""Delete photo rows whose files are no longer on disk *and* whose
|
||||
source root is currently mounted. Common cause: PHOTO_DIRS in .env
|
||||
was repointed at a different library, leaving every old row orphaned.
|
||||
|
||||
Set dry_run=False to actually delete. The default is intentionally
|
||||
safe so the matching count can be surfaced in the UI before the
|
||||
user commits to it.
|
||||
"""
|
||||
from sqlalchemy import delete
|
||||
async with AsyncSessionLocal() as session:
|
||||
try:
|
||||
deletable, skipped = await find_missing_photos(session)
|
||||
if not dry_run and deletable:
|
||||
# Chunked delete to keep the IN clause within SQLite limits.
|
||||
CHUNK = 500
|
||||
for i in range(0, len(deletable), CHUNK):
|
||||
await session.execute(
|
||||
delete(Photo).where(Photo.id.in_(deletable[i:i + CHUNK]))
|
||||
)
|
||||
await session.commit()
|
||||
logger.info(f"Pruned {len(deletable)} orphaned photo rows")
|
||||
return {
|
||||
"would_delete" if dry_run else "deleted": len(deletable),
|
||||
"skipped_unmounted": len(skipped),
|
||||
"dry_run": dry_run,
|
||||
}
|
||||
except Exception as e:
|
||||
logger.error(f"prune_missing_photos failed: {e}")
|
||||
await session.rollback()
|
||||
raise
|
||||
|
||||
|
||||
async def cleanup_data_integrity() -> dict:
|
||||
"""Top-level entry point. Runs the dedupe + count refresh in a single
|
||||
transaction. Returns a small summary dict for logging."""
|
||||
|
||||
Reference in New Issue
Block a user