Two related fixes:
1. Prevention — scanner now normalizes paths before lookup/insert
in get_or_create_source_root and get_or_create_folder. Trailing
slashes, redundant separators, and `.` segments all collapse to
the same row. _normalize_path uses os.path.normpath; symlinks
are intentionally NOT resolved so mount paths stay intact for
cross-machine portability.
2. Cleanup — new app/services/cleanup.py runs on backend startup
(idempotent) and merges any pre-existing duplicates left over
from older scanner versions:
- Groups source_roots by normalized path. Picks the canonical
row (preferring one with a non-empty name and the earliest
added_at), re-points child Folder rows via UPDATE, and
deletes the duplicates.
- Same for folders, with photo_count as the tiebreaker. Photos
get re-pointed to the canonical folder via UPDATE.
- Recomputes folder.photo_count from the actual non-discarded
photo membership so the sidebar count matches reality.
Wired into main.py's lifespan handler. On the dev DB this merged
the empty-name "/host/Pictures/MulitaTest/" duplicate that was
showing up alongside the canonical MulitaTest source root.
Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
142 lines
4.8 KiB
Python
142 lines
4.8 KiB
Python
"""
|
|
One-shot data integrity cleanup for source_roots / folders / photos.
|
|
|
|
Earlier versions of the scanner stored paths verbatim, so trailing slashes
|
|
and redundant separators produced duplicate SourceRoot and Folder rows for
|
|
the same physical directory. The watcher also auto-created source roots
|
|
when fired with a parent dir. This module merges the duplicates and
|
|
re-points photos to the canonical folder so the data lines up with the
|
|
post-fix scanner.
|
|
|
|
Idempotent: safe to run on every backend startup.
|
|
"""
|
|
import os
|
|
import logging
|
|
from datetime import datetime
|
|
from sqlalchemy import select, update, func
|
|
from sqlalchemy.ext.asyncio import AsyncSession
|
|
|
|
from app.database import AsyncSessionLocal
|
|
from app.models import Photo, Folder, SourceRoot
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
|
|
def _normalize_path(path: str) -> str:
|
|
return os.path.normpath(path)
|
|
|
|
|
|
async def _dedupe_source_roots(session: AsyncSession) -> int:
|
|
"""Group source roots by normalized path and merge duplicates. Returns
|
|
the number of rows deleted."""
|
|
result = await session.execute(select(SourceRoot))
|
|
rows = result.scalars().all()
|
|
|
|
groups: dict[str, list[SourceRoot]] = {}
|
|
for sr in rows:
|
|
norm = _normalize_path(sr.path)
|
|
groups.setdefault(norm, []).append(sr)
|
|
|
|
deleted = 0
|
|
for norm, srs in groups.items():
|
|
if len(srs) == 1:
|
|
# Make sure the canonical row's path is normalized too.
|
|
if srs[0].path != norm:
|
|
srs[0].path = norm
|
|
continue
|
|
# Pick the canonical row: prefer one with a non-empty name and the
|
|
# earliest added_at (most likely the original).
|
|
canonical = sorted(
|
|
srs,
|
|
key=lambda s: (not bool(s.name), s.added_at or datetime.max),
|
|
)[0]
|
|
canonical.path = norm
|
|
for sr in srs:
|
|
if sr.id == canonical.id:
|
|
continue
|
|
# Re-point folders that referenced the duplicate root.
|
|
await session.execute(
|
|
update(Folder)
|
|
.where(Folder.source_root_id == sr.id)
|
|
.values(source_root_id=canonical.id)
|
|
)
|
|
await session.delete(sr)
|
|
deleted += 1
|
|
|
|
return deleted
|
|
|
|
|
|
async def _dedupe_folders(session: AsyncSession) -> int:
|
|
"""Group folders by normalized path and merge duplicates. Returns the
|
|
number of rows deleted."""
|
|
result = await session.execute(select(Folder))
|
|
rows = result.scalars().all()
|
|
|
|
groups: dict[str, list[Folder]] = {}
|
|
for f in rows:
|
|
norm = _normalize_path(f.path)
|
|
groups.setdefault(norm, []).append(f)
|
|
|
|
deleted = 0
|
|
for norm, folders in groups.items():
|
|
if len(folders) == 1:
|
|
if folders[0].path != norm:
|
|
folders[0].path = norm
|
|
continue
|
|
# Canonical = the one with the most photos already attached, then
|
|
# the lowest-id (deterministic tiebreaker).
|
|
canonical = sorted(
|
|
folders,
|
|
key=lambda f: (-(f.photo_count or 0), f.id),
|
|
)[0]
|
|
canonical.path = norm
|
|
for f in folders:
|
|
if f.id == canonical.id:
|
|
continue
|
|
# Re-point photos to the canonical folder.
|
|
await session.execute(
|
|
update(Photo)
|
|
.where(Photo.folder_id == f.id)
|
|
.values(folder_id=canonical.id)
|
|
)
|
|
await session.delete(f)
|
|
deleted += 1
|
|
|
|
return deleted
|
|
|
|
|
|
async def _recompute_folder_counts(session: AsyncSession) -> None:
|
|
"""Set folder.photo_count to the actual non-discarded photo count."""
|
|
result = await session.execute(select(Folder))
|
|
folders = result.scalars().all()
|
|
for f in folders:
|
|
count_result = await session.execute(
|
|
select(func.count(Photo.id)).where(
|
|
Photo.folder_id == f.id,
|
|
Photo.is_discarded == False, # noqa: E712
|
|
)
|
|
)
|
|
f.photo_count = int(count_result.scalar() or 0)
|
|
|
|
|
|
async def cleanup_data_integrity() -> dict:
|
|
"""Top-level entry point. Runs the dedupe + count refresh in a single
|
|
transaction. Returns a small summary dict for logging."""
|
|
async with AsyncSessionLocal() as session:
|
|
try:
|
|
sr_deleted = await _dedupe_source_roots(session)
|
|
f_deleted = await _dedupe_folders(session)
|
|
await _recompute_folder_counts(session)
|
|
await session.commit()
|
|
summary = {
|
|
"source_roots_merged": sr_deleted,
|
|
"folders_merged": f_deleted,
|
|
}
|
|
if sr_deleted or f_deleted:
|
|
logger.info(f"Cleanup merged duplicates: {summary}")
|
|
return summary
|
|
except Exception as e:
|
|
logger.error(f"Cleanup failed: {e}")
|
|
await session.rollback()
|
|
raise
|