Adds a per-folder "hide from views" toggle so noisy subtrees
(screenshots, WhatsApp dumps, work archives) can be excluded from
cross-cutting views without losing indexing. Photos under a hidden
folder are still scanned, thumbnailed, embedded, OCR'd, face-
extracted — they just stop appearing in All Photos, Rated, Colors,
Map, Tags, People, Search, Duplicates, and the sidebar counts.
Navigating directly into the folder still shows every photo.
Schema (migration 0007_folder_hidden):
- folders.is_hidden user-set toggle, default false
- photos.is_hidden denormalized effective flag (true iff any
ancestor folder is hidden), indexed so cross-
cutting queries stay on the existing planner
paths
The denorm is maintained by two paths:
- The scanner walks the ancestry chain on insert, with a per-scan
memoized cache so each folder is resolved once per scan.
- POST /api/v1/folders/{id}/hide flips folders.is_hidden and runs a
WITH RECURSIVE CTE to recompute every folder's effective state in
one query, then bulk-updates photos WHERE IS DISTINCT FROM. Runs
in ~10 ms on a 13k-photo library.
Filters added (cross-cutting queries):
- /library/stats — every sidebar badge via a shared `visible` filter
- /photos (list) — only when neither folder_id nor heap_id is set;
folder browse and heap browse always show everything
- /photos/map
- /library/duplicates/groups
- /folders/tree photo_count subquery
- /tags count_subq (drives Tags + People sidebar counts)
- services/duplicates.regroup_duplicates (so hidden dupes never
contaminate the Duplicates view)
- services/search.hybrid_search — both semantic (pgvector) and FTS
legs join photos so rankings don't include hidden results
Intentionally NOT filtered:
- /photos?folder_id=X and /photos?heap_id=X (user-intentional browse)
- /library/maintenance/pipeline-stats (tracks real worker state)
- cleanup service (disk-level ops, not views)
Frontend:
- sourceFolders.setHidden(id, hidden) API client method
- FolderTreeNode.is_hidden carried through the tree into TreeItem
- LeftSidebar kebab menu: "Hide from views" / "Show in views" with a
mutation that invalidates folders, photos, stats, and tags caches
- Hidden folder rows swap the Folder icon for EyeOff and render the
label italic/muted so the state is visible at a glance
Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
162 lines
6.3 KiB
Python
162 lines
6.3 KiB
Python
"""
|
|
Unified search service — hybrid FTS + semantic (RRF) search.
|
|
|
|
Phase 1 (PR4): semantic-only via pgvector cosine similarity.
|
|
Phase 2 (PR5): adds FTS via tsvector, enables RRF fusion.
|
|
"""
|
|
import logging
|
|
from typing import Optional
|
|
|
|
import numpy as np
|
|
from sqlalchemy import select, text, func
|
|
from sqlalchemy.ext.asyncio import AsyncSession
|
|
|
|
from app.models import Photo
|
|
from app.models.embeddings import Embedding
|
|
from app.config import settings
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
|
|
async def hybrid_search(
|
|
db: AsyncSession,
|
|
q: Optional[str] = None,
|
|
tag_ids: Optional[list[str]] = None,
|
|
date_from: Optional[str] = None,
|
|
date_to: Optional[str] = None,
|
|
limit: int = 50,
|
|
offset: int = 0,
|
|
) -> list[dict]:
|
|
"""Run hybrid search (FTS + semantic) with RRF fusion.
|
|
|
|
Currently semantic-only; FTS leg added in PR5.
|
|
"""
|
|
model_name = settings.vision.embedder.name
|
|
results = {}
|
|
|
|
# ── Semantic search (CLIP text → pgvector cosine) ─────────────────
|
|
if q:
|
|
try:
|
|
from app.services.vision.registry import registry
|
|
embedder = registry.get_embedder()
|
|
query_vec = embedder.embed_text(q)
|
|
|
|
# pgvector cosine distance: <=> returns distance (lower = closer).
|
|
# Join photos so we can filter out discarded / hidden rows
|
|
# inside the same query — otherwise a hidden-folder photo
|
|
# can take a top-N rank and starve the visible results.
|
|
vec_str = "[" + ",".join(str(float(v)) for v in query_vec) + "]"
|
|
stmt = text("""
|
|
SELECT e.photo_id,
|
|
(e.vector <=> :qvec::vector) AS distance
|
|
FROM embeddings e
|
|
JOIN photos p ON p.id = e.photo_id
|
|
WHERE e.model = :model
|
|
AND p.is_trashed = false
|
|
AND p.is_hidden = false
|
|
ORDER BY e.vector <=> :qvec::vector
|
|
LIMIT 200
|
|
""")
|
|
rows = (await db.execute(stmt, {"qvec": vec_str, "model": model_name})).fetchall()
|
|
|
|
for rank, (photo_id, distance) in enumerate(rows):
|
|
if photo_id not in results:
|
|
results[photo_id] = {"semantic_rank": rank, "fts_rank": None}
|
|
else:
|
|
results[photo_id]["semantic_rank"] = rank
|
|
|
|
except Exception as e:
|
|
logger.warning("Semantic search failed (models may not be loaded): %s", e)
|
|
|
|
# ── FTS search (photos.search_vector + ocr_text) ────────────────
|
|
if q:
|
|
try:
|
|
# Same discarded/hidden filter as the semantic leg.
|
|
# The OCR branch joins photos (through photo_id) so we can
|
|
# filter there too; otherwise OCR hits in hidden folders
|
|
# would leak into results.
|
|
fts_stmt = text("""
|
|
SELECT id, ts_rank(search_vector, plainto_tsquery('english', :q)) AS rank
|
|
FROM photos
|
|
WHERE search_vector @@ plainto_tsquery('english', :q)
|
|
AND is_trashed = false
|
|
AND is_hidden = false
|
|
UNION
|
|
SELECT o.photo_id AS id,
|
|
MAX(o.confidence) AS rank
|
|
FROM ocr_text o
|
|
JOIN photos p ON p.id = o.photo_id
|
|
WHERE to_tsvector('english', o.text) @@ plainto_tsquery('english', :q)
|
|
AND p.is_trashed = false
|
|
AND p.is_hidden = false
|
|
GROUP BY o.photo_id
|
|
ORDER BY rank DESC
|
|
LIMIT 200
|
|
""")
|
|
fts_rows = (await db.execute(fts_stmt, {"q": q})).fetchall()
|
|
for rank, (photo_id, score) in enumerate(fts_rows):
|
|
if photo_id not in results:
|
|
results[photo_id] = {"semantic_rank": None, "fts_rank": rank}
|
|
else:
|
|
results[photo_id]["fts_rank"] = rank
|
|
except Exception as e:
|
|
logger.warning("FTS search failed: %s", e)
|
|
|
|
# ── RRF fusion ────────────────────────────────────────────────────
|
|
k = 60
|
|
scored = []
|
|
for photo_id, ranks in results.items():
|
|
score = 0.0
|
|
if ranks["semantic_rank"] is not None:
|
|
score += 1.0 / (k + ranks["semantic_rank"])
|
|
if ranks.get("fts_rank") is not None:
|
|
score += 1.0 / (k + ranks["fts_rank"])
|
|
scored.append((photo_id, score))
|
|
|
|
scored.sort(key=lambda x: -x[1])
|
|
|
|
# If no text query, fall back to recent photos. Always filter out
|
|
# discarded + hidden here — this path backs the "Tags" and "People"
|
|
# browse views, which should honor the folder hide flag.
|
|
if not q:
|
|
if tag_ids:
|
|
from app.models.tags import photo_tags
|
|
# Subquery to get distinct photo_ids matching the tag filter
|
|
sub = select(photo_tags.c.photo_id).where(
|
|
photo_tags.c.tag_id.in_(tag_ids)
|
|
).distinct().subquery()
|
|
stmt = select(Photo.id).join(sub, Photo.id == sub.c.photo_id)
|
|
else:
|
|
stmt = select(Photo.id)
|
|
stmt = stmt.where(
|
|
Photo.is_discarded.is_(False),
|
|
Photo.is_hidden.is_(False),
|
|
)
|
|
stmt = stmt.order_by(Photo.added_at.desc())
|
|
if date_from:
|
|
stmt = stmt.where(Photo.taken_at >= date_from)
|
|
if date_to:
|
|
stmt = stmt.where(Photo.taken_at <= date_to)
|
|
stmt = stmt.offset(offset).limit(limit)
|
|
rows = (await db.execute(stmt)).fetchall()
|
|
return [{"photo_id": row[0], "score": 0.0} for row in rows]
|
|
|
|
# Apply filters to scored results
|
|
photo_ids = [pid for pid, _ in scored]
|
|
if not photo_ids:
|
|
return []
|
|
|
|
# Filter by tags if requested
|
|
if tag_ids:
|
|
from app.models.tags import photo_tags
|
|
stmt = select(photo_tags.c.photo_id).where(
|
|
photo_tags.c.photo_id.in_(photo_ids),
|
|
photo_tags.c.tag_id.in_(tag_ids),
|
|
).distinct()
|
|
valid_ids = {row[0] for row in (await db.execute(stmt)).fetchall()}
|
|
scored = [(pid, s) for pid, s in scored if pid in valid_ids]
|
|
|
|
# Paginate
|
|
page = scored[offset : offset + limit]
|
|
return [{"photo_id": pid, "score": score} for pid, score in page]
|