Three overlapping fixes so the ingestion pipeline actually runs and the
user can see what it's doing:
Pipeline recovery
- app/database.py: use NullPool when MULITA_CELERY_WORKER=1 so each
Celery task opens a fresh asyncpg connection on its own event loop.
Fixes "another operation in progress" and "Future attached to a
different loop" errors that were dropping ~every thumbnail +
extract_metadata task on the floor.
- app/tasks/thumbs.py: initialize photo=None before the try and rollback
on error so a transport failure in the initial SELECT doesn't raise
UnboundLocalError in the except block and leak rows stuck in 'pending'.
- app/services/vision/bootstrap_models.py: on missing model files,
invoke export_models automatically instead of just warning. First
boot of a fresh install now self-heals.
- app/services/vision/export_models.py: shutil.move instead of
Path.rename so the YOLO export survives the /app → /data/models
cross-volume hop.
- requirements.txt: add ultralytics so export works in a stock image.
Worker topology
- docker-compose.yml: replace the single worker with worker-light
(default/high/low queues, c=2, IO-bound) and worker-vision (vision
queue, c=5, OMP_NUM_THREADS=1 to avoid oversubscription on 6 cores).
Vision is pinned to ≤5 parallel inferences so ONNX doesn't each
spawn an all-cores intra-op pool.
- .env / .env.example: CELERYD_CONCURRENCY replaced with
CELERY_LIGHT_CONCURRENCY + CELERY_VISION_CONCURRENCY.
- Backfill queries in thumbs / scan / vision now ORDER BY taken_at
DESC NULLS LAST so newest photos finish first — the library fills
in top-down in the UI instead of arbitrary insertion order.
Settings visibility
- routers/library.py: new GET /maintenance/pipeline-stats returning
done/total per stage (thumbnails, exif, gps, phash, embeddings,
tags, ocr, faces, face clusters, duplicate groups). Worker-status
now also reports the `vision` queue depth, which was missing.
- services/api.ts: PipelineStats / PipelineStage / ScanStatus types
and the matching client call.
- components/dialogs/SettingsDialog.tsx:
- new Pipeline Progress card with one progress bar per stage
- inline scan banner (processed/total/current folder) inside the
Library section while a scan is running
- Tasks/min throughput computed by diffing worker processed counters
between polls
- Workers section calls out the vision queue and documents the
CELERY_LIGHT/VISION_CONCURRENCY + docker compose up -d scale path
Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
109 lines
3.5 KiB
Python
109 lines
3.5 KiB
Python
"""
|
|
Download vision model weights on first worker boot.
|
|
|
|
Run as: python -m app.services.vision.bootstrap_models
|
|
|
|
Or called from the vision worker entrypoint before Celery starts.
|
|
Downloads are idempotent — existing files are skipped.
|
|
|
|
For models that require export (OpenCLIP, YOLOv8n), see export_models.py.
|
|
Those must be exported once on any machine with pip, then placed in
|
|
the models volume before the worker starts.
|
|
"""
|
|
import logging
|
|
import os
|
|
from pathlib import Path
|
|
from urllib.request import urlretrieve
|
|
|
|
from app.config import settings
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
# (relative_path, url, description)
|
|
# Models with url=None must be pre-exported via export_models.py.
|
|
# InsightFace (RetinaFace + ArcFace) auto-downloads via the insightface
|
|
# package on first use — no manual download entries needed.
|
|
DOWNLOADS = []
|
|
|
|
# Models that need manual export via export_models.py
|
|
EXPORTS = [
|
|
("embed/visual.onnx", "OpenCLIP ViT-B/32 visual encoder"),
|
|
("embed/textual.onnx", "OpenCLIP ViT-B/32 textual encoder"),
|
|
("detect/yolov8n.onnx", "YOLOv8n object detector"),
|
|
]
|
|
|
|
|
|
def bootstrap(models_dir: str | None = None):
|
|
"""Ensure all model files are present. Download what we can, warn about
|
|
files that need manual export."""
|
|
base = Path(models_dir or settings.vision.models_dir)
|
|
base.mkdir(parents=True, exist_ok=True)
|
|
|
|
# Download auto-downloadable models
|
|
for rel_path, url, desc in DOWNLOADS:
|
|
dest = base / rel_path
|
|
dest.parent.mkdir(parents=True, exist_ok=True)
|
|
|
|
if dest.exists():
|
|
logger.debug("Already exists: %s (%s)", dest, desc)
|
|
continue
|
|
|
|
logger.info("Downloading %s → %s", desc, dest)
|
|
try:
|
|
urlretrieve(url, str(dest))
|
|
size_kb = dest.stat().st_size / 1024
|
|
logger.info("Downloaded %s (%.0f KB)", desc, size_kb)
|
|
except Exception as e:
|
|
logger.error("Failed to download %s: %s", desc, e)
|
|
if dest.exists():
|
|
dest.unlink()
|
|
|
|
# Check for manually-exported models
|
|
missing = []
|
|
for rel_path, desc in EXPORTS:
|
|
dest = base / rel_path
|
|
if not dest.exists():
|
|
missing.append((rel_path, desc))
|
|
|
|
if missing:
|
|
logger.warning(
|
|
"Missing %d model file(s); attempting automatic export:",
|
|
len(missing),
|
|
)
|
|
for rel_path, desc in missing:
|
|
logger.warning(" %s — %s", base / rel_path, desc)
|
|
|
|
try:
|
|
from app.services.vision import export_models
|
|
|
|
export_models.export_openclip(base)
|
|
export_models.export_yolov8n(base)
|
|
except Exception as e:
|
|
logger.error(
|
|
"Automatic export failed: %s. "
|
|
"Run `python -m app.services.vision.export_models "
|
|
"--models-dir %s` manually before starting the worker.",
|
|
e,
|
|
base,
|
|
)
|
|
return
|
|
|
|
# Re-check what's still missing after the export pass.
|
|
still_missing = [
|
|
(rel_path, desc)
|
|
for rel_path, desc in EXPORTS
|
|
if not (base / rel_path).exists()
|
|
]
|
|
if still_missing:
|
|
for rel_path, desc in still_missing:
|
|
logger.error(" still missing: %s — %s", base / rel_path, desc)
|
|
else:
|
|
logger.info("All model files present in %s", base)
|
|
else:
|
|
logger.info("All model files present in %s", base)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
logging.basicConfig(level=logging.INFO)
|
|
bootstrap()
|