Addresses 16 robustness, transparency, and performance issues across the Celery media processing pipeline: Critical: - Singleton DB engine in vision tasks (was leaking one per task call) - acks_late + task_reject_on_worker_lost so crashed workers don't lose tasks - Global soft/hard time limits (5/10 min) to prevent hung worker slots - Thumbnail copy-before-resize (in-place mutation degraded larger sizes) - backfill_vision now checks each task type independently (OCR, faces, etc.) - Parameterized LIMIT in backfill_vision (was f-string SQL injection) High: - try/except + retry(max=3) on all vision inference tasks - extract_metadata writes processing_error on exiftool failure - PIL Image handles closed in _load_thumb/_load_original - Scan progress Redis keys auto-expire after 1 hour - Watcher lock renewal is wall-clock based (30s) not event-count based - worker_process_init signal warms up vision models on startup Medium: - Explicit task_routes for every task name (wildcards never matched) - app.services.metadata added to Celery include list - POST /maintenance/recover-stuck endpoint for photos stuck in processing - Docker healthchecks for worker-light, worker-vision, and Redis - Task ID in vision log lines for distributed tracing - Bare except:pass narrowed to specific exceptions Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
270 lines
9.8 KiB
YAML
270 lines
9.8 KiB
YAML
services:
|
||
frontend:
|
||
build:
|
||
context: ./frontend
|
||
dockerfile: Dockerfile
|
||
container_name: mulita-frontend
|
||
ports:
|
||
# Host port is configurable via FRONTEND_PORT in .env so multiple
|
||
# instances / other services on the same host don't collide.
|
||
- "${FRONTEND_PORT:-3000}:80"
|
||
depends_on:
|
||
- backend
|
||
networks:
|
||
- mulita-network
|
||
restart: unless-stopped
|
||
|
||
backend:
|
||
build:
|
||
context: ./backend
|
||
dockerfile: Dockerfile
|
||
container_name: mulita-backend
|
||
ports:
|
||
# Direct backend access on the host is rarely needed (the frontend
|
||
# talks to it through the nginx /api proxy on the same network),
|
||
# but it's exposed for debugging / curl. Override with BACKEND_PORT.
|
||
- "${BACKEND_PORT:-8001}:8000"
|
||
volumes:
|
||
- ./mulita.yml:/app/config/mulita.yml:ro
|
||
# The single host → container mount for your photo library. Set
|
||
# PHOTO_DIRS in .env to your library root. Mounted :rw because file
|
||
# operations (rename, move, empty discard pile) need to mutate the
|
||
# filesystem; flip to :ro for a strict read-only library and the
|
||
# write endpoints will return EROFS.
|
||
- ${PHOTO_DIRS:-./photos}:/photos:rw
|
||
- thumbs_data:/data/thumbs
|
||
- proxies_data:/data/proxies
|
||
- db_data:/data/db # retained so the docker-compose.sqlite.yml override has somewhere to put mulita.db
|
||
# Run Alembic migrations before starting uvicorn. On a fresh Postgres
|
||
# the empty 0001 baseline is a no-op stamp; create_all in init_db then
|
||
# builds the schema.
|
||
# init_db creates all tables from models (idempotent create_all),
|
||
# then Alembic runs migrations for existing installs. On fresh DBs
|
||
# create_all already built the full schema, so bootstrap.py stamps
|
||
# alembic head to skip redundant ALTER statements.
|
||
command: sh -c "python -c 'import asyncio; from app.database import init_db; asyncio.run(init_db())' && python bootstrap.py && uvicorn app.main:app --host 0.0.0.0 --port 8000 --reload"
|
||
environment:
|
||
- DATABASE_URL=postgresql+asyncpg://mulita:mulita@db:5432/mulita
|
||
- REDIS_URL=redis://redis:6379
|
||
- CELERY_BROKER_URL=redis://redis:6379
|
||
- CELERY_RESULT_BACKEND=redis://redis:6379
|
||
- PHOTO_DIRS=/photos
|
||
- ALLOWED_ORIGINS=${ALLOWED_ORIGINS:-*}
|
||
- SECRET_KEY=${SECRET_KEY:-mulita-dev-secret-change-me}
|
||
- ACCESS_TOKEN_EXPIRE_MINUTES=${ACCESS_TOKEN_EXPIRE_MINUTES:-60}
|
||
- REFRESH_TOKEN_EXPIRE_DAYS=${REFRESH_TOKEN_EXPIRE_DAYS:-30}
|
||
- LOG_LEVEL=${LOG_LEVEL:-INFO}
|
||
- TZ=${TZ:-UTC}
|
||
depends_on:
|
||
redis:
|
||
condition: service_started
|
||
db:
|
||
condition: service_healthy
|
||
networks:
|
||
- mulita-network
|
||
restart: unless-stopped
|
||
|
||
# ── Celery workers ─────────────────────────────────────────────────────
|
||
#
|
||
# The ingestion pipeline is split across two worker services so CPU-heavy
|
||
# vision tasks (embed / detect / OCR / faces / classify) cannot starve
|
||
# the fast IO-bound tasks (scan / thumbnails / EXIF / phash / duplicates).
|
||
#
|
||
# worker-light listens on default,high,low — IO-bound, cheap
|
||
# worker-vision listens on vision — CPU-bound, loads ONNX
|
||
#
|
||
# Both share the same image, photo volume, and model cache, so there's
|
||
# no disk duplication and model weights are loaded lazily only by
|
||
# worker-vision. Each service has its own concurrency knob; both
|
||
# workers ship their heartbeat to the same Redis broker so the
|
||
# Settings > Workers panel lists them side-by-side.
|
||
#
|
||
# Sizing defaults target a 6-core / 16 GB host:
|
||
# CELERY_LIGHT_CONCURRENCY=2 (enough for parallel thumbnail + EXIF)
|
||
# CELERY_VISION_CONCURRENCY=5 (5 × ~2GB ONNX = ~10GB RAM, 5/6 cores)
|
||
# Raise these in .env and run `docker compose up -d worker-light worker-vision`
|
||
# to scale. Keep light under ~4 and vision under your physical core
|
||
# count; more just thrashes.
|
||
worker-light:
|
||
build:
|
||
context: ./backend
|
||
dockerfile: Dockerfile
|
||
image: mule-image-worker
|
||
container_name: mulita-worker-light
|
||
command: sh -c "python -m app.services.vision.bootstrap_models && celery -A app.tasks.celery worker --loglevel=${LOG_LEVEL:-info} --concurrency=${CELERY_LIGHT_CONCURRENCY:-2} -Q default,high,low -n light@%h"
|
||
volumes:
|
||
- ./mulita.yml:/app/config/mulita.yml:ro
|
||
- ${PHOTO_DIRS:-./photos}:/photos:rw
|
||
- thumbs_data:/data/thumbs
|
||
- proxies_data:/data/proxies
|
||
- db_data:/data/db
|
||
- models_data:/data/models
|
||
environment:
|
||
- DATABASE_URL=postgresql+asyncpg://mulita:mulita@db:5432/mulita
|
||
- REDIS_URL=redis://redis:6379
|
||
- CELERY_BROKER_URL=redis://redis:6379
|
||
- CELERY_RESULT_BACKEND=redis://redis:6379
|
||
- PHOTO_DIRS=/photos
|
||
- LOG_LEVEL=${LOG_LEVEL:-INFO}
|
||
- TZ=${TZ:-UTC}
|
||
# NullPool — see app/database.py for rationale.
|
||
- MULITA_CELERY_WORKER=1
|
||
depends_on:
|
||
redis:
|
||
condition: service_started
|
||
backend:
|
||
condition: service_started
|
||
db:
|
||
condition: service_healthy
|
||
healthcheck:
|
||
test: ["CMD-SHELL", "celery -A app.tasks.celery inspect ping -d light@$$HOSTNAME 2>/dev/null | grep -q OK"]
|
||
interval: 30s
|
||
timeout: 10s
|
||
retries: 3
|
||
start_period: 120s
|
||
networks:
|
||
- mulita-network
|
||
restart: unless-stopped
|
||
|
||
# Dedicated watcher worker — runs the long-lived watch_folders task
|
||
# on its own queue so it never blocks scan/thumbnail workers.
|
||
worker-watcher:
|
||
build:
|
||
context: ./backend
|
||
dockerfile: Dockerfile
|
||
image: mule-image-worker
|
||
container_name: mulita-worker-watcher
|
||
command: sh -c "celery -A app.tasks.celery worker --loglevel=${LOG_LEVEL:-info} --concurrency=1 -Q watcher -n watcher@%h"
|
||
volumes:
|
||
- ./mulita.yml:/app/config/mulita.yml:ro
|
||
- ${PHOTO_DIRS:-./photos}:/photos:rw
|
||
- db_data:/data/db
|
||
environment:
|
||
- DATABASE_URL=postgresql+asyncpg://mulita:mulita@db:5432/mulita
|
||
- REDIS_URL=redis://redis:6379
|
||
- CELERY_BROKER_URL=redis://redis:6379
|
||
- CELERY_RESULT_BACKEND=redis://redis:6379
|
||
- PHOTO_DIRS=/photos
|
||
- LOG_LEVEL=${LOG_LEVEL:-INFO}
|
||
- TZ=${TZ:-UTC}
|
||
- MULITA_CELERY_WORKER=1
|
||
depends_on:
|
||
redis:
|
||
condition: service_started
|
||
db:
|
||
condition: service_healthy
|
||
networks:
|
||
- mulita-network
|
||
restart: unless-stopped
|
||
|
||
worker-vision:
|
||
build:
|
||
context: ./backend
|
||
dockerfile: Dockerfile
|
||
image: mule-image-worker
|
||
container_name: mulita-worker-vision
|
||
command: sh -c "python -m app.services.vision.bootstrap_models && celery -A app.tasks.celery worker --loglevel=${LOG_LEVEL:-info} --concurrency=${CELERY_VISION_CONCURRENCY:-5} -Q vision -n vision@%h"
|
||
volumes:
|
||
- ./mulita.yml:/app/config/mulita.yml:ro
|
||
- ${PHOTO_DIRS:-./photos}:/photos:rw
|
||
- thumbs_data:/data/thumbs
|
||
- proxies_data:/data/proxies
|
||
- db_data:/data/db
|
||
- models_data:/data/models
|
||
environment:
|
||
- DATABASE_URL=postgresql+asyncpg://mulita:mulita@db:5432/mulita
|
||
- REDIS_URL=redis://redis:6379
|
||
- CELERY_BROKER_URL=redis://redis:6379
|
||
- CELERY_RESULT_BACKEND=redis://redis:6379
|
||
- PHOTO_DIRS=/photos
|
||
- LOG_LEVEL=${LOG_LEVEL:-INFO}
|
||
- TZ=${TZ:-UTC}
|
||
- MULITA_CELERY_WORKER=1
|
||
# ONNX Runtime execution providers. Set to "auto" to auto-detect
|
||
# GPU (CUDA > ROCm > OpenVINO > CPU), or explicitly:
|
||
# "CUDAExecutionProvider,CPUExecutionProvider"
|
||
# "ROCMExecutionProvider,CPUExecutionProvider"
|
||
# Default: CPU only. To enable GPU, also uncomment the deploy
|
||
# section below and install nvidia-container-toolkit on the host.
|
||
- VISION_EXECUTION_PROVIDERS=${VISION_EXECUTION_PROVIDERS:-CPUExecutionProvider}
|
||
# Pin each ONNX session to one intra-op thread so N prefork children
|
||
# × default-all-cores doesn't oversubscribe the box. With
|
||
# concurrency=5 and OMP=1, vision peaks at 5 busy cores, leaving
|
||
# one for worker-light + system. These env vars cover the three
|
||
# threading runtimes ONNX Runtime might pick up on first use.
|
||
- OMP_NUM_THREADS=1
|
||
- OPENBLAS_NUM_THREADS=1
|
||
- MKL_NUM_THREADS=1
|
||
# Uncomment for NVIDIA GPU passthrough:
|
||
# deploy:
|
||
# resources:
|
||
# reservations:
|
||
# devices:
|
||
# - driver: nvidia
|
||
# count: all
|
||
# capabilities: [gpu]
|
||
healthcheck:
|
||
test: ["CMD-SHELL", "celery -A app.tasks.celery inspect ping -d vision@$$HOSTNAME 2>/dev/null | grep -q OK"]
|
||
interval: 30s
|
||
timeout: 10s
|
||
retries: 3
|
||
start_period: 300s
|
||
depends_on:
|
||
redis:
|
||
condition: service_started
|
||
backend:
|
||
condition: service_started
|
||
db:
|
||
condition: service_healthy
|
||
networks:
|
||
- mulita-network
|
||
restart: unless-stopped
|
||
|
||
db:
|
||
image: pgvector/pgvector:pg16
|
||
container_name: mulita-db
|
||
environment:
|
||
POSTGRES_USER: mulita
|
||
POSTGRES_PASSWORD: mulita
|
||
POSTGRES_DB: mulita
|
||
volumes:
|
||
- pg_data:/var/lib/postgresql/data
|
||
networks:
|
||
- mulita-network
|
||
restart: unless-stopped
|
||
healthcheck:
|
||
test: ["CMD-SHELL", "pg_isready -U mulita -d mulita"]
|
||
interval: 5s
|
||
timeout: 5s
|
||
retries: 10
|
||
|
||
redis:
|
||
image: redis:7-alpine
|
||
container_name: mulita-redis
|
||
# Host port exposed only for local debugging; the backend / worker
|
||
# reach Redis via the internal mulita-network on its container name.
|
||
ports:
|
||
- "${REDIS_PORT:-6379}:6379"
|
||
volumes:
|
||
- redis_data:/data
|
||
networks:
|
||
- mulita-network
|
||
restart: unless-stopped
|
||
command: redis-server --appendonly yes
|
||
healthcheck:
|
||
test: ["CMD", "redis-cli", "ping"]
|
||
interval: 10s
|
||
timeout: 5s
|
||
retries: 5
|
||
|
||
networks:
|
||
mulita-network:
|
||
driver: bridge
|
||
|
||
volumes:
|
||
thumbs_data:
|
||
proxies_data:
|
||
db_data:
|
||
redis_data:
|
||
pg_data:
|
||
models_data: |