feat: add vision pipeline scaffolding with ONNX backend
Introduce the app/services/vision/ module with ABC interfaces, ONNX Runtime backend, model registry, and per-task implementations: - OpenCLIP ViT-B/32 embedder (image + text, 512-d) - RapidOCR engine (PP-OCRv4 via ONNX, no PaddlePaddle) - YOLOv8n object detector (raw ONNX, no ultralytics runtime) - YuNet + SFace face processor (Apache 2.0, opencv_zoo, 128-d) - DBSCAN face clustering helper Add VisionSettings to config (mulita.yml + Pydantic), bootstrap_models.py for first-boot weight downloads, models_data Docker volume, and ROCm backend stub for future GPU acceleration. No Celery tasks wired yet — models load but nothing invokes them. Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -30,7 +30,7 @@ RUN pip install --no-cache-dir -r requirements.txt
|
||||
COPY . .
|
||||
|
||||
# Create necessary directories
|
||||
RUN mkdir -p /data/thumbs /data/db /data/proxies /app/config
|
||||
RUN mkdir -p /data/thumbs /data/db /data/proxies /data/models /app/config
|
||||
|
||||
# Expose port
|
||||
EXPOSE 8000
|
||||
|
||||
@@ -29,6 +29,42 @@ class PerformanceSettings(BaseModel):
|
||||
db_pool_size: int = 20
|
||||
db_pool_recycle: int = 3600
|
||||
|
||||
class EmbedderSettings(BaseModel):
|
||||
"""CLIP / SigLIP embedding model settings"""
|
||||
name: str = "openclip_vitb32"
|
||||
batch_size: int = 8
|
||||
|
||||
class OCRSettings(BaseModel):
|
||||
"""PaddleOCR / rapidocr settings"""
|
||||
enabled: bool = True
|
||||
languages: list[str] = ["en"]
|
||||
min_confidence: float = 0.5
|
||||
|
||||
class DetectorSettings(BaseModel):
|
||||
"""YOLOv8n object detection settings"""
|
||||
enabled: bool = True
|
||||
min_confidence: float = 0.35
|
||||
max_detections: int = 50
|
||||
|
||||
class FacesSettings(BaseModel):
|
||||
"""YuNet + SFace face detection/recognition settings"""
|
||||
enabled: bool = True
|
||||
min_face_size: int = 40
|
||||
recognition_threshold: float = 0.4
|
||||
cluster_eps: float = 0.35
|
||||
|
||||
class VisionSettings(BaseModel):
|
||||
"""AI vision pipeline settings. Disabled when running on SQLite
|
||||
(pgvector is required for embedding storage)."""
|
||||
enabled: bool = True
|
||||
backend: str = "onnx" # "onnx" | "rocm" (future)
|
||||
models_dir: str = "/data/models"
|
||||
embedder: EmbedderSettings = EmbedderSettings()
|
||||
ocr: OCRSettings = OCRSettings()
|
||||
detector: DetectorSettings = DetectorSettings()
|
||||
faces: FacesSettings = FacesSettings()
|
||||
worker_concurrency: int = 2
|
||||
|
||||
class MulitaConfig(BaseModel):
|
||||
"""Main configuration from YAML file. Source roots and the discard
|
||||
workflow are owned by the database now — only operational settings
|
||||
@@ -36,6 +72,7 @@ class MulitaConfig(BaseModel):
|
||||
thumbnails: ThumbnailSettings = ThumbnailSettings()
|
||||
scanner: ScannerSettings = ScannerSettings()
|
||||
performance: PerformanceSettings = PerformanceSettings()
|
||||
vision: VisionSettings = VisionSettings()
|
||||
|
||||
class Settings(BaseSettings):
|
||||
"""Application settings"""
|
||||
@@ -127,6 +164,10 @@ class Settings(BaseSettings):
|
||||
def performance(self) -> PerformanceSettings:
|
||||
return self.config.performance
|
||||
|
||||
@property
|
||||
def vision(self) -> VisionSettings:
|
||||
return self.config.vision
|
||||
|
||||
class Config:
|
||||
env_file = ".env"
|
||||
case_sensitive = False
|
||||
|
||||
7
backend/app/services/vision/__init__.py
Normal file
7
backend/app/services/vision/__init__.py
Normal file
@@ -0,0 +1,7 @@
|
||||
"""
|
||||
Vision pipeline services — embedding, OCR, object detection, face recognition.
|
||||
|
||||
All inference is done through the ModelRegistry singleton, which lazy-loads
|
||||
ONNX Runtime sessions on first use and caches them for the lifetime of the
|
||||
worker process.
|
||||
"""
|
||||
89
backend/app/services/vision/base.py
Normal file
89
backend/app/services/vision/base.py
Normal file
@@ -0,0 +1,89 @@
|
||||
"""
|
||||
Abstract base classes for vision backends.
|
||||
|
||||
Each ABC defines the contract a backend must satisfy. The default
|
||||
implementation is ONNXBackend (onnx_backend.py). A ROCm backend can be
|
||||
added later by subclassing these ABCs and registering via
|
||||
settings.vision.backend.
|
||||
"""
|
||||
from abc import ABC, abstractmethod
|
||||
from dataclasses import dataclass
|
||||
import numpy as np
|
||||
|
||||
|
||||
@dataclass
|
||||
class DetectionBox:
|
||||
"""A single object detection result."""
|
||||
label: str
|
||||
confidence: float
|
||||
bbox: list[float] # [x1, y1, x2, y2] normalized 0-1
|
||||
|
||||
|
||||
@dataclass
|
||||
class OCRResult:
|
||||
"""A single OCR text region."""
|
||||
text: str
|
||||
confidence: float
|
||||
bbox: list[float] # [x1, y1, x2, y2] normalized 0-1
|
||||
language: str = ""
|
||||
|
||||
|
||||
@dataclass
|
||||
class FaceDetection:
|
||||
"""A detected face with its recognition embedding."""
|
||||
bbox: list[float] # [x1, y1, x2, y2] normalized 0-1
|
||||
embedding: np.ndarray # float32 vector (128-d for SFace)
|
||||
quality: float
|
||||
|
||||
|
||||
class Embedder(ABC):
|
||||
"""Generates image and text embeddings (e.g. OpenCLIP ViT-B/32)."""
|
||||
|
||||
@abstractmethod
|
||||
def embed_image(self, image: np.ndarray) -> np.ndarray:
|
||||
"""Return a normalized float32 embedding vector for an RGB image."""
|
||||
...
|
||||
|
||||
@abstractmethod
|
||||
def embed_text(self, text: str) -> np.ndarray:
|
||||
"""Return a normalized float32 embedding vector for a text query."""
|
||||
...
|
||||
|
||||
@property
|
||||
@abstractmethod
|
||||
def dim(self) -> int:
|
||||
"""Dimensionality of the output embedding."""
|
||||
...
|
||||
|
||||
|
||||
class OCREngine(ABC):
|
||||
"""Extracts text from images (e.g. rapidocr-onnxruntime)."""
|
||||
|
||||
@abstractmethod
|
||||
def run(self, image: np.ndarray) -> list[OCRResult]:
|
||||
"""Return OCR results for an RGB image."""
|
||||
...
|
||||
|
||||
|
||||
class ObjectDetector(ABC):
|
||||
"""Detects objects in images (e.g. YOLOv8n)."""
|
||||
|
||||
@abstractmethod
|
||||
def detect(self, image: np.ndarray) -> list[DetectionBox]:
|
||||
"""Return detections for an RGB image."""
|
||||
...
|
||||
|
||||
|
||||
class FaceProcessor(ABC):
|
||||
"""Detects faces and extracts recognition embeddings (e.g. YuNet + SFace)."""
|
||||
|
||||
@abstractmethod
|
||||
def process(self, image: np.ndarray) -> list[FaceDetection]:
|
||||
"""Return face detections with embeddings for an RGB image."""
|
||||
...
|
||||
|
||||
@property
|
||||
@abstractmethod
|
||||
def embedding_dim(self) -> int:
|
||||
"""Dimensionality of face embedding vectors."""
|
||||
...
|
||||
83
backend/app/services/vision/bootstrap_models.py
Normal file
83
backend/app/services/vision/bootstrap_models.py
Normal file
@@ -0,0 +1,83 @@
|
||||
"""
|
||||
Download vision model weights on first worker boot.
|
||||
|
||||
Run as: python -m app.services.vision.bootstrap_models
|
||||
|
||||
Or called from the vision worker entrypoint before Celery starts.
|
||||
Downloads are idempotent — existing files with matching sizes are skipped.
|
||||
"""
|
||||
import logging
|
||||
import os
|
||||
from pathlib import Path
|
||||
from urllib.request import urlretrieve
|
||||
|
||||
from app.config import settings
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# (relative_path, url, expected_size_bytes_approx)
|
||||
# Sizes are approximate — used only for skip-if-exists checks, not integrity.
|
||||
MODELS = [
|
||||
# OpenCLIP ViT-B/32 — visual and textual encoders (ONNX)
|
||||
# These must be exported manually via export_openclip.py (see below).
|
||||
# Placeholder entries — bootstrap will warn if missing.
|
||||
("embed/visual.onnx", None, None),
|
||||
("embed/textual.onnx", None, None),
|
||||
|
||||
# YOLOv8n — object detection
|
||||
# Export: `yolo export model=yolov8n.pt format=onnx imgsz=640`
|
||||
# Placeholder — must be exported from ultralytics offline.
|
||||
("detect/yolov8n.onnx", None, None),
|
||||
|
||||
# YuNet — face detection (Apache 2.0, opencv_zoo)
|
||||
(
|
||||
"face/yunet.onnx",
|
||||
"https://github.com/opencv/opencv_zoo/raw/main/models/face_detection_yunet/face_detection_yunet_2023mar.onnx",
|
||||
233_000,
|
||||
),
|
||||
|
||||
# SFace — face recognition (Apache 2.0, opencv_zoo)
|
||||
(
|
||||
"face/sface.onnx",
|
||||
"https://github.com/opencv/opencv_zoo/raw/main/models/face_recognition_sface/face_recognition_sface_2021dec.onnx",
|
||||
37_000_000,
|
||||
),
|
||||
]
|
||||
|
||||
|
||||
def bootstrap(models_dir: str | None = None):
|
||||
"""Ensure all model files are present. Download what we can, warn about
|
||||
files that need manual export."""
|
||||
base = Path(models_dir or settings.vision.models_dir)
|
||||
base.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
for rel_path, url, expected_size in MODELS:
|
||||
dest = base / rel_path
|
||||
dest.parent.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
if dest.exists():
|
||||
logger.debug("Model already exists: %s", dest)
|
||||
continue
|
||||
|
||||
if url is None:
|
||||
logger.warning(
|
||||
"Model file %s not found and has no auto-download URL. "
|
||||
"See bootstrap_models.py for export instructions.",
|
||||
dest,
|
||||
)
|
||||
continue
|
||||
|
||||
logger.info("Downloading %s → %s", url, dest)
|
||||
try:
|
||||
urlretrieve(url, str(dest))
|
||||
actual = dest.stat().st_size
|
||||
logger.info("Downloaded %s (%d bytes)", rel_path, actual)
|
||||
except Exception as e:
|
||||
logger.error("Failed to download %s: %s", rel_path, e)
|
||||
if dest.exists():
|
||||
dest.unlink()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
logging.basicConfig(level=logging.INFO)
|
||||
bootstrap()
|
||||
42
backend/app/services/vision/clustering.py
Normal file
42
backend/app/services/vision/clustering.py
Normal file
@@ -0,0 +1,42 @@
|
||||
"""
|
||||
Face embedding clustering using DBSCAN with cosine distance.
|
||||
|
||||
Called by the periodic `recluster_faces` Celery task (PR7).
|
||||
"""
|
||||
import logging
|
||||
|
||||
import numpy as np
|
||||
from sklearn.cluster import DBSCAN
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def cluster_faces(
|
||||
embeddings: np.ndarray,
|
||||
eps: float = 0.35,
|
||||
min_samples: int = 2,
|
||||
) -> np.ndarray:
|
||||
"""Cluster face embeddings using DBSCAN with cosine metric.
|
||||
|
||||
Args:
|
||||
embeddings: (N, D) float32 array of L2-normalized face embeddings.
|
||||
eps: Maximum cosine distance between two samples to be in the
|
||||
same neighborhood. Lower = tighter clusters.
|
||||
min_samples: Minimum cluster size.
|
||||
|
||||
Returns:
|
||||
(N,) int array of cluster labels. -1 = noise / unclustered.
|
||||
"""
|
||||
if len(embeddings) < min_samples:
|
||||
return np.full(len(embeddings), -1, dtype=int)
|
||||
|
||||
db = DBSCAN(eps=eps, min_samples=min_samples, metric="cosine")
|
||||
labels = db.fit_predict(embeddings)
|
||||
|
||||
n_clusters = len(set(labels) - {-1})
|
||||
n_noise = (labels == -1).sum()
|
||||
logger.info(
|
||||
"Face clustering: %d embeddings → %d clusters, %d noise",
|
||||
len(embeddings), n_clusters, n_noise,
|
||||
)
|
||||
return labels
|
||||
140
backend/app/services/vision/detect.py
Normal file
140
backend/app/services/vision/detect.py
Normal file
@@ -0,0 +1,140 @@
|
||||
"""
|
||||
YOLOv8n object detector using raw ONNX Runtime.
|
||||
|
||||
Expects {models_dir}/detect/yolov8n.onnx, exported from ultralytics
|
||||
via bootstrap_models.py. We do NOT ship ultralytics at runtime to
|
||||
avoid dragging in torch.
|
||||
"""
|
||||
import logging
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
import onnxruntime as ort
|
||||
|
||||
from app.config import VisionSettings
|
||||
from app.services.vision.base import ObjectDetector, DetectionBox
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
_INPUT_SIZE = 640
|
||||
|
||||
# COCO class names (80 classes)
|
||||
COCO_LABELS = [
|
||||
"person", "bicycle", "car", "motorcycle", "airplane", "bus", "train",
|
||||
"truck", "boat", "traffic light", "fire hydrant", "stop sign",
|
||||
"parking meter", "bench", "bird", "cat", "dog", "horse", "sheep",
|
||||
"cow", "elephant", "bear", "zebra", "giraffe", "backpack", "umbrella",
|
||||
"handbag", "tie", "suitcase", "frisbee", "skis", "snowboard",
|
||||
"sports ball", "kite", "baseball bat", "baseball glove", "skateboard",
|
||||
"surfboard", "tennis racket", "bottle", "wine glass", "cup", "fork",
|
||||
"knife", "spoon", "bowl", "banana", "apple", "sandwich", "orange",
|
||||
"broccoli", "carrot", "hot dog", "pizza", "donut", "cake", "chair",
|
||||
"couch", "potted plant", "bed", "dining table", "toilet", "tv",
|
||||
"laptop", "mouse", "remote", "keyboard", "cell phone", "microwave",
|
||||
"oven", "toaster", "sink", "refrigerator", "book", "clock", "vase",
|
||||
"scissors", "teddy bear", "hair drier", "toothbrush",
|
||||
]
|
||||
|
||||
|
||||
def _preprocess(image: np.ndarray) -> tuple[np.ndarray, float, float]:
|
||||
"""Letterbox-resize + normalize to NCHW float32. Returns input tensor
|
||||
and scale factors for mapping boxes back to original coords."""
|
||||
from PIL import Image
|
||||
|
||||
img = Image.fromarray(image).convert("RGB")
|
||||
orig_w, orig_h = img.size
|
||||
|
||||
scale = min(_INPUT_SIZE / orig_w, _INPUT_SIZE / orig_h)
|
||||
new_w = int(orig_w * scale)
|
||||
new_h = int(orig_h * scale)
|
||||
img = img.resize((new_w, new_h), Image.BICUBIC)
|
||||
|
||||
# Paste onto gray canvas
|
||||
canvas = np.full((_INPUT_SIZE, _INPUT_SIZE, 3), 114, dtype=np.uint8)
|
||||
pad_x = (_INPUT_SIZE - new_w) // 2
|
||||
pad_y = (_INPUT_SIZE - new_h) // 2
|
||||
canvas[pad_y : pad_y + new_h, pad_x : pad_x + new_w] = np.array(img)
|
||||
|
||||
blob = canvas.astype(np.float32) / 255.0
|
||||
blob = blob.transpose(2, 0, 1)[np.newaxis] # NCHW
|
||||
return blob, scale, pad_x, pad_y
|
||||
|
||||
|
||||
def _postprocess(
|
||||
outputs: np.ndarray,
|
||||
scale: float,
|
||||
pad_x: int,
|
||||
pad_y: int,
|
||||
orig_w: int,
|
||||
orig_h: int,
|
||||
conf_threshold: float,
|
||||
max_detections: int,
|
||||
) -> list[DetectionBox]:
|
||||
"""Parse YOLOv8 output (1, 84, N) → list of DetectionBox."""
|
||||
# outputs shape: (1, 84, N) where 84 = 4 box coords + 80 class scores
|
||||
preds = outputs[0] # (84, N)
|
||||
preds = preds.T # (N, 84)
|
||||
|
||||
boxes_xywh = preds[:, :4]
|
||||
scores = preds[:, 4:]
|
||||
|
||||
class_ids = np.argmax(scores, axis=1)
|
||||
confidences = scores[np.arange(len(scores)), class_ids]
|
||||
|
||||
mask = confidences >= conf_threshold
|
||||
boxes_xywh = boxes_xywh[mask]
|
||||
class_ids = class_ids[mask]
|
||||
confidences = confidences[mask]
|
||||
|
||||
if len(confidences) == 0:
|
||||
return []
|
||||
|
||||
# Sort by confidence, take top N
|
||||
order = np.argsort(-confidences)[:max_detections]
|
||||
boxes_xywh = boxes_xywh[order]
|
||||
class_ids = class_ids[order]
|
||||
confidences = confidences[order]
|
||||
|
||||
results = []
|
||||
for i in range(len(confidences)):
|
||||
cx, cy, w, h = boxes_xywh[i]
|
||||
# Remove letterbox padding and rescale to original image
|
||||
x1 = (cx - w / 2 - pad_x) / scale
|
||||
y1 = (cy - h / 2 - pad_y) / scale
|
||||
x2 = (cx + w / 2 - pad_x) / scale
|
||||
y2 = (cy + h / 2 - pad_y) / scale
|
||||
# Normalize to 0-1
|
||||
bbox = [
|
||||
max(0, x1 / orig_w),
|
||||
max(0, y1 / orig_h),
|
||||
min(1, x2 / orig_w),
|
||||
min(1, y2 / orig_h),
|
||||
]
|
||||
label = COCO_LABELS[class_ids[i]] if class_ids[i] < len(COCO_LABELS) else f"class_{class_ids[i]}"
|
||||
results.append(DetectionBox(label=label, confidence=float(confidences[i]), bbox=bbox))
|
||||
|
||||
return results
|
||||
|
||||
|
||||
class YOLOv8Detector(ObjectDetector):
|
||||
def __init__(self, settings: VisionSettings):
|
||||
model_path = Path(settings.models_dir) / "detect" / "yolov8n.onnx"
|
||||
|
||||
opts = ort.SessionOptions()
|
||||
opts.inter_op_num_threads = 2
|
||||
opts.intra_op_num_threads = 2
|
||||
|
||||
logger.info("Loading YOLOv8n from %s", model_path)
|
||||
self._session = ort.InferenceSession(str(model_path), opts, providers=["CPUExecutionProvider"])
|
||||
self._conf_threshold = settings.detector.min_confidence
|
||||
self._max_detections = settings.detector.max_detections
|
||||
|
||||
def detect(self, image: np.ndarray) -> list[DetectionBox]:
|
||||
orig_h, orig_w = image.shape[:2]
|
||||
blob, scale, pad_x, pad_y = _preprocess(image)
|
||||
input_name = self._session.get_inputs()[0].name
|
||||
outputs = self._session.run(None, {input_name: blob})[0]
|
||||
return _postprocess(
|
||||
outputs, scale, pad_x, pad_y, orig_w, orig_h,
|
||||
self._conf_threshold, self._max_detections,
|
||||
)
|
||||
81
backend/app/services/vision/embed.py
Normal file
81
backend/app/services/vision/embed.py
Normal file
@@ -0,0 +1,81 @@
|
||||
"""
|
||||
OpenCLIP ViT-B/32 embedder using ONNX Runtime.
|
||||
|
||||
Expects two ONNX files under {models_dir}/embed/:
|
||||
- visual.onnx (image encoder)
|
||||
- textual.onnx (text encoder)
|
||||
|
||||
These are exported from open_clip via bootstrap_models.py.
|
||||
"""
|
||||
import logging
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
import onnxruntime as ort
|
||||
|
||||
from app.config import VisionSettings
|
||||
from app.services.vision.base import Embedder
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# OpenCLIP ViT-B/32 preprocessing constants (ImageNet norm)
|
||||
_MEAN = np.array([0.48145466, 0.4578275, 0.40821073], dtype=np.float32)
|
||||
_STD = np.array([0.26862954, 0.26130258, 0.27577711], dtype=np.float32)
|
||||
_INPUT_SIZE = 224
|
||||
|
||||
|
||||
def _preprocess_image(image: np.ndarray) -> np.ndarray:
|
||||
"""Resize, center-crop, normalize an RGB uint8 image to NCHW float32."""
|
||||
from PIL import Image
|
||||
|
||||
img = Image.fromarray(image).convert("RGB")
|
||||
# Resize shortest edge to _INPUT_SIZE, then center crop
|
||||
w, h = img.size
|
||||
scale = _INPUT_SIZE / min(w, h)
|
||||
img = img.resize((int(w * scale), int(h * scale)), Image.BICUBIC)
|
||||
w, h = img.size
|
||||
left = (w - _INPUT_SIZE) // 2
|
||||
top = (h - _INPUT_SIZE) // 2
|
||||
img = img.crop((left, top, left + _INPUT_SIZE, top + _INPUT_SIZE))
|
||||
|
||||
arr = np.array(img, dtype=np.float32) / 255.0
|
||||
arr = (arr - _MEAN) / _STD
|
||||
arr = arr.transpose(2, 0, 1) # HWC → CHW
|
||||
return arr[np.newaxis] # NCHW
|
||||
|
||||
|
||||
class OpenCLIPEmbedder(Embedder):
|
||||
def __init__(self, settings: VisionSettings):
|
||||
model_dir = Path(settings.models_dir) / "embed"
|
||||
visual_path = model_dir / "visual.onnx"
|
||||
textual_path = model_dir / "textual.onnx"
|
||||
|
||||
opts = ort.SessionOptions()
|
||||
opts.inter_op_num_threads = 2
|
||||
opts.intra_op_num_threads = 2
|
||||
|
||||
logger.info("Loading visual encoder from %s", visual_path)
|
||||
self._visual = ort.InferenceSession(str(visual_path), opts, providers=["CPUExecutionProvider"])
|
||||
|
||||
logger.info("Loading textual encoder from %s", textual_path)
|
||||
self._textual = ort.InferenceSession(str(textual_path), opts, providers=["CPUExecutionProvider"])
|
||||
|
||||
def embed_image(self, image: np.ndarray) -> np.ndarray:
|
||||
inp = _preprocess_image(image)
|
||||
input_name = self._visual.get_inputs()[0].name
|
||||
out = self._visual.run(None, {input_name: inp})[0][0]
|
||||
out = out / np.linalg.norm(out)
|
||||
return out.astype(np.float32)
|
||||
|
||||
def embed_text(self, text: str) -> np.ndarray:
|
||||
import open_clip
|
||||
tokenizer = open_clip.get_tokenizer("ViT-B-32")
|
||||
tokens = tokenizer([text]).numpy().astype(np.int64)
|
||||
input_name = self._textual.get_inputs()[0].name
|
||||
out = self._textual.run(None, {input_name: tokens})[0][0]
|
||||
out = out / np.linalg.norm(out)
|
||||
return out.astype(np.float32)
|
||||
|
||||
@property
|
||||
def dim(self) -> int:
|
||||
return 512
|
||||
153
backend/app/services/vision/faces.py
Normal file
153
backend/app/services/vision/faces.py
Normal file
@@ -0,0 +1,153 @@
|
||||
"""
|
||||
Face detection (YuNet) + recognition (SFace) using ONNX Runtime.
|
||||
|
||||
Both models are from opencv_zoo (Apache 2.0 license).
|
||||
Expects {models_dir}/face/:
|
||||
- yunet.onnx (~75 KB)
|
||||
- sface.onnx (~37 MB, 128-d embeddings)
|
||||
"""
|
||||
import logging
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
import onnxruntime as ort
|
||||
|
||||
from app.config import VisionSettings
|
||||
from app.services.vision.base import FaceProcessor, FaceDetection
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
_YUNET_INPUT_SIZE = 640
|
||||
|
||||
|
||||
def _align_face(image: np.ndarray, landmarks: np.ndarray) -> np.ndarray:
|
||||
"""Align and crop a 112x112 face patch using 5-point landmarks.
|
||||
landmarks shape: (5, 2) — left_eye, right_eye, nose, left_mouth, right_mouth."""
|
||||
from PIL import Image
|
||||
import math
|
||||
|
||||
left_eye = landmarks[0]
|
||||
right_eye = landmarks[1]
|
||||
|
||||
dx = right_eye[0] - left_eye[0]
|
||||
dy = right_eye[1] - left_eye[1]
|
||||
angle = math.degrees(math.atan2(dy, dx))
|
||||
eye_center = ((left_eye[0] + right_eye[0]) / 2, (left_eye[1] + right_eye[1]) / 2)
|
||||
eye_dist = math.sqrt(dx * dx + dy * dy)
|
||||
scale = 64.0 / max(eye_dist, 1e-6) # target: eyes at ~64px apart in 112x112
|
||||
|
||||
img = Image.fromarray(image)
|
||||
img = img.rotate(-angle, center=eye_center, resample=Image.BICUBIC)
|
||||
|
||||
cx, cy = eye_center
|
||||
half = 56.0 / scale
|
||||
crop = img.crop((int(cx - half), int(cy - half * 0.8), int(cx + half), int(cy + half * 1.2)))
|
||||
crop = crop.resize((112, 112), Image.BICUBIC)
|
||||
return np.array(crop, dtype=np.float32)
|
||||
|
||||
|
||||
class YuNetSFaceProcessor(FaceProcessor):
|
||||
def __init__(self, settings: VisionSettings):
|
||||
face_dir = Path(settings.models_dir) / "face"
|
||||
yunet_path = face_dir / "yunet.onnx"
|
||||
sface_path = face_dir / "sface.onnx"
|
||||
|
||||
opts = ort.SessionOptions()
|
||||
opts.inter_op_num_threads = 2
|
||||
opts.intra_op_num_threads = 2
|
||||
|
||||
logger.info("Loading YuNet from %s", yunet_path)
|
||||
self._detector = ort.InferenceSession(str(yunet_path), opts, providers=["CPUExecutionProvider"])
|
||||
|
||||
logger.info("Loading SFace from %s", sface_path)
|
||||
self._recognizer = ort.InferenceSession(str(sface_path), opts, providers=["CPUExecutionProvider"])
|
||||
|
||||
self._min_face_size = settings.faces.min_face_size
|
||||
self._score_threshold = settings.faces.recognition_threshold
|
||||
|
||||
def process(self, image: np.ndarray) -> list[FaceDetection]:
|
||||
orig_h, orig_w = image.shape[:2]
|
||||
|
||||
# Scale image for YuNet (expects fixed input size)
|
||||
scale = min(_YUNET_INPUT_SIZE / orig_w, _YUNET_INPUT_SIZE / orig_h)
|
||||
new_w = int(orig_w * scale)
|
||||
new_h = int(orig_h * scale)
|
||||
|
||||
from PIL import Image as PILImage
|
||||
resized = np.array(
|
||||
PILImage.fromarray(image).resize((new_w, new_h), PILImage.BICUBIC),
|
||||
dtype=np.uint8,
|
||||
)
|
||||
|
||||
# YuNet expects BGR, uint8, NHWC
|
||||
bgr = resized[:, :, ::-1].copy()
|
||||
|
||||
# Run detection
|
||||
det_input = self._detector.get_inputs()[0]
|
||||
# YuNet uses dynamic input — reshape
|
||||
blob = bgr.astype(np.float32)[np.newaxis] # (1, H, W, 3)
|
||||
# Some YuNet ONNX exports expect (1, 3, H, W)
|
||||
if det_input.shape and len(det_input.shape) == 4 and det_input.shape[1] == 3:
|
||||
blob = blob.transpose(0, 3, 1, 2)
|
||||
|
||||
detections_raw = self._detector.run(None, {det_input.name: blob})
|
||||
dets = detections_raw[0] # (N, 15): x,y,w,h,score, 5x landmark pairs
|
||||
|
||||
if dets is None or len(dets) == 0:
|
||||
return []
|
||||
|
||||
results = []
|
||||
for det in dets:
|
||||
score = float(det[4]) if len(det) > 4 else float(det[-1])
|
||||
if score < self._score_threshold:
|
||||
continue
|
||||
|
||||
x, y, w, h = det[0], det[1], det[2], det[3]
|
||||
|
||||
# Filter small faces
|
||||
face_size = max(w, h) / scale
|
||||
if face_size < self._min_face_size:
|
||||
continue
|
||||
|
||||
# Rescale to original image coords
|
||||
x1 = x / scale
|
||||
y1 = y / scale
|
||||
x2 = (x + w) / scale
|
||||
y2 = (y + h) / scale
|
||||
|
||||
bbox = [
|
||||
max(0, x1 / orig_w),
|
||||
max(0, y1 / orig_h),
|
||||
min(1, x2 / orig_w),
|
||||
min(1, y2 / orig_h),
|
||||
]
|
||||
|
||||
# Extract landmarks (5 points) for alignment
|
||||
if len(det) >= 15:
|
||||
landmarks = det[5:15].reshape(5, 2) / scale
|
||||
else:
|
||||
# Fallback: no landmarks, skip recognition
|
||||
continue
|
||||
|
||||
# Align face for recognition
|
||||
face_crop = _align_face(image, landmarks)
|
||||
|
||||
# SFace expects (1, 3, 112, 112) float32, BGR, normalized
|
||||
face_bgr = face_crop[:, :, ::-1].copy()
|
||||
face_blob = (face_bgr / 255.0).transpose(2, 0, 1)[np.newaxis].astype(np.float32)
|
||||
|
||||
rec_input = self._recognizer.get_inputs()[0].name
|
||||
embedding = self._recognizer.run(None, {rec_input: face_blob})[0][0]
|
||||
embedding = embedding / np.linalg.norm(embedding)
|
||||
|
||||
results.append(FaceDetection(
|
||||
bbox=bbox,
|
||||
embedding=embedding.astype(np.float32),
|
||||
quality=score,
|
||||
))
|
||||
|
||||
return results
|
||||
|
||||
@property
|
||||
def embedding_dim(self) -> int:
|
||||
return 128
|
||||
45
backend/app/services/vision/ocr.py
Normal file
45
backend/app/services/vision/ocr.py
Normal file
@@ -0,0 +1,45 @@
|
||||
"""
|
||||
OCR engine using rapidocr-onnxruntime (PP-OCRv4 weights).
|
||||
|
||||
No PaddlePaddle dependency — pure ONNX Runtime. Language packs are
|
||||
downloaded automatically by rapidocr on first use.
|
||||
"""
|
||||
import logging
|
||||
|
||||
import numpy as np
|
||||
|
||||
from app.config import VisionSettings
|
||||
from app.services.vision.base import OCREngine, OCRResult
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class RapidOCREngine(OCREngine):
|
||||
def __init__(self, settings: VisionSettings):
|
||||
from rapidocr_onnxruntime import RapidOCR
|
||||
|
||||
self._min_confidence = settings.ocr.min_confidence
|
||||
self._engine = RapidOCR()
|
||||
logger.info("RapidOCR engine initialized")
|
||||
|
||||
def run(self, image: np.ndarray) -> list[OCRResult]:
|
||||
result, _ = self._engine(image)
|
||||
if not result:
|
||||
return []
|
||||
|
||||
out = []
|
||||
for box, text, score in result:
|
||||
if score < self._min_confidence:
|
||||
continue
|
||||
# box is [[x1,y1],[x2,y2],[x3,y3],[x4,y4]] — take bounding rect
|
||||
xs = [p[0] for p in box]
|
||||
ys = [p[1] for p in box]
|
||||
h, w = image.shape[:2]
|
||||
bbox = [
|
||||
min(xs) / w,
|
||||
min(ys) / h,
|
||||
max(xs) / w,
|
||||
max(ys) / h,
|
||||
]
|
||||
out.append(OCRResult(text=text, confidence=float(score), bbox=bbox))
|
||||
return out
|
||||
37
backend/app/services/vision/onnx_backend.py
Normal file
37
backend/app/services/vision/onnx_backend.py
Normal file
@@ -0,0 +1,37 @@
|
||||
"""
|
||||
ONNX Runtime backend — default CPU inference for all vision models.
|
||||
|
||||
Each create_* method returns a concrete implementation of the
|
||||
corresponding ABC from base.py. Models are loaded from ONNX files
|
||||
under settings.vision.models_dir, downloaded on first boot by
|
||||
bootstrap_models.py.
|
||||
"""
|
||||
import logging
|
||||
|
||||
from app.config import VisionSettings
|
||||
from app.services.vision.base import Embedder, OCREngine, ObjectDetector, FaceProcessor
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class ONNXBackend:
|
||||
"""Factory for ONNX Runtime-based vision model instances."""
|
||||
|
||||
def __init__(self, vision_settings: VisionSettings):
|
||||
self._settings = vision_settings
|
||||
|
||||
def create_embedder(self) -> Embedder:
|
||||
from app.services.vision.embed import OpenCLIPEmbedder
|
||||
return OpenCLIPEmbedder(self._settings)
|
||||
|
||||
def create_ocr(self) -> OCREngine:
|
||||
from app.services.vision.ocr import RapidOCREngine
|
||||
return RapidOCREngine(self._settings)
|
||||
|
||||
def create_detector(self) -> ObjectDetector:
|
||||
from app.services.vision.detect import YOLOv8Detector
|
||||
return YOLOv8Detector(self._settings)
|
||||
|
||||
def create_face_processor(self) -> FaceProcessor:
|
||||
from app.services.vision.faces import YuNetSFaceProcessor
|
||||
return YuNetSFaceProcessor(self._settings)
|
||||
77
backend/app/services/vision/registry.py
Normal file
77
backend/app/services/vision/registry.py
Normal file
@@ -0,0 +1,77 @@
|
||||
"""
|
||||
ModelRegistry — singleton that lazy-loads vision models per worker process.
|
||||
|
||||
Usage from Celery tasks:
|
||||
|
||||
from app.services.vision.registry import registry
|
||||
embedder = registry.get_embedder()
|
||||
vec = embedder.embed_image(img)
|
||||
|
||||
Models are created on first access and cached for the worker's lifetime.
|
||||
The registry reads settings.vision to decide which backend to use and
|
||||
where model weights live.
|
||||
"""
|
||||
import logging
|
||||
from functools import lru_cache
|
||||
|
||||
from app.config import settings
|
||||
from app.services.vision.base import Embedder, OCREngine, ObjectDetector, FaceProcessor
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class ModelRegistry:
|
||||
"""Central access point for all vision models."""
|
||||
|
||||
def __init__(self):
|
||||
self._vision = settings.vision
|
||||
|
||||
@lru_cache(maxsize=1)
|
||||
def get_embedder(self) -> Embedder:
|
||||
logger.info("Loading embedder: %s (backend=%s)", self._vision.embedder.name, self._vision.backend)
|
||||
return self._load_backend().create_embedder()
|
||||
|
||||
@lru_cache(maxsize=1)
|
||||
def get_ocr(self) -> OCREngine:
|
||||
logger.info("Loading OCR engine (backend=%s)", self._vision.backend)
|
||||
return self._load_backend().create_ocr()
|
||||
|
||||
@lru_cache(maxsize=1)
|
||||
def get_detector(self) -> ObjectDetector:
|
||||
logger.info("Loading object detector (backend=%s)", self._vision.backend)
|
||||
return self._load_backend().create_detector()
|
||||
|
||||
@lru_cache(maxsize=1)
|
||||
def get_face_processor(self) -> FaceProcessor:
|
||||
logger.info("Loading face processor (backend=%s)", self._vision.backend)
|
||||
return self._load_backend().create_face_processor()
|
||||
|
||||
@lru_cache(maxsize=1)
|
||||
def _load_backend(self):
|
||||
"""Import and instantiate the configured backend."""
|
||||
backend_name = self._vision.backend
|
||||
if backend_name == "onnx":
|
||||
from app.services.vision.onnx_backend import ONNXBackend
|
||||
return ONNXBackend(self._vision)
|
||||
elif backend_name == "rocm":
|
||||
from app.services.vision.rocm_backend import ROCmBackend
|
||||
return ROCmBackend(self._vision)
|
||||
else:
|
||||
raise ValueError(f"Unknown vision backend: {backend_name}")
|
||||
|
||||
def warmup(self):
|
||||
"""Pre-load all enabled models. Called from Celery worker_process_init
|
||||
on the vision queue to avoid cold-start latency on the first task."""
|
||||
logger.info("Warming up vision models...")
|
||||
self.get_embedder()
|
||||
if self._vision.ocr.enabled:
|
||||
self.get_ocr()
|
||||
if self._vision.detector.enabled:
|
||||
self.get_detector()
|
||||
if self._vision.faces.enabled:
|
||||
self.get_face_processor()
|
||||
logger.info("Vision model warmup complete")
|
||||
|
||||
|
||||
# Module-level singleton. Import this from tasks.
|
||||
registry = ModelRegistry()
|
||||
25
backend/app/services/vision/rocm_backend.py
Normal file
25
backend/app/services/vision/rocm_backend.py
Normal file
@@ -0,0 +1,25 @@
|
||||
"""
|
||||
ROCm backend — GPU-accelerated inference for Radeon 760M-class hardware.
|
||||
|
||||
Stub: raises NotImplementedError on all factory methods. To enable,
|
||||
set `vision.backend: rocm` in mulita.yml once ROCm support is implemented.
|
||||
"""
|
||||
from app.config import VisionSettings
|
||||
from app.services.vision.base import Embedder, OCREngine, ObjectDetector, FaceProcessor
|
||||
|
||||
|
||||
class ROCmBackend:
|
||||
def __init__(self, vision_settings: VisionSettings):
|
||||
self._settings = vision_settings
|
||||
|
||||
def create_embedder(self) -> Embedder:
|
||||
raise NotImplementedError("ROCm backend not yet implemented — use 'onnx'")
|
||||
|
||||
def create_ocr(self) -> OCREngine:
|
||||
raise NotImplementedError("ROCm backend not yet implemented — use 'onnx'")
|
||||
|
||||
def create_detector(self) -> ObjectDetector:
|
||||
raise NotImplementedError("ROCm backend not yet implemented — use 'onnx'")
|
||||
|
||||
def create_face_processor(self) -> FaceProcessor:
|
||||
raise NotImplementedError("ROCm backend not yet implemented — use 'onnx'")
|
||||
@@ -34,6 +34,13 @@ pyexiftool==0.5.6
|
||||
# File watching
|
||||
watchfiles==0.21.0
|
||||
|
||||
# Vision pipeline (ONNX Runtime CPU inference)
|
||||
onnxruntime==1.17.1
|
||||
open-clip-torch==2.24.0 # tokenizer + export helper; inference via ONNX
|
||||
rapidocr-onnxruntime==1.3.22
|
||||
scikit-learn==1.4.0 # DBSCAN for face clustering
|
||||
numpy>=1.26.0,<2.0
|
||||
|
||||
# Utilities
|
||||
pyyaml==6.0.1
|
||||
pydantic==2.5.3
|
||||
|
||||
@@ -69,6 +69,7 @@ services:
|
||||
- thumbs_data:/data/thumbs
|
||||
- proxies_data:/data/proxies
|
||||
- db_data:/data/db
|
||||
- models_data:/data/models
|
||||
environment:
|
||||
- DATABASE_URL=postgresql+asyncpg://mulita:mulita@db:5432/mulita
|
||||
- REDIS_URL=redis://redis:6379
|
||||
@@ -130,4 +131,5 @@ volumes:
|
||||
proxies_data:
|
||||
db_data:
|
||||
redis_data:
|
||||
pg_data:
|
||||
pg_data:
|
||||
models_data:
|
||||
25
mulita.yml
25
mulita.yml
@@ -22,3 +22,28 @@ performance:
|
||||
cache_ttl: 3600
|
||||
db_pool_size: 20
|
||||
db_pool_recycle: 3600
|
||||
|
||||
# AI vision pipeline — embedding, OCR, object detection, face recognition.
|
||||
# Runs on the dedicated `vision` Celery queue (PR4+). Set enabled: false
|
||||
# to disable all vision processing.
|
||||
vision:
|
||||
enabled: true
|
||||
backend: onnx # "onnx" (CPU) | "rocm" (future GPU)
|
||||
models_dir: /data/models
|
||||
embedder:
|
||||
name: openclip_vitb32
|
||||
batch_size: 8
|
||||
ocr:
|
||||
enabled: true
|
||||
languages: [en]
|
||||
min_confidence: 0.5
|
||||
detector:
|
||||
enabled: true
|
||||
min_confidence: 0.35
|
||||
max_detections: 50
|
||||
faces:
|
||||
enabled: true
|
||||
min_face_size: 40
|
||||
recognition_threshold: 0.4
|
||||
cluster_eps: 0.35
|
||||
worker_concurrency: 2
|
||||
|
||||
Reference in New Issue
Block a user