Adds Redis-backed feature flags for vision stages with admin UI toggles and manual backfill trigger, photo upload and download routers with frontend upload modal, and rawpy-based RAW decoding with JPEG fallback for misnamed DNGs. Fixes pgvector serialization, is_trashed filter, and naive-datetime bind in incremental duplicate regrouping; bumps Celery time limits on regroup tasks beyond the 5-minute default. Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
304 lines
10 KiB
Python
304 lines
10 KiB
Python
"""
|
|
Upload router — lets users drop files (or whole folders) from their
|
|
desktop into a destination Folder, preserving any sub-folder structure
|
|
they bring with them.
|
|
|
|
Each POST handles one file. The frontend fans out many parallel requests
|
|
per drop, giving it per-file progress without the server having to
|
|
invent a chunking protocol. For folder uploads, the browser passes
|
|
`webkitRelativePath` under the `relative_path` field; any leading
|
|
sub-directories there are materialised on disk (and as Folder rows)
|
|
under the destination.
|
|
|
|
Uploaded files are placed under the destination folder on the owner's
|
|
media mount, indexed immediately (Photo row created), and queued for
|
|
the same thumb + metadata pipeline that the scanner uses. An optional
|
|
`heap_id` also drops them into a heap in the same request.
|
|
"""
|
|
import hashlib
|
|
import logging
|
|
import os
|
|
from pathlib import Path
|
|
from datetime import datetime
|
|
from typing import Optional
|
|
|
|
from fastapi import APIRouter, Depends, File, Form, HTTPException, UploadFile
|
|
from sqlalchemy import insert, select
|
|
from sqlalchemy.ext.asyncio import AsyncSession
|
|
|
|
from app.database import get_db
|
|
from app.dependencies import get_current_user
|
|
from app.models import Folder, Heap, Photo, SourceRoot
|
|
from app.models.heaps import heap_photos
|
|
from app.models.user import User
|
|
from app.services.date_guess import has_date_warning
|
|
from app.tasks.scan import SUPPORTED_EXTENSIONS, get_media_type
|
|
from app.tasks.thumbs import generate_thumbnails
|
|
from app.services.metadata import extract_metadata
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
router = APIRouter()
|
|
|
|
|
|
MAX_UPLOAD_BYTES = 500 * 1024 * 1024 # 500 MB per file cap.
|
|
|
|
|
|
def _validate_segment(segment: str) -> str:
|
|
"""Reject path segments that would escape the destination directory."""
|
|
segment = segment.strip()
|
|
if not segment or segment in ('.', '..') or '/' in segment or '\\' in segment:
|
|
raise HTTPException(status_code=400, detail=f"Invalid path segment: {segment!r}")
|
|
return segment
|
|
|
|
|
|
def _sanitize_relative_path(rel: Optional[str]) -> list[str]:
|
|
"""Split `relative_path` into safe segments (dirs + filename).
|
|
|
|
Empty or missing → []. Any absolute path, backslash, or `..` segment
|
|
raises 400 — we never want an upload to escape the destination.
|
|
"""
|
|
if not rel:
|
|
return []
|
|
# Normalise backslashes to forward slashes; browsers on Windows send
|
|
# webkitRelativePath with forward slashes anyway, but defend in depth.
|
|
rel = rel.replace('\\', '/').strip('/')
|
|
if not rel:
|
|
return []
|
|
segs = [_validate_segment(s) for s in rel.split('/') if s]
|
|
return segs
|
|
|
|
|
|
async def _resolve_destination(
|
|
folder_id: str,
|
|
user: User,
|
|
db: AsyncSession,
|
|
) -> Folder:
|
|
"""Resolve `folder_id` to a concrete Folder row the user owns.
|
|
|
|
Accepts both Folder ids and SourceRoot ids (for source roots, we
|
|
return the Folder row at the mount path — the scanner creates one
|
|
for every source root it walks). Raises 404 if neither matches.
|
|
"""
|
|
folder = (await db.execute(
|
|
select(Folder).where(Folder.id == folder_id, Folder.user_id == user.id)
|
|
)).scalar_one_or_none()
|
|
if folder is not None:
|
|
return folder
|
|
|
|
sr = (await db.execute(
|
|
select(SourceRoot).where(SourceRoot.id == folder_id, SourceRoot.user_id == user.id)
|
|
)).scalar_one_or_none()
|
|
if sr is None:
|
|
raise HTTPException(status_code=404, detail="Destination folder not found")
|
|
|
|
root_folder = (await db.execute(
|
|
select(Folder).where(
|
|
Folder.source_root_id == sr.id,
|
|
Folder.user_id == user.id,
|
|
Folder.path == os.path.normpath(sr.path),
|
|
)
|
|
)).scalar_one_or_none()
|
|
if root_folder is None:
|
|
# First-time source root with no walk yet — create the row now so
|
|
# uploads work even before the initial scan has run.
|
|
root_folder = Folder(
|
|
name=sr.name or os.path.basename(sr.path),
|
|
path=os.path.normpath(sr.path),
|
|
source_root_id=sr.id,
|
|
user_id=user.id,
|
|
)
|
|
os.makedirs(root_folder.path, exist_ok=True)
|
|
db.add(root_folder)
|
|
await db.flush()
|
|
return root_folder
|
|
|
|
|
|
async def _ensure_subfolder(
|
|
parent: Folder,
|
|
name: str,
|
|
user: User,
|
|
db: AsyncSession,
|
|
) -> Folder:
|
|
"""Return (or create) a Folder row named `name` under `parent`.
|
|
|
|
Also mkdirs the directory on disk. Idempotent — safe to call for a
|
|
path segment that already exists as a Folder row or directory.
|
|
"""
|
|
child_path = os.path.normpath(os.path.join(parent.path, name))
|
|
|
|
existing = (await db.execute(
|
|
select(Folder).where(
|
|
Folder.path == child_path,
|
|
Folder.user_id == user.id,
|
|
)
|
|
)).scalar_one_or_none()
|
|
if existing is not None:
|
|
os.makedirs(child_path, exist_ok=True)
|
|
return existing
|
|
|
|
os.makedirs(child_path, exist_ok=True)
|
|
child = Folder(
|
|
name=name,
|
|
path=child_path,
|
|
parent_id=parent.id,
|
|
source_root_id=parent.source_root_id,
|
|
user_id=user.id,
|
|
is_hidden=parent.is_hidden,
|
|
)
|
|
db.add(child)
|
|
await db.flush()
|
|
return child
|
|
|
|
|
|
def _unique_path(target_dir: str, filename: str) -> tuple[str, str]:
|
|
"""Return a (filepath, filename) that doesn't collide with an
|
|
existing file on disk. Suffixes " (2)", " (3)", ... until a free
|
|
slot is found. Prevents upload-over-existing and keeps the user's
|
|
original file intact.
|
|
"""
|
|
base, ext = os.path.splitext(filename)
|
|
candidate = os.path.join(target_dir, filename)
|
|
n = 2
|
|
while os.path.exists(candidate):
|
|
new_name = f"{base} ({n}){ext}"
|
|
candidate = os.path.join(target_dir, new_name)
|
|
n += 1
|
|
return candidate, os.path.basename(candidate)
|
|
|
|
|
|
@router.post("")
|
|
async def upload_file(
|
|
file: UploadFile = File(...),
|
|
destination_folder_id: str = Form(...),
|
|
relative_path: Optional[str] = Form(None),
|
|
heap_id: Optional[str] = Form(None),
|
|
db: AsyncSession = Depends(get_db),
|
|
current_user: User = Depends(get_current_user),
|
|
):
|
|
"""Upload a single file into a destination folder (and optionally a
|
|
heap). For folder uploads, `relative_path` carries the sub-folder
|
|
chain from the browser's `webkitRelativePath`, and we materialise
|
|
it under the destination on disk + as Folder rows.
|
|
|
|
Returns the created photo's id on success. 4xx on unsupported file
|
|
type, bad path, missing destination, or too-large file.
|
|
"""
|
|
# --- validate inputs -------------------------------------------------
|
|
raw_name = file.filename or ''
|
|
if not raw_name:
|
|
raise HTTPException(status_code=400, detail="Missing filename")
|
|
|
|
# Prefer the leaf of relative_path when present (it contains the
|
|
# original filename as the browser saw it inside the picked folder).
|
|
segs = _sanitize_relative_path(relative_path)
|
|
if segs:
|
|
leaf = segs[-1]
|
|
subdirs = segs[:-1]
|
|
else:
|
|
leaf = _validate_segment(os.path.basename(raw_name))
|
|
subdirs = []
|
|
|
|
ext = Path(leaf).suffix.lower()
|
|
if ext not in SUPPORTED_EXTENSIONS:
|
|
raise HTTPException(
|
|
status_code=400,
|
|
detail=f"Unsupported file type: {ext or '(none)'}",
|
|
)
|
|
|
|
dest_folder = await _resolve_destination(destination_folder_id, current_user, db)
|
|
|
|
target_folder = dest_folder
|
|
for seg in subdirs:
|
|
target_folder = await _ensure_subfolder(target_folder, seg, current_user, db)
|
|
|
|
target_dir = target_folder.path
|
|
os.makedirs(target_dir, exist_ok=True)
|
|
filepath, final_name = _unique_path(target_dir, leaf)
|
|
|
|
# --- stream to disk, hash as we go ----------------------------------
|
|
hasher = hashlib.sha256()
|
|
total = 0
|
|
try:
|
|
with open(filepath, 'wb') as out:
|
|
while True:
|
|
chunk = await file.read(1024 * 1024)
|
|
if not chunk:
|
|
break
|
|
total += len(chunk)
|
|
if total > MAX_UPLOAD_BYTES:
|
|
out.close()
|
|
os.unlink(filepath)
|
|
raise HTTPException(
|
|
status_code=413,
|
|
detail=f"File exceeds {MAX_UPLOAD_BYTES // (1024*1024)}MB limit",
|
|
)
|
|
hasher.update(chunk)
|
|
out.write(chunk)
|
|
except HTTPException:
|
|
raise
|
|
except Exception as e:
|
|
logger.error(f"Upload write failed for {filepath}: {e}")
|
|
if os.path.exists(filepath):
|
|
try:
|
|
os.unlink(filepath)
|
|
except OSError:
|
|
pass
|
|
raise HTTPException(status_code=500, detail=f"Upload failed: {e}")
|
|
|
|
file_hash = hasher.hexdigest()
|
|
|
|
# --- validate heap before committing the DB row ---------------------
|
|
if heap_id:
|
|
heap = (await db.execute(
|
|
select(Heap).where(Heap.id == heap_id, Heap.user_id == current_user.id)
|
|
)).scalar_one_or_none()
|
|
if heap is None:
|
|
# Destination heap vanished — still keep the file + photo row,
|
|
# but tell the caller so the UI can surface the mismatch.
|
|
heap_id = None
|
|
|
|
# --- create Photo row ------------------------------------------------
|
|
mtime_dt = datetime.fromtimestamp(os.stat(filepath).st_mtime)
|
|
photo = Photo(
|
|
filepath=filepath,
|
|
filename=final_name,
|
|
folder_id=target_folder.id,
|
|
user_id=current_user.id,
|
|
file_hash=file_hash,
|
|
media_type=get_media_type(filepath),
|
|
original_format=Path(filepath).suffix.upper()[1:],
|
|
file_size=total,
|
|
taken_at=mtime_dt,
|
|
taken_at_source='filesystem',
|
|
has_date_warning=has_date_warning(filepath, mtime_dt),
|
|
is_hidden=bool(target_folder.is_hidden),
|
|
processing_status='pending',
|
|
)
|
|
db.add(photo)
|
|
await db.flush()
|
|
|
|
if heap_id:
|
|
await db.execute(
|
|
insert(heap_photos),
|
|
[{"heap_id": heap_id, "photo_id": photo.id}],
|
|
)
|
|
|
|
await db.commit()
|
|
|
|
# Queue the same background work the scanner does so thumbnails +
|
|
# EXIF show up without the user having to trigger a rescan.
|
|
try:
|
|
generate_thumbnails.delay(photo.id)
|
|
extract_metadata.delay(photo.id)
|
|
except Exception as e:
|
|
logger.warning(f"Failed to queue post-upload tasks for {photo.id}: {e}")
|
|
|
|
return {
|
|
"photo_id": photo.id,
|
|
"filename": final_name,
|
|
"folder_id": target_folder.id,
|
|
"folder_path": target_folder.path,
|
|
"heap_id": heap_id,
|
|
}
|