Files
mule-image/backend/app/database.py
dtoro 7cf546af7a feat: map view with GPS extraction fix
Adds a new Map sidebar entry that plots photos by their EXIF GPS
coordinates on a clustered Leaflet map. While wiring this up, the
metadata extractor was reading unprefixed GPS keys that never exist
in `exiftool -G -j` output AND assumed coordinates were already
floats — every photo silently lost its GPS. The new extract_gps
helper handles Composite/EXIF group prefixes and parses DMS strings,
and lat/lon are stored as first-class indexed columns so the map
can query them without parsing exif_json on every request.

Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
2026-04-09 23:44:29 +02:00

154 lines
6.2 KiB
Python

"""
Database configuration and session management
"""
from sqlalchemy.ext.asyncio import AsyncSession, create_async_engine, async_sessionmaker
from sqlalchemy.orm import declarative_base
from sqlalchemy import event, text
import logging
import os
from pathlib import Path
from app.config import settings
logger = logging.getLogger(__name__)
# Create database directory if it doesn't exist
db_path = Path(settings.database_url.replace("sqlite+aiosqlite:///", ""))
db_path.parent.mkdir(parents=True, exist_ok=True)
# Create async engine
# SQLite doesn't support pool configuration
if "sqlite" in settings.database_url:
engine = create_async_engine(
settings.database_url,
echo=False, # Set to True for SQL debugging
connect_args={
"check_same_thread": False, # SQLite specific
"timeout": 30
}
)
else:
engine = create_async_engine(
settings.database_url,
echo=False, # Set to True for SQL debugging
pool_size=settings.performance.db_pool_size,
pool_recycle=settings.performance.db_pool_recycle
)
# Create async session factory
AsyncSessionLocal = async_sessionmaker(
engine,
class_=AsyncSession,
expire_on_commit=False
)
# Base class for models
Base = declarative_base()
async def get_db() -> AsyncSession:
"""Dependency to get database session"""
async with AsyncSessionLocal() as session:
try:
yield session
finally:
await session.close()
async def init_db():
"""Initialize database, create tables if they don't exist"""
async with engine.begin() as conn:
# Import all models to register them with Base
from app.models import Photo, Folder, SourceRoot, Tag, PhotoTag, Heap, HeapPhoto, Embedding
# Create all tables. Note: create_all only creates *missing* tables —
# it does NOT add new columns to existing tables when the model gains
# them. Anything new on an existing table needs an explicit ALTER
# below.
await conn.run_sync(Base.metadata.create_all)
# Enable WAL mode for SQLite (better concurrency)
if "sqlite" in settings.database_url:
await conn.execute(text("PRAGMA journal_mode=WAL"))
await conn.execute(text("PRAGMA synchronous=NORMAL"))
await conn.execute(text("PRAGMA cache_size=10000"))
await conn.execute(text("PRAGMA temp_store=MEMORY"))
# ── Idempotent column adds ────────────────────────────────────────
# The project does not use Alembic; we lean on create_all + a small
# set of inline ALTER TABLE statements for the columns we've added
# post-launch. SQLite supports ADD COLUMN but not "IF NOT EXISTS"
# for columns, so introspect via PRAGMA first. Each entry is
# (column_name, ALTER statement). Add new columns at the bottom.
if "sqlite" in settings.database_url:
existing_cols = {
row[1]
for row in (
await conn.execute(text("PRAGMA table_info(photos)"))
).fetchall()
}
pending_alters: list[tuple[str, str]] = [
("phash", "ALTER TABLE photos ADD COLUMN phash VARCHAR(16)"),
(
"duplicate_group_id",
"ALTER TABLE photos ADD COLUMN duplicate_group_id VARCHAR",
),
("latitude", "ALTER TABLE photos ADD COLUMN latitude REAL"),
("longitude", "ALTER TABLE photos ADD COLUMN longitude REAL"),
]
# Track whether the GPS columns were just added so we can kick
# off a one-shot backfill of existing photos at the end of init.
gps_columns_added = False
for col_name, alter_sql in pending_alters:
if col_name not in existing_cols:
logger.info(f"Adding photos.{col_name} column")
await conn.execute(text(alter_sql))
if col_name in ("latitude", "longitude"):
gps_columns_added = True
# Indexes for the new duplicate-detection columns. CREATE INDEX
# IF NOT EXISTS is supported on SQLite so this is safe to run
# every startup.
await conn.execute(
text("CREATE INDEX IF NOT EXISTS ix_photos_phash ON photos(phash)")
)
await conn.execute(
text(
"CREATE INDEX IF NOT EXISTS ix_photos_duplicate_group_id "
"ON photos(duplicate_group_id)"
)
)
await conn.execute(
text(
"CREATE INDEX IF NOT EXISTS ix_photos_lat_lon "
"ON photos(latitude, longitude)"
)
)
logger.info("Database initialized successfully")
# If we just introduced the GPS columns on an existing install, kick
# off a one-shot backfill so the Map view is populated without a
# manual full re-scan. Imported lazily to avoid pulling Celery into
# the import graph for non-worker processes that don't need it.
if "sqlite" in settings.database_url and gps_columns_added:
try:
from app.tasks.scan import backfill_gps
backfill_gps.delay()
logger.info("Queued one-shot backfill_gps task after column add")
except Exception as e:
logger.warning(f"Could not queue backfill_gps task: {e}")
async def create_fts_table():
"""Create Full-Text Search table for SQLite"""
if "sqlite" in settings.database_url:
async with engine.begin() as conn:
# Create FTS5 virtual table for full-text search
await conn.execute(text("""
CREATE VIRTUAL TABLE IF NOT EXISTS photos_fts USING fts5(
photo_id UNINDEXED,
filename,
user_title,
user_notes,
exif_text,
tokenize='unicode61'
)
"""))
logger.info("FTS5 table created successfully")