refactor frontend and backend modules

This commit is contained in:
2026-05-04 22:57:57 +02:00
parent e0e461502b
commit 8f0a9650b0
17 changed files with 4560 additions and 4416 deletions
@@ -1,458 +1 @@
"""SQLite-backed media inventory service.
This is the main step toward a frontend-agnostic architecture. The Streamlit UI
asks this service to build/query an index, but the same class could be exposed
through FastAPI to a React frontend without rewriting Jellyfin indexing logic.
"""
from __future__ import annotations
import logging
import sqlite3
import time
from dataclasses import dataclass
from pathlib import Path
from typing import Any, Callable, Iterable
from media_library_viewer_api.clients.jellyfin import JellyfinClient
from media_library_viewer_api.domain.media import display_media_row, normalize_media_item
from media_library_viewer_api.path_utils import resolve_remote_media_path
logger = logging.getLogger(__name__)
# Local generated database. It is ignored by git and can be rebuilt from
# Jellyfin metadata whenever needed.
DEFAULT_INDEX_PATH = Path(".cache/media_library_viewer/media_index.sqlite")
MEDIA_TYPES = "Movie,Episode,Video"
# Only values from this whitelist are interpolated into ORDER BY. User-selected
# sort keys map to these known SQL snippets to avoid SQL injection.
SORT_COLUMNS = {
"title": "title COLLATE NOCASE",
"series": "series COLLATE NOCASE",
"season": "season_number",
"episode": "episode",
"type": "type COLLATE NOCASE",
"year": "year",
"runtime": "runtime_min",
"size": "size_bytes",
"bitrate": "bitrate_bps",
"hdr": "hdr",
"video": "video COLLATE NOCASE",
"resolution": "height",
"date_added": "date_added_ts",
"library": "library_name COLLATE NOCASE",
"path": "path COLLATE NOCASE",
}
def _estimate_remaining_seconds(elapsed_seconds: float, progress: float | None) -> float | None:
if progress is None:
return None
progress = max(0.0, min(1.0, progress))
if progress <= 0.0:
return None
return max(0.0, elapsed_seconds * (1.0 - progress) / progress)
class MediaIndexBuildCancelled(Exception):
"""Raised when a media index build is requested to stop."""
@dataclass(frozen=True)
class MediaIndexStatus:
"""Lightweight status object displayed by the Media tab."""
exists: bool
item_count: int = 0
updated_at: int | None = None
updated_at_label: str = ""
build_duration_seconds: float | None = None
build_running: bool = False
build_stage: str = ""
build_message: str = ""
build_progress: float | None = None
build_items_processed: int = 0
build_items_total: int = 0
build_current_library: str = ""
build_library_index: int = 0
build_libraries_total: int = 0
build_library_progress: float | None = None
build_library_items_processed: int = 0
build_library_items_total: int = 0
build_elapsed_seconds: float | None = None
build_eta_seconds: float | None = None
build_library_elapsed_seconds: float | None = None
build_library_eta_seconds: float | None = None
build_cancel_requested: bool = False
build_pid: int | None = None
build_error: str = ""
class MediaIndex:
"""SQLite-backed media inventory.
This class is UI-framework independent. Streamlit, a future FastAPI backend,
or a React-facing API can all use this service.
"""
def __init__(self, db_path: Path | str = DEFAULT_INDEX_PATH):
self.db_path = Path(db_path)
self.db_path.parent.mkdir(parents=True, exist_ok=True)
def connect(self) -> sqlite3.Connection:
"""Open a sqlite connection configured to return Row objects."""
conn = sqlite3.connect(self.db_path, timeout=30)
conn.row_factory = sqlite3.Row
conn.execute("PRAGMA journal_mode=WAL")
conn.execute("PRAGMA busy_timeout=30000")
return conn
def init_schema(self) -> None:
"""Create tables/indexes if this is the first use of the index."""
with self.connect() as conn:
conn.executescript(
"""
CREATE TABLE IF NOT EXISTS media_items (
id TEXT PRIMARY KEY,
title TEXT,
series TEXT,
season TEXT,
season_number INTEGER,
episode INTEGER,
type TEXT,
year INTEGER,
runtime_ticks INTEGER,
runtime_min INTEGER,
size_bytes INTEGER,
bitrate_bps INTEGER,
hdr INTEGER,
video TEXT,
width INTEGER,
height INTEGER,
resolution TEXT,
date_added TEXT,
date_added_ts INTEGER,
path TEXT,
library_id TEXT,
library_name TEXT
);
CREATE TABLE IF NOT EXISTS index_metadata (
key TEXT PRIMARY KEY,
value TEXT
);
CREATE INDEX IF NOT EXISTS idx_media_type ON media_items(type);
CREATE INDEX IF NOT EXISTS idx_media_library ON media_items(library_id);
CREATE INDEX IF NOT EXISTS idx_media_title ON media_items(title COLLATE NOCASE);
CREATE INDEX IF NOT EXISTS idx_media_series ON media_items(series COLLATE NOCASE);
CREATE INDEX IF NOT EXISTS idx_media_date_added ON media_items(date_added_ts);
CREATE INDEX IF NOT EXISTS idx_media_size ON media_items(size_bytes);
CREATE INDEX IF NOT EXISTS idx_media_bitrate ON media_items(bitrate_bps);
"""
)
def set_metadata(self, key: str, value: str | int | float) -> None:
"""Store a small string metadata value, e.g. build duration."""
self.init_schema()
with self.connect() as conn:
conn.execute(
"INSERT OR REPLACE INTO index_metadata (key, value) VALUES (?, ?)",
(key, str(value)),
)
def replace_items(self, rows: Iterable[dict[str, Any]]) -> int:
"""Atomically replace indexed media rows with a freshly built set."""
self.init_schema()
row_list = list(rows)
columns = [
"id",
"title",
"series",
"season",
"season_number",
"episode",
"type",
"year",
"runtime_ticks",
"runtime_min",
"size_bytes",
"bitrate_bps",
"hdr",
"video",
"width",
"height",
"resolution",
"date_added",
"date_added_ts",
"path",
"library_id",
"library_name",
]
placeholders = ",".join(["?"] * len(columns))
with self.connect() as conn:
conn.execute("DELETE FROM media_items")
conn.executemany(
f"INSERT OR REPLACE INTO media_items ({','.join(columns)}) VALUES ({placeholders})",
[[row.get(column) for column in columns] for row in row_list],
)
conn.execute(
"INSERT OR REPLACE INTO index_metadata (key, value) VALUES ('updated_at', ?)",
(str(int(time.time())),),
)
return len(row_list)
def status(self) -> MediaIndexStatus:
"""Return existence, count, update time, and last build duration."""
if not self.db_path.exists():
return MediaIndexStatus(exists=False)
try:
with self.connect() as conn:
item_count = int(conn.execute("SELECT COUNT(*) FROM media_items").fetchone()[0])
meta = {
row[0]: row[1]
for row in conn.execute("SELECT key, value FROM index_metadata").fetchall()
}
except sqlite3.Error:
return MediaIndexStatus(exists=False)
updated_at_raw = meta.get("updated_at", "")
updated_at = int(updated_at_raw) if str(updated_at_raw).isdigit() else None
label = time.strftime("%Y-%m-%d %H:%M:%S", time.localtime(updated_at)) if updated_at else ""
duration_raw = meta.get("build_duration_seconds")
build_duration = None
if duration_raw is not None:
try:
build_duration = float(duration_raw)
except (TypeError, ValueError):
build_duration = None
def _bool(key: str, default: bool = False) -> bool:
value = str(meta.get(key, str(default))).strip().lower()
return value in {"1", "true", "yes", "on"}
def _int(key: str, default: int = 0) -> int:
value = meta.get(key, default)
try:
return int(value)
except (TypeError, ValueError):
return default
def _float(key: str) -> float | None:
value = meta.get(key)
if value in (None, ""):
return None
try:
return float(value)
except (TypeError, ValueError):
return None
return MediaIndexStatus(
exists=True,
item_count=item_count,
updated_at=updated_at,
updated_at_label=label,
build_duration_seconds=build_duration,
build_running=_bool("build_running"),
build_stage=str(meta.get("build_stage", "")),
build_message=str(meta.get("build_message", "")),
build_progress=_float("build_progress"),
build_items_processed=_int("build_items_processed"),
build_items_total=_int("build_items_total"),
build_current_library=str(meta.get("build_current_library", "")),
build_library_index=_int("build_library_index"),
build_libraries_total=_int("build_libraries_total"),
build_library_progress=_float("build_library_progress"),
build_library_items_processed=_int("build_library_items_processed"),
build_library_items_total=_int("build_library_items_total"),
build_elapsed_seconds=_float("build_elapsed_seconds"),
build_eta_seconds=_float("build_eta_seconds"),
build_library_elapsed_seconds=_float("build_library_elapsed_seconds"),
build_library_eta_seconds=_float("build_library_eta_seconds"),
build_cancel_requested=_bool("build_cancel_requested"),
build_pid=_int("build_pid") or None,
build_error=str(meta.get("build_error", "")),
)
def query(
self,
library_id: str | None = None,
library_ids: list[str] | None = None,
media_types: list[str] | None = None,
search: str = "",
hdr_filter: str = "All",
sort_key: str = "title",
sort_order: str = "Ascending",
limit: int = 100,
offset: int = 0,
) -> tuple[list[dict[str, Any]], int]:
"""Query indexed media with full-index filters, sorting, and pagination."""
self.init_schema()
where = []
params: list[Any] = []
if library_ids:
where.append("library_id IN (" + ",".join(["?"] * len(library_ids)) + ")")
params.extend(library_ids)
elif library_id:
where.append("library_id = ?")
params.append(library_id)
if media_types:
where.append("type IN (" + ",".join(["?"] * len(media_types)) + ")")
params.extend(media_types)
if search:
needle = f"%{search.lower()}%"
where.append("(LOWER(title) LIKE ? OR LOWER(series) LIKE ? OR LOWER(path) LIKE ?)")
params.extend([needle, needle, needle])
if hdr_filter == "HDR only":
where.append("hdr = 1")
elif hdr_filter == "SDR/unknown only":
where.append("(hdr IS NULL OR hdr = 0)")
where_sql = " WHERE " + " AND ".join(where) if where else ""
sort_sql = SORT_COLUMNS.get(sort_key, SORT_COLUMNS["title"])
direction = "DESC" if sort_order == "Descending" else "ASC"
# Always add stable tie-breakers.
order_sql = f" ORDER BY {sort_sql} {direction}, series COLLATE NOCASE ASC, season_number ASC, episode ASC, title COLLATE NOCASE ASC"
with self.connect() as conn:
total = int(conn.execute("SELECT COUNT(*) FROM media_items" + where_sql, params).fetchone()[0])
rows = conn.execute(
"SELECT * FROM media_items" + where_sql + order_sql + " LIMIT ? OFFSET ?",
[*params, int(limit), int(offset)],
).fetchall()
return [display_media_row(dict(row)) for row in rows], total
def build_media_index(
client: JellyfinClient,
user_id: str,
libraries: list[dict[str, Any]],
index: MediaIndex | None = None,
page_size: int = 500,
media_root: str = "",
fallback_prefix: str = "",
progress_callback: Callable[[dict[str, Any]], None] | None = None,
should_cancel: Callable[[], bool] | None = None,
) -> int:
"""Fetch Jellyfin pages for all selected libraries and rebuild the index."""
index = index or MediaIndex()
started_at = time.perf_counter()
normalized_rows: list[dict[str, Any]] = []
processed_total = 0
expected_total = 0
current_library_name = ""
current_library_index = 0
current_library_processed = 0
current_library_total = 0
current_library_started_at = started_at
def ensure_not_cancelled() -> None:
if should_cancel and should_cancel():
raise MediaIndexBuildCancelled()
def emit(stage: str, message: str) -> None:
if not progress_callback:
return
elapsed_seconds = time.perf_counter() - started_at
library_elapsed_seconds = time.perf_counter() - current_library_started_at
overall_progress = (processed_total / expected_total) if expected_total else None
library_progress = (current_library_processed / current_library_total) if current_library_total else None
progress_callback(
{
"stage": stage,
"message": message,
"processed": processed_total,
"total": expected_total,
"progress": overall_progress,
"elapsed_seconds": elapsed_seconds,
"eta_seconds": _estimate_remaining_seconds(elapsed_seconds, overall_progress),
"library": current_library_name,
"library_index": current_library_index,
"libraries_total": len(libraries),
"library_processed": current_library_processed,
"library_total": current_library_total,
"library_progress": library_progress,
"library_elapsed_seconds": library_elapsed_seconds if current_library_total else None,
"library_eta_seconds": _estimate_remaining_seconds(library_elapsed_seconds, library_progress),
}
)
ensure_not_cancelled()
logger.info("Media index build starting libraries=%s page_size=%s", len(libraries), page_size)
emit("starting", "Starting media index build")
for library_index, library in enumerate(libraries, start=1):
library_id = library.get("Id")
current_library_name = library.get("Name", "")
current_library_index = library_index
current_library_processed = 0
current_library_total = 0
current_library_started_at = time.perf_counter()
if not library_id:
continue
ensure_not_cancelled()
logger.info(
"Media index scanning library index=%s/%s name=%s id=%s",
library_index,
len(libraries),
current_library_name or "Library",
library_id,
)
emit("library-starting", f"Scanning {current_library_name or 'Library'}")
start = 0
discovered_library_total = None
while True:
ensure_not_cancelled()
response = client.items(
user_id=user_id,
parent_id=library_id,
start_index=start,
limit=page_size,
include_item_types=MEDIA_TYPES,
recursive=True,
sort_by="SortName",
sort_order="Ascending",
)
ensure_not_cancelled()
items = response.get("Items", [])
if discovered_library_total is None:
discovered_library_total = int(response.get("TotalRecordCount", len(items)))
current_library_total = max(discovered_library_total, 0)
expected_total += current_library_total
normalized_rows.extend(
{
**row,
"path": resolve_remote_media_path(row.get("path", ""), media_root, fallback_prefix),
}
for row in (
normalize_media_item(item, library_id, current_library_name)
for item in items
)
)
processed_total += len(items)
current_library_processed += len(items)
start += len(items)
ensure_not_cancelled()
emit(
"building",
f"{current_library_name or 'Library'}: {current_library_processed} / {current_library_total or '?'} items",
)
logger.debug(
"Media index progress library=%s processed=%s/%s total_processed=%s",
current_library_name or "Library",
current_library_processed,
current_library_total,
processed_total,
)
total = int(response.get("TotalRecordCount", start))
if not items or start >= total:
break
ensure_not_cancelled()
logger.info("Media index finalizing rows=%s", len(normalized_rows))
emit("finalizing", "Writing index to disk")
ensure_not_cancelled()
count = index.replace_items(normalized_rows)
duration = time.perf_counter() - started_at
index.set_metadata("build_duration_seconds", f"{duration:.3f}")
processed_total = count
current_library_processed = current_library_total
emit("completed", f"Indexed {count} items in {duration:.1f}s")
logger.info("Media index build completed count=%s duration=%.2fs", count, duration)
return count
from media_library_viewer_api.services.media_index_impl import * # noqa: F401,F403