mirror of
https://github.com/myronblair/kino-app
synced 2026-07-27 21:18:43 -05:00
Fix Library Cleanup scan freezing the entire server
scan() was doing hundreds of blocking Path.is_file()/stat() calls directly in an async function -- movie/episode/track files often live on the NFS/CIFS-mounted NAS, so each stat call has real network latency. Since the backend runs a single-threaded asyncio event loop, this froze every other request (confirmed live: a plain GET /api/movies hung for 60s+ while a cleanup scan was in flight) for as long as the scan took. Moved every filesystem-touching check (missing files, orphaned HLS dirs) into a plain sync function run via asyncio.to_thread, so the scan can no longer block anything else.
This commit is contained in:
+44
-23
@@ -7,6 +7,8 @@ from datetime import datetime, timezone
|
|||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Dict, List
|
from typing import Dict, List
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
|
||||||
|
|
||||||
def _file_path(storage_type: str, storage_path: str, videos_dir: Path) -> Path:
|
def _file_path(storage_type: str, storage_path: str, videos_dir: Path) -> Path:
|
||||||
if storage_type in ("radarr", "sonarr", "scan"):
|
if storage_type in ("radarr", "sonarr", "scan"):
|
||||||
@@ -14,44 +16,63 @@ def _file_path(storage_type: str, storage_path: str, videos_dir: Path) -> Path:
|
|||||||
return videos_dir / storage_path
|
return videos_dir / storage_path
|
||||||
|
|
||||||
|
|
||||||
|
def _scan_filesystem(movies: List[Dict], episodes: List[Dict], tracks: List[Dict],
|
||||||
|
videos_dir: Path, hls_dir: Path, live_ids: set) -> Dict[str, List[Dict]]:
|
||||||
|
"""Every filesystem stat call in the whole scan lives here, run off the event loop via
|
||||||
|
asyncio.to_thread — movie/episode/track paths are often on an NFS/CIFS-mounted NAS, and
|
||||||
|
doing hundreds of blocking stats directly in an async function would freeze the entire
|
||||||
|
server (every other request) for as long as the scan takes."""
|
||||||
|
out: Dict[str, List[Dict]] = {"missing_files": [], "orphaned_hls": []}
|
||||||
|
|
||||||
|
for m in movies:
|
||||||
|
if m.get("storage_type") not in ("local", "radarr", "scan") or not m.get("storage_path"):
|
||||||
|
continue
|
||||||
|
if not _file_path(m["storage_type"], m["storage_path"], videos_dir).is_file():
|
||||||
|
out["missing_files"].append({"id": m["id"], "content_type": "movie", "label": m.get("title", "Untitled"), "detail": m["storage_path"]})
|
||||||
|
|
||||||
|
for e in episodes:
|
||||||
|
if e.get("_orphaned"):
|
||||||
|
continue # already reported as orphaned_episodes; don't also report a missing file
|
||||||
|
if e.get("storage_type") not in ("local", "sonarr", "scan") or not e.get("storage_path"):
|
||||||
|
continue
|
||||||
|
label = f"S{e.get('season_number')}E{e.get('episode_number')} {e.get('title') or ''}".strip()
|
||||||
|
if not _file_path(e["storage_type"], e["storage_path"], videos_dir).is_file():
|
||||||
|
out["missing_files"].append({"id": e["id"], "content_type": "episode", "label": label, "detail": e["storage_path"]})
|
||||||
|
|
||||||
|
for t in tracks:
|
||||||
|
if not Path(t["storage_path"]).is_file():
|
||||||
|
out["missing_files"].append({"id": t["id"], "content_type": "track", "label": f"{t.get('artist','')} — {t.get('title','Untitled')}", "detail": t["storage_path"]})
|
||||||
|
|
||||||
|
if hls_dir.is_dir():
|
||||||
|
for d in hls_dir.iterdir():
|
||||||
|
if d.is_dir() and d.name not in live_ids:
|
||||||
|
size = sum(f.stat().st_size for f in d.rglob("*") if f.is_file())
|
||||||
|
out["orphaned_hls"].append({"id": d.name, "content_type": "hls_dir", "label": d.name, "detail": f"{round(size / 1_000_000)} MB reclaimable"})
|
||||||
|
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
async def scan(db, videos_dir: Path, hls_dir: Path) -> Dict[str, List[Dict]]:
|
async def scan(db, videos_dir: Path, hls_dir: Path) -> Dict[str, List[Dict]]:
|
||||||
findings: Dict[str, List[Dict]] = {
|
findings: Dict[str, List[Dict]] = {
|
||||||
"missing_files": [], "orphaned_episodes": [], "orphaned_hls": [],
|
"missing_files": [], "orphaned_episodes": [], "orphaned_hls": [],
|
||||||
"stuck_jobs": [], "duplicate_paths": [], "orphaned_subtitles": [], "dangling_refs": [],
|
"stuck_jobs": [], "duplicate_paths": [], "orphaned_subtitles": [], "dangling_refs": [],
|
||||||
}
|
}
|
||||||
|
|
||||||
# --- Missing files ---
|
|
||||||
movies = await db.movies.find({}, {"_id": 0, "id": 1, "title": 1, "storage_type": 1, "storage_path": 1}).to_list(10000)
|
movies = await db.movies.find({}, {"_id": 0, "id": 1, "title": 1, "storage_type": 1, "storage_path": 1}).to_list(10000)
|
||||||
for m in movies:
|
|
||||||
if m.get("storage_type") not in ("local", "radarr", "scan") or not m.get("storage_path"):
|
|
||||||
continue
|
|
||||||
if not _file_path(m["storage_type"], m["storage_path"], videos_dir).is_file():
|
|
||||||
findings["missing_files"].append({"id": m["id"], "content_type": "movie", "label": m.get("title", "Untitled"), "detail": m["storage_path"]})
|
|
||||||
|
|
||||||
episodes = await db.episodes.find({}, {"_id": 0, "id": 1, "title": 1, "show_id": 1, "season_number": 1, "episode_number": 1, "storage_type": 1, "storage_path": 1}).to_list(20000)
|
episodes = await db.episodes.find({}, {"_id": 0, "id": 1, "title": 1, "show_id": 1, "season_number": 1, "episode_number": 1, "storage_type": 1, "storage_path": 1}).to_list(20000)
|
||||||
|
tracks = await db.tracks.find({}, {"_id": 0, "id": 1, "title": 1, "artist": 1, "storage_path": 1}).to_list(20000)
|
||||||
show_ids = set(await db.shows.distinct("id"))
|
show_ids = set(await db.shows.distinct("id"))
|
||||||
|
|
||||||
for e in episodes:
|
for e in episodes:
|
||||||
label = f"S{e.get('season_number')}E{e.get('episode_number')} {e.get('title') or ''}".strip()
|
label = f"S{e.get('season_number')}E{e.get('episode_number')} {e.get('title') or ''}".strip()
|
||||||
if e.get("show_id") not in show_ids:
|
if e.get("show_id") not in show_ids:
|
||||||
findings["orphaned_episodes"].append({"id": e["id"], "content_type": "episode", "label": label, "detail": f"show_id {e.get('show_id')} not found"})
|
findings["orphaned_episodes"].append({"id": e["id"], "content_type": "episode", "label": label, "detail": f"show_id {e.get('show_id')} not found"})
|
||||||
continue
|
e["_orphaned"] = True
|
||||||
if e.get("storage_type") not in ("local", "sonarr", "scan") or not e.get("storage_path"):
|
|
||||||
continue
|
|
||||||
if not _file_path(e["storage_type"], e["storage_path"], videos_dir).is_file():
|
|
||||||
findings["missing_files"].append({"id": e["id"], "content_type": "episode", "label": label, "detail": e["storage_path"]})
|
|
||||||
|
|
||||||
tracks = await db.tracks.find({}, {"_id": 0, "id": 1, "title": 1, "artist": 1, "storage_path": 1}).to_list(20000)
|
|
||||||
for t in tracks:
|
|
||||||
if not Path(t["storage_path"]).is_file():
|
|
||||||
findings["missing_files"].append({"id": t["id"], "content_type": "track", "label": f"{t.get('artist','')} — {t.get('title','Untitled')}", "detail": t["storage_path"]})
|
|
||||||
|
|
||||||
# --- Orphaned HLS output dirs (no matching movie or episode id) ---
|
|
||||||
live_ids = {m["id"] for m in movies} | {e["id"] for e in episodes}
|
live_ids = {m["id"] for m in movies} | {e["id"] for e in episodes}
|
||||||
if hls_dir.is_dir():
|
fs_findings = await asyncio.to_thread(_scan_filesystem, movies, episodes, tracks, videos_dir, hls_dir, live_ids)
|
||||||
for d in hls_dir.iterdir():
|
findings["missing_files"] = fs_findings["missing_files"]
|
||||||
if d.is_dir() and d.name not in live_ids:
|
findings["orphaned_hls"] = fs_findings["orphaned_hls"]
|
||||||
size = sum(f.stat().st_size for f in d.rglob("*") if f.is_file())
|
|
||||||
findings["orphaned_hls"].append({"id": d.name, "content_type": "hls_dir", "label": d.name, "detail": f"{round(size / 1_000_000)} MB reclaimable"})
|
|
||||||
|
|
||||||
# --- Stuck transcode jobs (running with no progress for 2+ hours — likely dead from a restart) ---
|
# --- Stuck transcode jobs (running with no progress for 2+ hours — likely dead from a restart) ---
|
||||||
now = datetime.now(timezone.utc)
|
now = datetime.now(timezone.utc)
|
||||||
|
|||||||
Reference in New Issue
Block a user