diff --git a/Dockerfile b/Dockerfile new file mode 100644 index 0000000..d7b47dc --- /dev/null +++ b/Dockerfile @@ -0,0 +1,41 @@ +# Multi-stage so build deps don't ship in the runtime image. +FROM python:3.12-slim AS build + +WORKDIR /build +COPY requirements.txt . +RUN pip install --no-cache-dir --prefix=/install -r requirements.txt + + +FROM python:3.12-slim + +LABEL org.opencontainers.image.title="MediaShelf" \ + org.opencontainers.image.description="Plex library analytics and reclaim reporting" \ + org.opencontainers.image.source="https://gitlab.hallsworth.ca/yrtria/MediaShelf" + +COPY --from=build /install /usr/local + +# Non-root. /data is the only writable path the app needs. +RUN useradd --create-home --uid 10001 mediashelf \ + && mkdir -p /data && chown mediashelf:mediashelf /data + +WORKDIR /app +COPY --chown=mediashelf:mediashelf mediashelf/ ./mediashelf/ +COPY --chown=mediashelf:mediashelf wsgi.py ./ + +USER mediashelf +VOLUME ["/data"] +EXPOSE 8080 + +ENV PYTHONUNBUFFERED=1 \ + PYTHONDONTWRITEBYTECODE=1 \ + DATABASE_PATH=/data/mediashelf.db + +HEALTHCHECK --interval=60s --timeout=10s --start-period=20s --retries=3 \ + CMD python -c "import urllib.request,sys; sys.exit(0 if urllib.request.urlopen('http://localhost:8080/healthz', timeout=8).status==200 else 1)" + +# Two workers: enough for one person browsing while a scan runs. The scheduler +# starts in exactly one of them, guarded by an flock on /data (see §11.4). +# --preload is deliberately NOT used: it would run the app factory before the +# fork, so both workers would inherit one already-open SQLite connection. +CMD ["gunicorn", "--workers", "2", "--threads", "4", "--bind", "0.0.0.0:8080", \ + "--timeout", "120", "--access-logfile", "-", "wsgi:app"] diff --git a/Makefile b/Makefile new file mode 100644 index 0000000..f8be945 --- /dev/null +++ b/Makefile @@ -0,0 +1,24 @@ +.PHONY: test lint run scan docker smoke clean + +test: + python3 -m pytest tests/ -q + +run: + python3 -m mediashelf.cli serve --port 8080 --no-scheduler + +scan: + python3 -m mediashelf.cli scan --full + +docker: + docker build -t mediashelf:$(or $(VERSION),0.1.0) . + +# The human sanity check no unit test replaces: a real scan, real top candidates. +smoke: + python3 -m mediashelf.cli scan --full + @echo "--- top 20 reclaim candidates ---" + @python3 -c "import json,urllib.request; \ +d=json.load(urllib.request.urlopen('http://127.0.0.1:8080/api/v1/items?page_size=20')); \ +[print('%8.1f GB score %-6s %s' % (i['size_bytes']/1e9, i['reclaim_score'], i['title'])) for i in d['items']]" + +clean: + rm -rf .pytest_cache **/__pycache__ *.egg-info diff --git a/README.md b/README.md index ace30a7..4babdf4 100644 --- a/README.md +++ b/README.md @@ -12,15 +12,46 @@ candidates. ## Status -**v1 is report-only.** MediaShelf does not delete, move, or modify anything. It produces -a ranked list, saved rule sets, and CSV export. Deletion is designed for in the roadmap -but deliberately not built, so the scanner and the scoring model can be trusted before -anything destructive is wired up. +**v1 is report-only.** MediaShelf does not delete, move, or modify anything — not a file, +not a Plex record. It produces a ranked list, saved rule sets, and CSV export. Deletion is +designed for in the roadmap but deliberately not built, so the scanner and the scoring +model can be trusted before anything destructive is wired up. -Nothing is implemented yet — this repository holds the design plus the read-only tools -used to validate it. The design has been checked against the live servers: 65.7 TB across -24 libraries, 2,930 movies and 2,807 TV seasons, 87,640 logged plays from 60 users, and -27.4 TB never played. See `docs/design.md` §2.1. +The application is implemented and tested; it has not yet been deployed. The design was +validated against the live servers first: 65.7 TB across 24 libraries, 2,930 movies and +2,807 TV seasons, 87,640 logged plays from 60 users, and 27.4 TB never played. See +`docs/design.md` §2.1. + +## Running it + +```bash +pip install -r requirements-dev.txt +cp .env.example .env # fill in PLEX_TOKEN and TAUTULLI_API_KEY + +python3 -m mediashelf.cli scan --full # first ingest +python3 -m mediashelf.cli serve # http://127.0.0.1:8080 +make test # 73 tests, no live server needed +``` + +Deployment is a single container behind Nginx Proxy Manager — see `docs/design.md` §11. +Build off-box and push the image to Nox's Portainer; `docker-compose.yml` is the stack. + +## Layout + +``` +mediashelf/ + config.py every setting, read once from the environment + db.py SQLite (WAL) + numbered SQL migrations + providers/ Plex (library) and Tautulli (history) behind two protocols + ingest.py scan orchestration, rollups, keep resolution + scoring.py the reclaim score, in SQL and in Python + rules.py saved-view grammar -> parameterized SQL, whitelist only + keeps.py GUID-keyed keep marks + queries.py the item query, live score, dashboard aggregates + api.py / web.py JSON API and the page shell +tools/ read-only probe, reclaim preview, fake Plex+Tautulli +tests/ 73 tests, run entirely against the fake server +``` ## What it does @@ -42,10 +73,10 @@ used to validate it. The design has been checked against the live servers: 65.7 GUIDs rather than rating keys, so they survive a library rebuild - CSV export of any view -## Planned stack +## Stack -Python + Flask, SQLite (WAL), vanilla JS front-end, single container deployed as a -Portainer stack behind Nginx Proxy Manager. +Python + Flask, SQLite (WAL), vanilla JS front-end (no build step), single container +deployed as a Portainer stack behind Nginx Proxy Manager. ## Roadmap @@ -53,9 +84,9 @@ Portainer stack behind Nginx Proxy Manager. allowlist, and an audit log - **v3** — Emby and Jellyfin support behind the existing `MediaProvider` abstraction -## Validating the design first +## Read-only LAN tools -Plex and Tautulli are both LAN-only, so `tools/probe.py` exists to check this design +Plex and Tautulli are both LAN-only, so these exist to check the design and the numbers against real data from inside the network. It is **read-only** — GET requests only, nothing is modified — and has no dependencies beyond the standard library. diff --git a/docker-compose.yml b/docker-compose.yml new file mode 100644 index 0000000..551866c --- /dev/null +++ b/docker-compose.yml @@ -0,0 +1,61 @@ +# Portainer stack for Nox. See docs/design.md §11.3. +# +# Every variable the app reads MUST be listed under `environment:` here. +# A variable set in Portainer's UI but absent from this file silently does +# nothing — the trap that bit PLAYERMAP_SHOW_GMS on the Mythica stack. +# +# The image is built off-box and pushed/loaded onto Nox (§11.2). Pin a real +# version tag; never deploy bare :latest. + +services: + mediashelf: + image: ${MEDIASHELF_IMAGE:-mediashelf:0.1.0} + container_name: mediashelf + restart: unless-stopped + ports: + - "${MEDIASHELF_PORT:-8085}:8080" + volumes: + - mediashelf_data:/data + environment: + - MEDIASHELF_SECRET_KEY=${MEDIASHELF_SECRET_KEY} + - LOG_LEVEL=${LOG_LEVEL:-INFO} + - TZ=${TZ:-America/Regina} + + - PLEX_BASE_URL=${PLEX_BASE_URL} + - PLEX_TOKEN=${PLEX_TOKEN} + - PLEX_VERIFY_SSL=${PLEX_VERIFY_SSL:-true} + - PLEX_TIMEOUT_S=${PLEX_TIMEOUT_S:-30} + - PLEX_PAGE_SIZE=${PLEX_PAGE_SIZE:-500} + - PLEX_REQUEST_DELAY_MS=${PLEX_REQUEST_DELAY_MS:-0} + + - TAUTULLI_BASE_URL=${TAUTULLI_BASE_URL} + - TAUTULLI_API_KEY=${TAUTULLI_API_KEY} + - TAUTULLI_TIMEOUT_S=${TAUTULLI_TIMEOUT_S:-30} + - TAUTULLI_PAGE_SIZE=${TAUTULLI_PAGE_SIZE:-1000} + + - HISTORY_SOURCE=${HISTORY_SOURCE:-auto} + - SESSION_MERGE_WINDOW_H=${SESSION_MERGE_WINDOW_H:-6} + - COMPLETION_THRESHOLD=${COMPLETION_THRESHOLD:-85} + - ABANDON_CEILING=${ABANDON_CEILING:-15} + + - DATABASE_PATH=/data/mediashelf.db + + - SCAN_SCHEDULE_CRON=${SCAN_SCHEDULE_CRON:-0 4 * * *} + - SCAN_FULL_SWEEP_CRON=${SCAN_FULL_SWEEP_CRON:-0 3 * * 0} + - SCAN_ON_STARTUP=${SCAN_ON_STARTUP:-false} + - SCAN_LOCK_TIMEOUT_S=${SCAN_LOCK_TIMEOUT_S:-7200} + - SCHEDULER_ENABLED=${SCHEDULER_ENABLED:-true} + + - SCORE_STALE_HORIZON_DAYS=${SCORE_STALE_HORIZON_DAYS:-730} + - SCORE_AGE_HORIZON_DAYS=${SCORE_AGE_HORIZON_DAYS:-1095} + - SCORE_POPULAR_AT=${SCORE_POPULAR_AT:-3} + - SCORE_REJECTED_AT=${SCORE_REJECTED_AT:-2} + - SCORE_SOLITUDE_AT=${SCORE_SOLITUDE_AT:-3} + - SCORE_GRACE_DAYS=${SCORE_GRACE_DAYS:-30} + - SCORE_RECENT_DAYS=${SCORE_RECENT_DAYS:-90} + + # Empty by default. Nothing is ever kept unless someone says so (§6.6). + - KEEP_ALL_LIBRARIES=${KEEP_ALL_LIBRARIES:-} + +volumes: + mediashelf_data: diff --git a/mediashelf/__init__.py b/mediashelf/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/mediashelf/api.py b/mediashelf/api.py new file mode 100644 index 0000000..246a7a5 --- /dev/null +++ b/mediashelf/api.py @@ -0,0 +1,497 @@ +"""JSON API (§8). No auth in v1 — LAN-only, single admin view (§12).""" + +from __future__ import annotations + +import csv +import io +import json +import logging +import time + +from flask import Blueprint, Response, current_app, jsonify, request + +from . import keeps, queries, scanner +from .rules import RuleError +from .keeps import KeepError + +log = logging.getLogger(__name__) + +bp = Blueprint("api", __name__, url_prefix="/api/v1") + + +def db(): + return current_app.extensions["mediashelf"]["db"] + + +def cfg(): + return current_app.extensions["mediashelf"]["config"] + + +@bp.errorhandler(RuleError) +def _rule_error(e): + return jsonify({"error": "invalid_rule", "message": str(e)}), 400 + + +@bp.errorhandler(KeepError) +def _keep_error(e): + return jsonify({"error": "invalid_keep", "message": str(e)}), 400 + + +def _json_arg(name): + raw = request.args.get(name) + if not raw: + return None + try: + return json.loads(raw) + except (TypeError, ValueError): + raise RuleError("%s must be valid JSON" % name) + + +def _bool_arg(name, default=False): + v = request.args.get(name) + if v is None: + return default + return v.lower() in ("1", "true", "yes") + + +def _query_kwargs(): + rule_group = _json_arg("rules") + weights = _json_arg("weights") + + view_id = request.args.get("view_id") + sort = request.args.get("sort") + if view_id: + row = db().one("SELECT * FROM saved_view WHERE id = ?", (view_id,)) + if row is None: + raise RuleError("no such view") + rule_group = json.loads(row["rules"]) if row["rules"] else None + sort = sort or row["sort"] + if row["weights"] and not weights: + weights = json.loads(row["weights"]) + + return dict( + library_ids=request.args.getlist("library_id") or None, + kinds=request.args.getlist("kind") or None, + q=request.args.get("q"), + rule_group=rule_group, + include_missing=_bool_arg("include_missing"), + include_kept=_bool_arg("include_kept"), + include_shows=_bool_arg("include_shows"), + weights=weights, + ), sort + + +# ── items ──────────────────────────────────────────────────────────────── + +@bp.get("/items") +def items(): + kw, sort = _query_kwargs() + q = queries.Query(db(), cfg()) + return jsonify(q.page( + sort=sort, + page=int(request.args.get("page", 1)), + page_size=int(request.args.get("page_size", 100)), + **kw, + )) + + +@bp.get("/items/") +def item_detail(item_id): + d = db() + q = queries.Query(d, cfg()) + expr, where, params = q.build(include_kept=True, include_missing=True, include_shows=True) + params["item_id"] = item_id + sql = ("SELECT " + queries.BASE_COLUMNS + ", " + expr + " AS reclaim_score, " + + "NULL AS grace, COALESCE(d.dupe_count,1) AS duplicate_count " + + queries.FROM_CLAUSE + " WHERE i.id = :item_id") + row = d.one(sql, params) + if row is None: + return jsonify({"error": "not_found"}), 404 + + out = q.serialize(row) + out["parts"] = [dict(r) for r in d.query( + "SELECT * FROM media_part WHERE media_item_id = ? ORDER BY file_path", (item_id,))] + + if out["kind"] == "season": + out["episodes"] = [dict(r) for r in d.query( + "SELECT * FROM episode WHERE season_item_id = ? ORDER BY episode_number", + (item_id,))] + pids = [e["provider_item_id"] for e in out["episodes"]] + else: + pids = [d.scalar("SELECT provider_item_id FROM media_item WHERE id=?", (item_id,))] + + if pids: + marks = ",".join("?" * len(pids)) + out["watch_history"] = [dict(r) for r in d.query( + "SELECT w.viewed_at, w.percent_complete, w.disposition, w.account_id, " + "COALESCE(a.friendly_name, a.name, w.account_id) AS who, w.platform " + "FROM watch_event w LEFT JOIN account a ON a.account_id = w.account_id " + f"WHERE w.provider_item_id IN ({marks}) ORDER BY w.viewed_at DESC LIMIT 500", + tuple(pids))] + else: + out["watch_history"] = [] + + if out.get("guid"): + out["duplicates"] = [dict(r) for r in d.query( + "SELECT i.id, i.title, i.size_bytes, i.resolution, i.watch_count, " + "lib.title AS library_title FROM media_item i " + "JOIN library lib ON lib.id=i.library_id " + "WHERE i.guid = ? AND i.id != ? AND i.kind='movie' AND i.status='present'", + (out["guid"], item_id))] + return jsonify(out) + + +@bp.get("/libraries") +def libraries(): + return jsonify({"libraries": queries.size_by_library(db())}) + + +@bp.get("/accounts") +def accounts(): + return jsonify({"accounts": [dict(r) for r in db().query( + "SELECT a.*, (SELECT COUNT(*) FROM watch_event w WHERE w.account_id=a.account_id) " + "AS plays FROM account a ORDER BY plays DESC")]}) + + +# ── stats ──────────────────────────────────────────────────────────────── + +@bp.get("/stats/overview") +def stats_overview(): + return jsonify(queries.overview(db(), cfg())) + + +@bp.get("/stats/size-by-library") +def stats_size_by_library(): + return jsonify({"libraries": queries.size_by_library(db())}) + + +@bp.get("/stats/added-over-time") +def stats_added_over_time(): + return jsonify({"buckets": queries.added_over_time( + db(), request.args.get("bucket", "month"))}) + + +@bp.get("/stats/completion") +def stats_completion(): + return jsonify(queries.completion_split(db())) + + +@bp.get("/stats/size-vs-lastwatched") +def stats_scatter(): + return jsonify({"points": queries.size_vs_lastwatched(db())}) + + +@bp.get("/stats/watch-distribution") +def stats_watch_distribution(): + return jsonify({"buckets": [dict(r) for r in db().query(""" + SELECT CASE WHEN watch_count = 0 THEN '0' + WHEN watch_count = 1 THEN '1' + WHEN watch_count <= 3 THEN '2-3' + WHEN watch_count <= 10 THEN '4-10' + ELSE '10+' END AS bucket, + COUNT(*) AS items, COALESCE(SUM(size_bytes),0) AS size_bytes + FROM media_item WHERE kind IN ('movie','season') AND status='present' + GROUP BY bucket""")]}) + + +@bp.get("/duplicates") +def duplicates(): + groups = queries.duplicate_groups(db()) + return jsonify({ + "groups": groups, + "total_redundant_bytes": sum(g["redundant_bytes"] for g in groups), + "group_count": len(groups), + }) + + +@bp.get("/sources") +def sources(): + d = db() + c = cfg() + cov = [dict(r) for r in d.query("SELECT * FROM history_coverage")] + return jsonify({ + "plex": {"configured": c.plex_configured, "base_url": c.plex_base_url}, + "tautulli": {"configured": c.tautulli_configured, "base_url": c.tautulli_base_url}, + "history_source": (cov[0]["source"] if cov else None), + "has_completion_data": queries.history_has_completion(d), + "coverage": cov, + "providers": [dict(r) for r in d.query("SELECT * FROM provider")], + }) + + +# ── keep marks (§8.2) ──────────────────────────────────────────────────── + +@bp.get("/keeps") +def list_keeps(): + d = db() + counts = keeps.mark_matches(d) + rows = [] + for r in d.query( + "SELECT k.*, lib.title AS library_title FROM keep_mark k " + "JOIN library lib ON lib.id = k.library_id ORDER BY k.created_at DESC" + ): + item = dict(r) + item["resolved_items"] = counts.get(r["id"], 0) + item["resolved_bytes"] = d.scalar( + "SELECT COALESCE(SUM(size_bytes),0) FROM media_item WHERE kept_mark_id = ?", + (r["id"],)) or 0 + item["orphaned"] = item["resolved_items"] == 0 + rows.append(item) + libs = [dict(r) for r in d.query( + "SELECT id, title, keep_all FROM library WHERE keep_all = 1 ORDER BY title")] + return jsonify({ + "marks": rows, + "library_rules": libs, + "orphan_count": sum(1 for r in rows if r["orphaned"]), + "kept_bytes": d.scalar("SELECT COALESCE(SUM(size_bytes),0) FROM media_item " + "WHERE kept=1 AND kind IN ('movie','season')") or 0, + "kept_items": d.scalar("SELECT COUNT(*) FROM media_item " + "WHERE kept=1 AND kind IN ('movie','season')") or 0, + }) + + +@bp.get("/keeps/orphans") +def keep_orphans(): + d = db() + counts = keeps.mark_matches(d) + ids = [k for k, v in counts.items() if v == 0] + if not ids: + return jsonify({"marks": []}) + marks = ",".join("?" * len(ids)) + return jsonify({"marks": [dict(r) for r in d.query( + f"SELECT k.*, lib.title AS library_title FROM keep_mark k " + f"JOIN library lib ON lib.id=k.library_id WHERE k.id IN ({marks})", tuple(ids))]}) + + +@bp.post("/keeps") +def create_keep(): + payload = request.get_json(silent=True) or {} + d = db() + mode = payload.get("mode", "keep") + note = payload.get("note") + if payload.get("item_id"): + mark_id = keeps.create_from_item(d, int(payload["item_id"]), mode, note) + else: + required = ("scope", "library_id", "guid") + if not all(payload.get(k) for k in required): + raise KeepError("need item_id, or scope + library_id + guid") + mark_id = keeps.upsert( + d, payload["scope"], mode, int(payload["library_id"]), payload["guid"], + payload.get("season_number"), label=payload.get("label") or payload["guid"], + note=note) + keeps.resolve_all(d) + row = d.one("SELECT * FROM keep_mark WHERE id = ?", (mark_id,)) + counts = keeps.mark_matches(d) + out = dict(row) + out["resolved_items"] = counts.get(mark_id, 0) + out["resolved_bytes"] = d.scalar( + "SELECT COALESCE(SUM(size_bytes),0) FROM media_item WHERE kept_mark_id=?", + (mark_id,)) or 0 + return jsonify(out), 201 + + +@bp.post("/keeps/bulk") +def bulk_keep(): + payload = request.get_json(silent=True) or {} + ids = payload.get("item_ids") or [] + if not isinstance(ids, list) or not ids: + raise KeepError("item_ids must be a non-empty list") + if len(ids) > 5000: + raise KeepError("too many items in one request") + mode = payload.get("mode", "keep") + note = payload.get("note") + d = db() + created, failed = [], [] + for item_id in ids: + try: + created.append(keeps.create_from_item(d, int(item_id), mode, note)) + except KeepError as e: + failed.append({"item_id": item_id, "reason": str(e)}) + keeps.resolve_all(d) + return jsonify({ + "created": len(created), "failed": failed, + "kept_bytes": d.scalar("SELECT COALESCE(SUM(size_bytes),0) FROM media_item " + "WHERE kept=1 AND kind IN ('movie','season')") or 0, + }) + + +@bp.patch("/keeps/") +def patch_keep(mark_id): + payload = request.get_json(silent=True) or {} + db().execute("UPDATE keep_mark SET note=?, updated_at=? WHERE id=?", + (payload.get("note"), int(time.time()), mark_id)) + return jsonify({"ok": True}) + + +@bp.delete("/keeps/") +def delete_keep(mark_id): + d = db() + ok = keeps.delete(d, mark_id) + keeps.resolve_all(d) + return jsonify({"deleted": ok}), (200 if ok else 404) + + +@bp.put("/libraries//keep_all") +def library_keep_all(library_id): + payload = request.get_json(silent=True) or {} + d = db() + keeps.set_library_keep_all(d, library_id, bool(payload.get("keep_all"))) + keeps.resolve_all(d) + return jsonify({"ok": True, "library_id": library_id, + "keep_all": bool(payload.get("keep_all"))}) + + +@bp.get("/keeps/export") +def export_keeps(): + return Response(json.dumps(keeps.export(db()), indent=2), + mimetype="application/json", + headers={"Content-Disposition": "attachment; filename=keeps.json"}) + + +@bp.post("/keeps/import") +def import_keeps(): + payload = request.get_json(silent=True) or {} + return jsonify(keeps.import_(db(), payload)) + + +# ── saved views (§8.3) ─────────────────────────────────────────────────── + +@bp.get("/views") +def list_views(): + return jsonify({"views": [dict(r) for r in db().query( + "SELECT * FROM saved_view ORDER BY builtin DESC, name")]}) + + +@bp.post("/views") +def create_view(): + p = request.get_json(silent=True) or {} + if not p.get("name"): + return jsonify({"error": "name is required"}), 400 + now = int(time.time()) + cur = db().execute( + "INSERT INTO saved_view (name, description, rules, sort, columns, weights, " + "created_at, updated_at) VALUES (?,?,?,?,?,?,?,?)", + (p["name"], p.get("description"), json.dumps(p.get("rules") or {}), + p.get("sort"), json.dumps(p.get("columns") or []), + json.dumps(p.get("weights") or {}), now, now)) + return jsonify(dict(db().one("SELECT * FROM saved_view WHERE id=?", (cur.lastrowid,)))), 201 + + +@bp.get("/views/") +def get_view(view_id): + row = db().one("SELECT * FROM saved_view WHERE id=?", (view_id,)) + return (jsonify(dict(row)), 200) if row else (jsonify({"error": "not_found"}), 404) + + +@bp.put("/views/") +def update_view(view_id): + p = request.get_json(silent=True) or {} + db().execute( + "UPDATE saved_view SET name=COALESCE(?,name), description=?, rules=?, sort=?, " + "columns=?, weights=?, updated_at=? WHERE id=?", + (p.get("name"), p.get("description"), json.dumps(p.get("rules") or {}), + p.get("sort"), json.dumps(p.get("columns") or []), + json.dumps(p.get("weights") or {}), int(time.time()), view_id)) + return jsonify({"ok": True}) + + +@bp.delete("/views/") +def delete_view(view_id): + cur = db().execute("DELETE FROM saved_view WHERE id=? AND builtin=0", (view_id,)) + return jsonify({"deleted": cur.rowcount > 0}) + + +# ── scans (§8.4) ───────────────────────────────────────────────────────── + +@bp.get("/scans") +def list_scans(): + return jsonify({"scans": [dict(r) for r in db().query( + "SELECT * FROM scan ORDER BY id DESC LIMIT 50")]}) + + +@bp.get("/scans/current") +def current_scan(): + row = db().one("SELECT * FROM scan WHERE status='running' ORDER BY id DESC LIMIT 1") + return jsonify(dict(row) if row else None) + + +@bp.post("/scans") +def start_scan(): + p = request.get_json(silent=True) or {} + mode = p.get("mode", "incremental") + if mode not in ("full", "incremental"): + return jsonify({"error": "mode must be 'full' or 'incremental'"}), 400 + started = scanner.start_background_scan(current_app, mode, "manual") + if not started: + return jsonify({"error": "a scan is already running"}), 409 + return jsonify({"status": "started", "mode": mode}), 202 + + +# ── export (§8.5) ──────────────────────────────────────────────────────── + +CSV_COLUMNS = [ + "id", "kind", "library", "title", "show_title", "season_number", "year", + "size_bytes", "size_human", "added_at_iso", "last_watched_at_iso", + "watch_count", "partial_count", "abandoned_count", "distinct_watcher_count", + "episode_count", "part_count", "resolution", "duplicate_count", + "pre_history", "kept", "kept_via", "reclaim_score", "primary_path", +] + + +def _human(n): + n = float(n or 0) + for u in ("B", "KB", "MB", "GB", "TB", "PB"): + if abs(n) < 1024: + return "%.1f %s" % (n, u) + n /= 1024.0 + return "%.1f EB" % n + + +def _iso(ts): + if not ts: + return "" + return time.strftime("%Y-%m-%d", time.localtime(int(ts))) + + +@bp.get("/export.csv") +def export_csv(): + kw, sort = _query_kwargs() + d = db() + q = queries.Query(d, cfg()) + + def generate(): + buf = io.StringIO() + w = csv.writer(buf) + w.writerow(CSV_COLUMNS) + yield buf.getvalue() + buf.seek(0), buf.truncate(0) + for row in q.iter_all(sort=sort, **kw): + r = dict(row) + w.writerow([ + r["id"], r["kind"], r["library_title"], r["title"], + r.get("show_title") or "", r.get("season_number") or "", + r.get("year") or "", r.get("size_bytes") or 0, + _human(r.get("size_bytes")), _iso(r.get("added_at")), + _iso(r.get("last_watched_at")), r.get("watch_count") or 0, + r.get("partial_count") or 0, r.get("abandoned_count") or 0, + r.get("distinct_watcher_count") or 0, + r.get("episode_count") if r["kind"] == "season" else "", + r.get("part_count") or 0, r.get("resolution") or "", + r.get("duplicate_count") or 1, + "yes" if r.get("pre_history") else "no", + "yes" if r.get("kept") else "no", r.get("kept_via") or "", + r.get("reclaim_score"), r.get("primary_path") or "", + ]) + yield buf.getvalue() + buf.seek(0), buf.truncate(0) + + stamp = time.strftime("%Y%m%d-%H%M") + return Response(generate(), mimetype="text/csv", headers={ + "Content-Disposition": "attachment; filename=mediashelf-%s.csv" % stamp}) + + +# ── settings ───────────────────────────────────────────────────────────── + +@bp.get("/settings") +def settings(): + return jsonify(cfg().redacted()) diff --git a/mediashelf/app.py b/mediashelf/app.py new file mode 100644 index 0000000..1e7bc9c --- /dev/null +++ b/mediashelf/app.py @@ -0,0 +1,153 @@ +"""Flask app factory.""" + +from __future__ import annotations + +import json +import logging +import time + +from flask import Flask, jsonify, render_template + +from . import queries, scanner +from .config import Config +from .db import Database +from .scoring import register_sqlite_functions + +log = logging.getLogger(__name__) + +# Shipped read-only so the first run has somewhere useful to start (§7.2). +SEED_VIEWS = [ + ("Confident reclaim", + "Added while watch history was being recorded, and never played since. " + "The list to act on first.", + {"op": "and", "rules": [ + {"field": "watch_count", "op": "eq", "value": 0}, + {"field": "pre_history", "op": "eq", "value": False}]}, + "reclaim_score:desc"), + ("Uncertain reclaim", + "Never played, but added before history began — it may have been watched " + "and nobody can tell. Needs judgement, not a bulk action.", + {"op": "and", "rules": [ + {"field": "watch_count", "op": "eq", "value": 0}, + {"field": "pre_history", "op": "eq", "value": True}]}, + "size_bytes:desc"), + ("Never watched, large", + "Nothing has ever played it and it is over 8 GB.", + {"op": "and", "rules": [ + {"field": "watch_count", "op": "eq", "value": 0}, + {"field": "size_bytes", "op": "gte", "value": 8 * 1024**3}]}, + "size_bytes:desc"), + ("Cold storage", + "Not watched in over three years.", + {"op": "and", "rules": [ + {"field": "last_watched_at", "op": "older_than_days", "value": 1095}]}, + "size_bytes:desc"), + ("One-and-done movies", + "Watched exactly once, over two years ago.", + {"op": "and", "rules": [ + {"field": "kind", "op": "eq", "value": "movie"}, + {"field": "watch_count", "op": "eq", "value": 1}, + {"field": "last_watched_at", "op": "older_than_days", "value": 730}]}, + "size_bytes:desc"), + ("Tried and rejected", + "Two or more people started it and nobody finished it. Needs Tautulli.", + {"op": "and", "rules": [ + {"field": "abandoned_count", "op": "gte", "value": 2}, + {"field": "watch_count", "op": "eq", "value": 0}]}, + "size_bytes:desc"), + ("Abandoned seasons", + "Seasons barely watched through, added over a year ago.", + {"op": "and", "rules": [ + {"field": "kind", "op": "eq", "value": "season"}, + {"field": "watch_ratio", "op": "lt", "value": 0.3}, + {"field": "added_at", "op": "older_than_days", "value": 365}]}, + "size_bytes:desc"), + ("Biggest 100", + "Simply the largest things on disk.", + {}, "size_bytes:desc"), + ("Recently added", + "The sanity check, not a delete list.", + {"op": "and", "rules": [ + {"field": "added_at", "op": "newer_than_days", "value": 30}]}, + "added_at:desc"), + ("Kept", + "What you have decided to keep, and what it costs.", + {"op": "and", "rules": [{"field": "kept", "op": "eq", "value": True}]}, + "size_bytes:desc"), +] + + +def seed_views(db) -> None: + now = int(time.time()) + for name, desc, rules, sort in SEED_VIEWS: + if db.one("SELECT id FROM saved_view WHERE name = ?", (name,)): + continue + db.execute( + "INSERT INTO saved_view (name, description, rules, sort, columns, weights, " + "builtin, created_at, updated_at) VALUES (?,?,?,?,'[]','{}',1,?,?)", + (name, desc, json.dumps(rules), sort, now, now)) + + +def create_app(config: Config | None = None, *, start_scheduler: bool = True) -> Flask: + cfg = config or Config.from_env() + logging.basicConfig( + level=getattr(logging, cfg.log_level, logging.INFO), + format="%(asctime)s %(levelname)-7s %(name)s: %(message)s") + + app = Flask(__name__) + app.config["SECRET_KEY"] = cfg.secret_key + app.config["SESSION_COOKIE_SAMESITE"] = "Strict" + app.config["JSON_SORT_KEYS"] = False + + db = Database(cfg.database_path) + db.migrate() + register_sqlite_functions(db.conn) + seed_views(db) + + app.extensions["mediashelf"] = { + "config": cfg, "db": db, "degraded_reason": None, + "started_at": int(time.time()), + } + + @app.before_request + def _ensure_functions(): + # Each thread gets its own connection, so LOG() must be registered on it. + register_sqlite_functions(db.conn) + + from .api import bp as api_bp + app.register_blueprint(api_bp) + + from .web import bp as web_bp + app.register_blueprint(web_bp) + + @app.get("/healthz") + def healthz(): + ext = app.extensions["mediashelf"] + out = {"status": "ok", "db": "ok", "plex": "unknown", "tautulli": "disabled"} + try: + db.scalar("SELECT 1") + except Exception as e: # noqa: BLE001 + out["status"], out["db"] = "error", str(e) + row = db.one("SELECT * FROM scan WHERE status='succeeded' ORDER BY id DESC LIMIT 1") + out["last_scan_at"] = row["finished_at"] if row else None + out["stale"] = bool(row is None or (time.time() - (row["finished_at"] or 0)) > 48 * 3600) + cov = db.one("SELECT source FROM history_coverage ORDER BY event_count DESC LIMIT 1") + out["history_source"] = cov["source"] if cov else None + out["plex"] = "configured" if cfg.plex_configured else "unconfigured" + out["tautulli"] = "configured" if cfg.tautulli_configured else "disabled" + out["degraded"] = ext.get("degraded_reason") + return jsonify(out), (200 if out["status"] == "ok" else 503) + + @app.errorhandler(404) + def _404(e): + return jsonify({"error": "not_found"}), 404 + + @app.errorhandler(500) + def _500(e): + log.exception("unhandled error") + return jsonify({"error": "internal_error"}), 500 + + if start_scheduler: + scanner.start_scheduler(app) + + return app diff --git a/mediashelf/cli.py b/mediashelf/cli.py new file mode 100644 index 0000000..771142d --- /dev/null +++ b/mediashelf/cli.py @@ -0,0 +1,112 @@ +"""Command line: run a scan, serve in development, or export keeps. + + python -m mediashelf.cli scan --full + python -m mediashelf.cli serve --port 8080 + python -m mediashelf.cli export-keeps > keeps.json +""" + +from __future__ import annotations + +import argparse +import json +import logging +import sys + +from . import ingest, keeps, providers +from .app import create_app +from .config import Config +from .db import Database + + +def _load_dotenv(path=".env"): + """Minimal .env support for development. Production uses the stack env.""" + import os + try: + with open(path) as fh: + for line in fh: + line = line.strip() + if not line or line.startswith("#") or "=" not in line: + continue + k, v = line.split("=", 1) + os.environ.setdefault(k.strip(), v.strip()) + except OSError: + pass + + +def cmd_scan(args) -> int: + cfg = Config.from_env() + db = Database(cfg.database_path) + db.migrate() + media = providers.build_media_provider(cfg) + history, degraded = providers.build_history_provider(cfg, media) + if degraded: + print("WARNING: history degraded — %s" % degraded, file=sys.stderr) + print("history source: %s" % (history.name if history else "none"), file=sys.stderr) + + result = ingest.Ingest(db, cfg, media, history).run( + "full" if args.full else "incremental", "manual") + print(json.dumps({ + "status": result.status, "scan_id": result.scan_id, + "items_seen": result.items_seen, "items_added": result.items_added, + "items_updated": result.items_updated, "items_missing": result.items_missing, + "events_added": result.events_added, "warnings": len(result.warnings or []), + "error": result.error, + }, indent=2)) + return 0 if result.status == "succeeded" else 1 + + +def cmd_serve(args) -> int: + app = create_app(start_scheduler=not args.no_scheduler) + app.run(host=args.host, port=args.port, debug=args.debug, threaded=True) + return 0 + + +def cmd_export_keeps(args) -> int: + cfg = Config.from_env() + db = Database(cfg.database_path) + db.migrate() + json.dump(keeps.export(db), sys.stdout, indent=2) + return 0 + + +def cmd_import_keeps(args) -> int: + cfg = Config.from_env() + db = Database(cfg.database_path) + db.migrate() + with open(args.path) as fh: + payload = json.load(fh) + print(json.dumps(keeps.import_(db, payload), indent=2)) + return 0 + + +def main(argv=None) -> int: + _load_dotenv() + logging.basicConfig(level=logging.INFO, + format="%(asctime)s %(levelname)-7s %(name)s: %(message)s") + p = argparse.ArgumentParser(prog="mediashelf") + sub = p.add_subparsers(dest="cmd", required=True) + + s = sub.add_parser("scan", help="run a scan now") + s.add_argument("--full", action="store_true", help="full sweep rather than incremental") + s.set_defaults(func=cmd_scan) + + s = sub.add_parser("serve", help="development server") + s.add_argument("--host", default="127.0.0.1") + s.add_argument("--port", type=int, default=8080) + s.add_argument("--debug", action="store_true") + s.add_argument("--no-scheduler", action="store_true") + s.set_defaults(func=cmd_serve) + + s = sub.add_parser("export-keeps", help="write keep marks to stdout as JSON") + s.set_defaults(func=cmd_export_keeps) + + s = sub.add_parser("import-keeps", help="restore keep marks from a JSON export") + s.add_argument("path") + s.set_defaults(func=cmd_import_keeps) + + args = p.parse_args(argv) + return args.func(args) + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/mediashelf/config.py b/mediashelf/config.py new file mode 100644 index 0000000..c69d41e --- /dev/null +++ b/mediashelf/config.py @@ -0,0 +1,173 @@ +"""Configuration, read once from the environment. + +Every value here is documented in docs/design.md §10. Defaults match that table. +The Portainer stack is the single source of truth in production; .env is used in +development via `python -m mediashelf.cli`. +""" + +from __future__ import annotations + +import os +from dataclasses import dataclass, field + + +def _b(name: str, default: bool) -> bool: + v = os.environ.get(name) + if v is None or v == "": + return default + return v.strip().lower() in ("1", "true", "yes", "on") + + +def _i(name: str, default: int) -> int: + v = os.environ.get(name) + if v is None or v == "": + return default + try: + return int(v) + except ValueError: + return default + + +def _f(name: str, default: float) -> float: + v = os.environ.get(name) + if v is None or v == "": + return default + try: + return float(v) + except ValueError: + return default + + +def _s(name: str, default: str = "") -> str: + v = os.environ.get(name) + return default if v is None else v.strip() + + +def _csv(name: str) -> list[str]: + raw = _s(name) + return [p.strip() for p in raw.split(",") if p.strip()] + + +@dataclass(frozen=True) +class ScoreConfig: + """Constants for the reclaim score (§6).""" + + stale_horizon_days: int = 730 + age_horizon_days: int = 1095 + popular_at: int = 3 + rejected_at: int = 2 + solitude_at: int = 3 + grace_days: int = 30 + recent_days: int = 90 + + # Default weight profile, calibrated against the measured library (§6.1/§6.2). + weights: dict[str, float] = field( + default_factory=lambda: { + "size": 0.28, + "staleness": 0.24, + "unpopularity": 0.22, + "solitude": 0.10, + "age": 0.10, + "rejection": 0.06, + } + ) + + @classmethod + def from_env(cls) -> "ScoreConfig": + return cls( + stale_horizon_days=_i("SCORE_STALE_HORIZON_DAYS", 730), + age_horizon_days=_i("SCORE_AGE_HORIZON_DAYS", 1095), + popular_at=_i("SCORE_POPULAR_AT", 3), + rejected_at=_i("SCORE_REJECTED_AT", 2), + solitude_at=_i("SCORE_SOLITUDE_AT", 3), + grace_days=_i("SCORE_GRACE_DAYS", 30), + recent_days=_i("SCORE_RECENT_DAYS", 90), + ) + + +@dataclass(frozen=True) +class Config: + secret_key: str = "dev-insecure" + log_level: str = "INFO" + tz: str = "America/Regina" + + plex_base_url: str = "" + plex_token: str = "" + plex_verify_ssl: bool = True + plex_timeout_s: int = 30 + plex_page_size: int = 500 + plex_request_delay_ms: int = 0 + + tautulli_base_url: str = "" + tautulli_api_key: str = "" + tautulli_timeout_s: int = 30 + tautulli_page_size: int = 1000 + + history_source: str = "auto" # auto | tautulli | plex + session_merge_window_h: int = 6 + completion_threshold: int = 85 + abandon_ceiling: int = 15 + + database_path: str = "/data/mediashelf.db" + + scan_schedule_cron: str = "0 4 * * *" + scan_full_sweep_cron: str = "0 3 * * 0" + scan_on_startup: bool = False + scan_lock_timeout_s: int = 7200 + scheduler_enabled: bool = True + + keep_all_libraries: list[str] = field(default_factory=list) + + score: ScoreConfig = field(default_factory=ScoreConfig) + + @classmethod + def from_env(cls) -> "Config": + return cls( + secret_key=_s("MEDIASHELF_SECRET_KEY", "dev-insecure"), + log_level=_s("LOG_LEVEL", "INFO").upper(), + tz=_s("TZ", "America/Regina"), + plex_base_url=_s("PLEX_BASE_URL").rstrip("/"), + plex_token=_s("PLEX_TOKEN"), + plex_verify_ssl=_b("PLEX_VERIFY_SSL", True), + plex_timeout_s=_i("PLEX_TIMEOUT_S", 30), + plex_page_size=_i("PLEX_PAGE_SIZE", 500), + plex_request_delay_ms=_i("PLEX_REQUEST_DELAY_MS", 0), + tautulli_base_url=_s("TAUTULLI_BASE_URL").rstrip("/"), + tautulli_api_key=_s("TAUTULLI_API_KEY"), + tautulli_timeout_s=_i("TAUTULLI_TIMEOUT_S", 30), + tautulli_page_size=_i("TAUTULLI_PAGE_SIZE", 1000), + history_source=_s("HISTORY_SOURCE", "auto").lower(), + session_merge_window_h=_i("SESSION_MERGE_WINDOW_H", 6), + completion_threshold=_i("COMPLETION_THRESHOLD", 85), + abandon_ceiling=_i("ABANDON_CEILING", 15), + database_path=_s("DATABASE_PATH", "/data/mediashelf.db"), + scan_schedule_cron=_s("SCAN_SCHEDULE_CRON", "0 4 * * *"), + scan_full_sweep_cron=_s("SCAN_FULL_SWEEP_CRON", "0 3 * * 0"), + scan_on_startup=_b("SCAN_ON_STARTUP", False), + scan_lock_timeout_s=_i("SCAN_LOCK_TIMEOUT_S", 7200), + scheduler_enabled=_b("SCHEDULER_ENABLED", True), + keep_all_libraries=_csv("KEEP_ALL_LIBRARIES"), + score=ScoreConfig.from_env(), + ) + + # ── derived ────────────────────────────────────────────────────────── + + @property + def tautulli_configured(self) -> bool: + return bool(self.tautulli_base_url and self.tautulli_api_key) + + @property + def plex_configured(self) -> bool: + return bool(self.plex_base_url and self.plex_token) + + def redacted(self) -> dict: + """Safe for API responses and logs. Never leaks a credential (§12).""" + out = {} + for k, v in self.__dict__.items(): + if k in ("plex_token", "tautulli_api_key", "secret_key"): + out[k] = "***" if v else "" + elif k == "score": + out[k] = v.__dict__ + else: + out[k] = v + return out diff --git a/mediashelf/db.py b/mediashelf/db.py new file mode 100644 index 0000000..3b96546 --- /dev/null +++ b/mediashelf/db.py @@ -0,0 +1,158 @@ +"""SQLite access and migrations. + +One writer (the scan job), many readers. WAL mode makes that work without +readers blocking. Migrations are plain numbered .sql files applied in order at +startup and tracked in schema_version — no ORM migration framework (§5). +""" + +from __future__ import annotations + +import logging +import os +import sqlite3 +import threading +from pathlib import Path + +log = logging.getLogger(__name__) + +MIGRATIONS_DIR = Path(__file__).parent / "migrations" + +_local = threading.local() + + +def connect(path: str, *, read_only: bool = False) -> sqlite3.Connection: + """Open a tuned connection. Callers own the connection lifecycle.""" + if path != ":memory:": + os.makedirs(os.path.dirname(os.path.abspath(path)) or ".", exist_ok=True) + conn = sqlite3.connect(path, timeout=30.0, isolation_level=None) + conn.row_factory = sqlite3.Row + conn.execute("PRAGMA journal_mode=WAL") + conn.execute("PRAGMA foreign_keys=ON") + conn.execute("PRAGMA synchronous=NORMAL") + conn.execute("PRAGMA busy_timeout=30000") + conn.execute("PRAGMA temp_store=MEMORY") + if read_only: + conn.execute("PRAGMA query_only=ON") + return conn + + +def migrate(conn: sqlite3.Connection) -> int: + """Apply any unapplied migrations. Returns the resulting version.""" + conn.execute( + "CREATE TABLE IF NOT EXISTS schema_version (" + " version INTEGER PRIMARY KEY, applied_at INTEGER NOT NULL)" + ) + row = conn.execute("SELECT COALESCE(MAX(version), 0) AS v FROM schema_version").fetchone() + current = row["v"] + + files = sorted(MIGRATIONS_DIR.glob("*.sql")) + applied = current + for f in files: + try: + version = int(f.name.split("_", 1)[0]) + except ValueError: + log.warning("skipping unnumbered migration %s", f.name) + continue + if version <= current: + continue + log.info("applying migration %s", f.name) + sql = f.read_text() + # Migrations contain triggers with internal semicolons, so the whole file + # runs as a script rather than being split on ';'. executescript() commits + # any pending transaction before it runs, so the BEGIN/COMMIT must live + # inside the script itself for the migration to be atomic. + script = ( + "BEGIN;\n" + + sql + + f"\nINSERT INTO schema_version (version, applied_at)" + f" VALUES ({version}, strftime('%s','now'));\nCOMMIT;" + ) + try: + conn.executescript(script) + except Exception: + if conn.in_transaction: + conn.execute("ROLLBACK") + raise + applied = version + return applied + + +class Database: + """Thin wrapper giving each thread its own connection. + + Flask serves requests on multiple threads and the scheduler runs on its own, + and sqlite3 connections are not shareable across threads. + """ + + def __init__(self, path: str): + self.path = path + + @property + def conn(self) -> sqlite3.Connection: + c = getattr(_local, "conn", None) + if c is None or getattr(_local, "path", None) != self.path: + c = connect(self.path) + _local.conn = c + _local.path = self.path + return c + + def close(self) -> None: + c = getattr(_local, "conn", None) + if c is not None: + c.close() + _local.conn = None + + # ── convenience ────────────────────────────────────────────────────── + + def query(self, sql: str, params: tuple | dict = ()) -> list[sqlite3.Row]: + return self.conn.execute(sql, params).fetchall() + + def one(self, sql: str, params: tuple | dict = ()) -> sqlite3.Row | None: + return self.conn.execute(sql, params).fetchone() + + def scalar(self, sql: str, params: tuple | dict = ()): + row = self.conn.execute(sql, params).fetchone() + return None if row is None else row[0] + + def execute(self, sql: str, params: tuple | dict = ()) -> sqlite3.Cursor: + return self.conn.execute(sql, params) + + def executemany(self, sql: str, seq) -> sqlite3.Cursor: + return self.conn.executemany(sql, seq) + + def migrate(self) -> int: + return migrate(self.conn) + + # ── settings ───────────────────────────────────────────────────────── + + def get_setting(self, key: str, default=None): + row = self.one("SELECT value FROM setting WHERE key = ?", (key,)) + return default if row is None else row["value"] + + def set_setting(self, key: str, value: str) -> None: + self.execute( + "INSERT INTO setting (key, value) VALUES (?, ?) " + "ON CONFLICT(key) DO UPDATE SET value = excluded.value", + (key, str(value)), + ) + + +class transaction: + """Context manager for an explicit write transaction. + + isolation_level is None (autocommit), so transactions are explicit here. + """ + + def __init__(self, db: Database): + self.db = db + + def __enter__(self) -> sqlite3.Connection: + self.db.conn.execute("BEGIN IMMEDIATE") + return self.db.conn + + def __exit__(self, exc_type, exc, tb) -> bool: + if exc_type is None: + self.db.conn.execute("COMMIT") + else: + self.db.conn.execute("ROLLBACK") + return False diff --git a/mediashelf/ingest.py b/mediashelf/ingest.py new file mode 100644 index 0000000..a48d85a --- /dev/null +++ b/mediashelf/ingest.py @@ -0,0 +1,696 @@ +"""Scan orchestration (§4.6). + +Pulls libraries, items and history; normalizes; upserts inside one transaction +per library; rolls episodes up to seasons; resolves keep marks; marks vanished +items missing. + +The property that matters most here is idempotency. A scanner that double-counts +sizes or duplicates history events produces a report that looks entirely +plausible and is wrong, which is worse than one that crashes. +""" + +from __future__ import annotations + +import json +import logging +import os +import time +from dataclasses import dataclass + +from . import keeps +from .db import Database +from .providers.base import ( + Account, + AuthError, + Coverage, + HistoryProvider, + Item, + Library, + MediaProvider, + ProviderError, + WatchEvent, +) + +log = logging.getLogger(__name__) + + +class ScanBusy(RuntimeError): + pass + + +@dataclass +class ScanResult: + scan_id: int + status: str + items_seen: int = 0 + items_added: int = 0 + items_updated: int = 0 + items_missing: int = 0 + events_added: int = 0 + warnings: list[str] = None + error: str | None = None + + +class Ingest: + def __init__(self, db: Database, cfg, media: MediaProvider, + history: HistoryProvider | None): + self.db = db + self.cfg = cfg + self.media = media + self.history = history + self.warnings: list[str] = [] + + # ── locking ────────────────────────────────────────────────────────── + + def _acquire_lock(self, scan_id: int) -> None: + now = int(time.time()) + row = self.db.one("SELECT * FROM scan_lock WHERE id = 1") + if row and row["scan_id"] is not None: + age = now - (row["acquired_at"] or 0) + if age < self.cfg.scan_lock_timeout_s: + raise ScanBusy("a scan is already running (started %ds ago)" % age) + # Stale lock: the previous scan died. Mark it failed and take over. + log.warning("breaking stale scan lock held by scan %s", row["scan_id"]) + self.db.execute( + "UPDATE scan SET status='failed', finished_at=?, " + "error='abandoned - lock timed out' WHERE id=? AND status='running'", + (now, row["scan_id"]), + ) + self.db.execute( + "INSERT INTO scan_lock (id, scan_id, holder, acquired_at) VALUES (1,?,?,?) " + "ON CONFLICT(id) DO UPDATE SET scan_id=excluded.scan_id, " + "holder=excluded.holder, acquired_at=excluded.acquired_at", + (scan_id, "%s:%s" % (os.uname().nodename, os.getpid()), now), + ) + + def _release_lock(self) -> None: + self.db.execute("UPDATE scan_lock SET scan_id=NULL, holder=NULL WHERE id=1") + + def _progress(self, scan_id: int, text: str) -> None: + self.db.execute("UPDATE scan SET progress=? WHERE id=?", (text, scan_id)) + + def _warn(self, msg: str) -> None: + log.warning("scan warning: %s", msg) + if len(self.warnings) < 200: + self.warnings.append(msg) + + # ── entry point ────────────────────────────────────────────────────── + + def run(self, mode: str = "full", trigger: str = "manual") -> ScanResult: + now = int(time.time()) + cur = self.db.execute( + "INSERT INTO scan (mode, trigger, status, started_at, history_source) " + "VALUES (?,?,'running',?,?)", + (mode, trigger, now, self.history.name if self.history else None), + ) + scan_id = cur.lastrowid + try: + self._acquire_lock(scan_id) + except ScanBusy: + self.db.execute( + "UPDATE scan SET status='failed', finished_at=?, error=? WHERE id=?", + (now, "another scan is already running", scan_id), + ) + raise + + result = ScanResult(scan_id=scan_id, status="running", warnings=[]) + try: + self._run_inner(scan_id, mode, result) + result.status = "succeeded" + except AuthError as e: + result.status, result.error = "failed", str(e) + log.error("scan %s failed on auth: %s", scan_id, e) + except Exception as e: # noqa: BLE001 + result.status, result.error = "failed", str(e) + log.exception("scan %s failed", scan_id) + finally: + result.warnings = self.warnings + self.db.execute( + "UPDATE scan SET status=?, finished_at=?, items_seen=?, items_added=?, " + "items_updated=?, items_missing=?, events_added=?, warning_count=?, " + "warnings=?, error=?, progress=NULL WHERE id=?", + (result.status, int(time.time()), result.items_seen, result.items_added, + result.items_updated, result.items_missing, result.events_added, + len(self.warnings), json.dumps(self.warnings[:200]), result.error, scan_id), + ) + self._release_lock() + return result + + def _run_inner(self, scan_id: int, mode: str, result: ScanResult) -> None: + self._progress(scan_id, "connecting") + info = self.media.server_info() + provider_id = self._upsert_provider(info) + self.db.execute("UPDATE scan SET provider_id=? WHERE id=?", (provider_id, scan_id)) + + self._check_history_pairing(provider_id, info) + + self._progress(scan_id, "reading libraries") + libraries = self.media.libraries() + lib_ids = self._upsert_libraries(provider_id, libraries) + self._seed_keep_all_libraries() + + # History first: item rollups need it in place. + self._progress(scan_id, "reading watch history") + coverage = self._ingest_history(provider_id, mode, result) + + for lib in libraries: + self._progress(scan_id, "scanning %s" % lib.title) + self._ingest_library(provider_id, lib, lib_ids[lib.provider_key], + scan_id, result) + + self._progress(scan_id, "rolling up") + self._apply_watch_rollups(provider_id) + self._rollup_seasons(provider_id) + self._rollup_shows(provider_id) + self._apply_pre_history(provider_id, coverage) + + if mode == "full": + self._mark_missing(provider_id, scan_id, result) + + self._progress(scan_id, "resolving keeps") + keeps.resolve_all(self.db) + orphans = keeps.stamp_matches(self.db, scan_id) + if orphans: + self._warn("%d keep mark(s) matched nothing this scan" % orphans) + + self.db.execute("UPDATE provider SET last_scan_id=? WHERE id=?", (scan_id, provider_id)) + + # ── provider / libraries ───────────────────────────────────────────── + + def _upsert_provider(self, info) -> int: + now = int(time.time()) + self.db.execute( + "INSERT INTO provider (kind, name, base_url, server_id, version, created_at) " + "VALUES (?,?,?,?,?,?) ON CONFLICT(kind, base_url) DO UPDATE SET " + "name=excluded.name, server_id=excluded.server_id, version=excluded.version", + (info.kind, info.name, info.base_url, info.server_id, info.version, now), + ) + return self.db.scalar( + "SELECT id FROM provider WHERE kind=? AND base_url=?", + (info.kind, info.base_url), + ) + + def _check_history_pairing(self, provider_id: int, media_info) -> None: + """Refuse to join history from a different Plex server (§4.11).""" + if self.history is None or self.history.name != "tautulli": + return + try: + hinfo = self.history.server_info() + except ProviderError as e: + self._warn("could not read Tautulli server info: %s" % e) + return + if hinfo.server_id and media_info.server_id and hinfo.server_id != media_info.server_id: + raise ProviderError( + "Tautulli is watching a different Plex server " + "(%s != %s) - refusing to join unrelated history data" + % (hinfo.server_id[:8], media_info.server_id[:8]) + ) + + def _upsert_libraries(self, provider_id: int, libraries: list[Library]) -> dict[str, int]: + out = {} + now = int(time.time()) + for lib in libraries: + self.db.execute( + "INSERT INTO library (provider_id, provider_key, title, kind, locations, scanned_at) " + "VALUES (?,?,?,?,?,?) ON CONFLICT(provider_id, provider_key) DO UPDATE SET " + "title=excluded.title, kind=excluded.kind, locations=excluded.locations, " + "scanned_at=excluded.scanned_at", + (provider_id, lib.provider_key, lib.title, lib.kind, + json.dumps(lib.locations), now), + ) + out[lib.provider_key] = self.db.scalar( + "SELECT id FROM library WHERE provider_id=? AND provider_key=?", + (provider_id, lib.provider_key), + ) + return out + + def _seed_keep_all_libraries(self) -> None: + """Apply KEEP_ALL_LIBRARIES once, on first run only. + + Empty by default: nothing is ever kept unless a person says so (§6.6). + Re-applying on every start would silently re-enable a rule the user + turned off in the UI, so a marker setting guards it. + """ + if not self.cfg.keep_all_libraries: + return + if self.db.get_setting("keep_all_seeded"): + return + for title in self.cfg.keep_all_libraries: + row = self.db.one("SELECT id FROM library WHERE title = ?", (title,)) + if row: + self.db.execute("UPDATE library SET keep_all = 1 WHERE id = ?", (row["id"],)) + log.info("seeded keep_all for library %r", title) + else: + self._warn("KEEP_ALL_LIBRARIES names %r, which is not a library" % title) + self.db.set_setting("keep_all_seeded", "1") + + # ── history ────────────────────────────────────────────────────────── + + def _disposition(self, pc: int | None) -> str: + if pc is None: + return "completed" # Plex fallback: only a play is recorded (§4.11) + if pc >= self.cfg.completion_threshold: + return "completed" + if pc < self.cfg.abandon_ceiling: + return "abandoned" + return "partial" + + def _ingest_history(self, provider_id: int, mode: str, result: ScanResult) -> Coverage | None: + if self.history is None: + return None + + try: + for acct in self.history.accounts(): + self.db.execute( + "INSERT INTO account (provider_id, account_id, name, friendly_name) " + "VALUES (?,?,?,?) ON CONFLICT(provider_id, account_id) DO UPDATE SET " + "name=excluded.name, friendly_name=excluded.friendly_name", + (provider_id, acct.account_id, acct.name, acct.friendly_name), + ) + except ProviderError as e: + self._warn("could not read accounts: %s" % e) + + since = None + if mode != "full": + since = self.db.scalar( + "SELECT MAX(viewed_at) FROM watch_event WHERE provider_id=? AND source=?", + (provider_id, self.history.name), + ) + + window = self.cfg.session_merge_window_h * 3600 + batch: list[tuple] = [] + added = 0 + + def flush(): + nonlocal added, batch + if not batch: + return + cur = self.db.executemany( + "INSERT INTO watch_event (provider_id, source, source_row_id, reference_id, " + "provider_item_id, account_id, viewed_at, stopped_at, play_duration_s, " + "paused_counter_s, percent_complete, watched_status, disposition, session_id, " + "media_type, platform) VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?) " + "ON CONFLICT(provider_id, source, source_row_id) DO NOTHING", + batch, + ) + added += cur.rowcount if cur.rowcount and cur.rowcount > 0 else 0 + batch = [] + + for ev in self.history.watch_events(since): + # Session key: same item + same user within the merge window (§4.9). + bucket = ev.viewed_at // window if window > 0 else ev.viewed_at + session_id = "%s:%s:%s" % (ev.provider_item_id, ev.account_id or "-", bucket) + batch.append(( + provider_id, ev.source, ev.source_row_id, ev.reference_id, + ev.provider_item_id, ev.account_id, ev.viewed_at, ev.stopped_at, + ev.play_duration_s, ev.paused_counter_s, ev.percent_complete, + ev.watched_status, self._disposition(ev.percent_complete), session_id, + ev.media_type, ev.platform, + )) + if len(batch) >= 2000: + flush() + flush() + result.events_added = added + + cov = self.db.one( + "SELECT MIN(viewed_at) AS lo, MAX(viewed_at) AS hi, COUNT(*) AS n " + "FROM watch_event WHERE provider_id=? AND source=?", + (provider_id, self.history.name), + ) + coverage = Coverage(cov["lo"], cov["hi"], cov["n"] or 0) + self.db.execute( + "INSERT INTO history_coverage (provider_id, source, earliest_event_at, " + "latest_event_at, event_count, updated_at) VALUES (?,?,?,?,?,?) " + "ON CONFLICT(provider_id, source) DO UPDATE SET " + "earliest_event_at=excluded.earliest_event_at, " + "latest_event_at=excluded.latest_event_at, " + "event_count=excluded.event_count, updated_at=excluded.updated_at", + (provider_id, self.history.name, coverage.earliest_event_at, + coverage.latest_event_at, coverage.event_count, int(time.time())), + ) + return coverage + + # ── items ──────────────────────────────────────────────────────────── + + def _ingest_library(self, provider_id: int, lib: Library, library_id: int, + scan_id: int, result: ScanResult) -> None: + show_guids: dict[str, str] = {} + if lib.kind == "show": + try: + show_guids = self.media.show_guids(lib) + except Exception as e: # noqa: BLE001 + self._warn("could not read show GUIDs for %s: %s" % (lib.title, e)) + + seasons: dict[str, dict] = {} + shows: dict[str, dict] = {} + n = 0 + + conn = self.db.conn + conn.execute("BEGIN IMMEDIATE") + try: + for item in self.media.items(lib): + n += 1 + if item.kind == "movie": + self._upsert_movie(provider_id, library_id, item, scan_id, result) + else: + self._collect_episode(provider_id, library_id, item, show_guids, + seasons, shows, scan_id, result) + # season/show container rows + for key, s in seasons.items(): + self._upsert_season(provider_id, library_id, s, scan_id, result) + for key, s in shows.items(): + self._upsert_show(provider_id, library_id, s, scan_id, result) + conn.execute("COMMIT") + except Exception: + conn.execute("ROLLBACK") + raise + + result.items_seen += n + + def _upsert_movie(self, provider_id, library_id, item: Item, scan_id, result) -> None: + existing = self.db.one( + "SELECT id FROM media_item WHERE provider_id=? AND provider_item_id=?", + (provider_id, item.provider_item_id), + ) + primary = item.parts[0].file_path if item.parts else None + vals = ( + provider_id, library_id, item.provider_item_id, "movie", item.guid, None, + item.title, item.sort_title, item.year, None, None, + item.added_at, item.updated_at, 0, item.size_bytes, item.duration_ms, + len(item.parts), primary, item.resolution, item.video_codec, + item.view_count, "present", scan_id, scan_id, + ) + if existing: + self.db.execute( + "UPDATE media_item SET library_id=?, guid=?, title=?, sort_title=?, year=?, " + "added_at=?, updated_at=?, size_bytes=?, duration_ms=?, part_count=?, " + "primary_path=?, resolution=?, video_codec=?, provider_view_count=?, " + "status='present', last_seen_scan_id=? WHERE id=?", + (library_id, item.guid, item.title, item.sort_title, item.year, + item.added_at, item.updated_at, item.size_bytes, item.duration_ms, + len(item.parts), primary, item.resolution, item.video_codec, + item.view_count, scan_id, existing["id"]), + ) + item_id = existing["id"] + result.items_updated += 1 + else: + cur = self.db.execute( + "INSERT INTO media_item (provider_id, library_id, provider_item_id, kind, " + "guid, show_guid, title, sort_title, year, parent_id, season_number, " + "added_at, updated_at, episode_count, size_bytes, duration_ms, part_count, " + "primary_path, resolution, video_codec, provider_view_count, status, " + "first_seen_scan_id, last_seen_scan_id) " + "VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)", vals, + ) + item_id = cur.lastrowid + result.items_added += 1 + + # Parts are replaced wholesale — cheap, and the only way to stay correct + # when a version is removed or a split file is re-encoded into one. + self.db.execute("DELETE FROM media_part WHERE media_item_id=?", (item_id,)) + self._insert_parts(item, media_item_id=item_id) + + def _insert_parts(self, item: Item, *, media_item_id=None, episode_id=None) -> None: + if not item.parts: + return + self.db.executemany( + "INSERT INTO media_part (media_item_id, episode_id, provider_part_id, file_path, " + "size_bytes, container, resolution, video_codec, audio_codec, bitrate) " + "VALUES (?,?,?,?,?,?,?,?,?,?)", + [(media_item_id, episode_id, p.provider_part_id, p.file_path, p.size_bytes, + p.container, p.resolution, p.video_codec, p.audio_codec, p.bitrate) + for p in item.parts], + ) + + def _collect_episode(self, provider_id, library_id, item: Item, show_guids, + seasons, shows, scan_id, result) -> None: + if not item.season_id: + self._warn("episode %s has no season; skipped" % item.provider_item_id) + return + show_guid = show_guids.get(item.show_id or "") or None + + s = seasons.setdefault(item.season_id, { + "provider_item_id": item.season_id, + "show_id": item.show_id, + "show_guid": show_guid, + "show_title": item.show_title, + "season_number": item.season_number, + "episodes": [], + }) + s["episodes"].append(item) + + sh = shows.setdefault(item.show_id or "?", { + "provider_item_id": item.show_id, + "guid": show_guid, + "title": item.show_title or "(unknown show)", + "seasons": set(), + }) + sh["seasons"].add(item.season_id) + + def _upsert_season(self, provider_id, library_id, s: dict, scan_id, result) -> None: + eps: list[Item] = s["episodes"] + title = "Season %s" % (s["season_number"] if s["season_number"] is not None else "?") + existing = self.db.one( + "SELECT id FROM media_item WHERE provider_id=? AND provider_item_id=?", + (provider_id, s["provider_item_id"]), + ) + if existing: + season_id = existing["id"] + self.db.execute( + "UPDATE media_item SET library_id=?, show_guid=?, title=?, season_number=?, " + "status='present', last_seen_scan_id=? WHERE id=?", + (library_id, s["show_guid"], title, s["season_number"], scan_id, season_id), + ) + result.items_updated += 1 + else: + cur = self.db.execute( + "INSERT INTO media_item (provider_id, library_id, provider_item_id, kind, " + "show_guid, title, season_number, status, first_seen_scan_id, last_seen_scan_id) " + "VALUES (?,?,?,'season',?,?,?,'present',?,?)", + (provider_id, library_id, s["provider_item_id"], s["show_guid"], + title, s["season_number"], scan_id, scan_id), + ) + season_id = cur.lastrowid + result.items_added += 1 + + for ep in eps: + row = self.db.one("SELECT id FROM episode WHERE provider_item_id=?", + (ep.provider_item_id,)) + primary_count = len(ep.parts) + if row: + ep_id = row["id"] + self.db.execute( + "UPDATE episode SET season_item_id=?, episode_number=?, title=?, " + "added_at=?, duration_ms=?, size_bytes=?, part_count=?, " + "status='present', last_seen_scan_id=? WHERE id=?", + (season_id, ep.episode_number, ep.title, ep.added_at, ep.duration_ms, + ep.size_bytes, primary_count, scan_id, ep_id), + ) + else: + cur = self.db.execute( + "INSERT INTO episode (season_item_id, provider_item_id, episode_number, " + "title, added_at, duration_ms, size_bytes, part_count, status, last_seen_scan_id) " + "VALUES (?,?,?,?,?,?,?,?, 'present', ?)", + (season_id, ep.provider_item_id, ep.episode_number, ep.title, + ep.added_at, ep.duration_ms, ep.size_bytes, primary_count, scan_id), + ) + ep_id = cur.lastrowid + self.db.execute("DELETE FROM media_part WHERE episode_id=?", (ep_id,)) + self._insert_parts(ep, episode_id=ep_id) + + def _upsert_show(self, provider_id, library_id, sh: dict, scan_id, result) -> None: + if not sh["provider_item_id"]: + return + existing = self.db.one( + "SELECT id FROM media_item WHERE provider_id=? AND provider_item_id=?", + (provider_id, sh["provider_item_id"]), + ) + if existing: + show_id = existing["id"] + self.db.execute( + "UPDATE media_item SET library_id=?, guid=?, title=?, status='present', " + "last_seen_scan_id=? WHERE id=?", + (library_id, sh["guid"], sh["title"], scan_id, show_id), + ) + else: + cur = self.db.execute( + "INSERT INTO media_item (provider_id, library_id, provider_item_id, kind, " + "guid, title, status, first_seen_scan_id, last_seen_scan_id) " + "VALUES (?,?,?,'show',?,?,'present',?,?)", + (provider_id, library_id, sh["provider_item_id"], sh["guid"], + sh["title"], scan_id, scan_id), + ) + show_id = cur.lastrowid + # link seasons to their show + self.db.execute( + "UPDATE media_item SET parent_id=? WHERE provider_id=? AND kind='season' " + "AND provider_item_id IN (%s)" % ",".join("?" * len(sh["seasons"])), + tuple([show_id, provider_id] + list(sh["seasons"])), + ) + + # ── rollups ────────────────────────────────────────────────────────── + + def _apply_watch_rollups(self, provider_id: int) -> None: + """Aggregate watch_event onto movies and episodes. + + Distinct sessions, not raw events: a paused-and-resumed play is one + viewing (§4.9). Everything is recomputed from scratch each scan, which is + what makes re-running a scan idempotent. + """ + c = self.db.conn + c.execute(""" + UPDATE media_item SET watch_count=0, partial_count=0, abandoned_count=0, + last_watched_at=NULL, last_touched_at=NULL, first_watched_at=NULL, + distinct_watcher_count=0, avg_percent_complete=NULL + WHERE kind='movie' + """) + c.execute(""" + UPDATE episode SET watch_count=0, partial_count=0, abandoned_count=0, + last_watched_at=NULL, last_touched_at=NULL + """) + + agg = """ + SELECT provider_item_id AS pid, + COUNT(DISTINCT CASE WHEN disposition='completed' THEN session_id END) AS completed, + COUNT(DISTINCT CASE WHEN disposition='partial' THEN session_id END) AS partial, + COUNT(DISTINCT CASE WHEN disposition='abandoned' THEN session_id END) AS abandoned, + MAX(CASE WHEN disposition='completed' THEN viewed_at END) AS last_watched, + MIN(CASE WHEN disposition='completed' THEN viewed_at END) AS first_watched, + MAX(viewed_at) AS last_touched, + COUNT(DISTINCT account_id) AS watchers, + AVG(percent_complete) AS avg_pc + FROM watch_event WHERE provider_id = ? + GROUP BY provider_item_id + """ + c.execute("DROP TABLE IF EXISTS _wagg") + c.execute("CREATE TEMP TABLE _wagg AS " + agg, (provider_id,)) + c.execute("CREATE INDEX _wagg_pid ON _wagg(pid)") + + c.execute(""" + UPDATE media_item SET + watch_count = COALESCE((SELECT completed FROM _wagg WHERE pid = media_item.provider_item_id), 0), + partial_count = COALESCE((SELECT partial FROM _wagg WHERE pid = media_item.provider_item_id), 0), + abandoned_count = COALESCE((SELECT abandoned FROM _wagg WHERE pid = media_item.provider_item_id), 0), + last_watched_at = (SELECT last_watched FROM _wagg WHERE pid = media_item.provider_item_id), + first_watched_at = (SELECT first_watched FROM _wagg WHERE pid = media_item.provider_item_id), + last_touched_at = (SELECT last_touched FROM _wagg WHERE pid = media_item.provider_item_id), + distinct_watcher_count = COALESCE((SELECT watchers FROM _wagg WHERE pid = media_item.provider_item_id), 0), + avg_percent_complete = (SELECT avg_pc FROM _wagg WHERE pid = media_item.provider_item_id) + WHERE kind = 'movie' + """) + c.execute(""" + UPDATE episode SET + watch_count = COALESCE((SELECT completed FROM _wagg WHERE pid = episode.provider_item_id), 0), + partial_count = COALESCE((SELECT partial FROM _wagg WHERE pid = episode.provider_item_id), 0), + abandoned_count = COALESCE((SELECT abandoned FROM _wagg WHERE pid = episode.provider_item_id), 0), + last_watched_at = (SELECT last_watched FROM _wagg WHERE pid = episode.provider_item_id), + last_touched_at = (SELECT last_touched FROM _wagg WHERE pid = episode.provider_item_id) + """) + + def _rollup_seasons(self, provider_id: int) -> None: + c = self.db.conn + c.execute(""" + UPDATE media_item SET + episode_count = COALESCE((SELECT COUNT(*) FROM episode e + WHERE e.season_item_id = media_item.id AND e.status='present'), 0), + size_bytes = COALESCE((SELECT SUM(e.size_bytes) FROM episode e + WHERE e.season_item_id = media_item.id AND e.status='present'), 0), + duration_ms = COALESCE((SELECT SUM(e.duration_ms) FROM episode e + WHERE e.season_item_id = media_item.id AND e.status='present'), 0), + part_count = COALESCE((SELECT SUM(e.part_count) FROM episode e + WHERE e.season_item_id = media_item.id AND e.status='present'), 0), + added_at = (SELECT MIN(e.added_at) FROM episode e + WHERE e.season_item_id = media_item.id AND e.status='present'), + watch_count = COALESCE((SELECT SUM(e.watch_count) FROM episode e + WHERE e.season_item_id = media_item.id), 0), + partial_count = COALESCE((SELECT SUM(e.partial_count) FROM episode e + WHERE e.season_item_id = media_item.id), 0), + abandoned_count = COALESCE((SELECT SUM(e.abandoned_count) FROM episode e + WHERE e.season_item_id = media_item.id), 0), + last_watched_at = (SELECT MAX(e.last_watched_at) FROM episode e + WHERE e.season_item_id = media_item.id), + last_touched_at = (SELECT MAX(e.last_touched_at) FROM episode e + WHERE e.season_item_id = media_item.id) + WHERE kind = 'season' AND provider_id = ? + """, (provider_id,)) + + # distinct watchers across the season's episodes + c.execute(""" + UPDATE media_item SET distinct_watcher_count = COALESCE(( + SELECT COUNT(DISTINCT w.account_id) FROM watch_event w + JOIN episode e ON e.provider_item_id = w.provider_item_id + WHERE e.season_item_id = media_item.id), 0) + WHERE kind = 'season' AND provider_id = ? + """, (provider_id,)) + + # representative path: the common directory of its episodes + c.execute(""" + UPDATE media_item SET primary_path = ( + SELECT p.file_path FROM media_part p + JOIN episode e ON e.id = p.episode_id + WHERE e.season_item_id = media_item.id + ORDER BY e.episode_number LIMIT 1) + WHERE kind = 'season' AND provider_id = ? + """, (provider_id,)) + c.execute(""" + UPDATE media_item SET resolution = ( + SELECT p.resolution FROM media_part p + JOIN episode e ON e.id = p.episode_id + WHERE e.season_item_id = media_item.id AND p.resolution IS NOT NULL + LIMIT 1) + WHERE kind = 'season' AND provider_id = ? + """, (provider_id,)) + + def _rollup_shows(self, provider_id: int) -> None: + self.db.execute(""" + UPDATE media_item SET + episode_count = COALESCE((SELECT SUM(s.episode_count) FROM media_item s + WHERE s.parent_id = media_item.id), 0), + size_bytes = COALESCE((SELECT SUM(s.size_bytes) FROM media_item s + WHERE s.parent_id = media_item.id), 0), + part_count = COALESCE((SELECT SUM(s.part_count) FROM media_item s + WHERE s.parent_id = media_item.id), 0), + duration_ms = COALESCE((SELECT SUM(s.duration_ms) FROM media_item s + WHERE s.parent_id = media_item.id), 0), + added_at = (SELECT MIN(s.added_at) FROM media_item s + WHERE s.parent_id = media_item.id), + watch_count = COALESCE((SELECT SUM(s.watch_count) FROM media_item s + WHERE s.parent_id = media_item.id), 0), + abandoned_count = COALESCE((SELECT SUM(s.abandoned_count) FROM media_item s + WHERE s.parent_id = media_item.id), 0), + last_watched_at = (SELECT MAX(s.last_watched_at) FROM media_item s + WHERE s.parent_id = media_item.id), + last_touched_at = (SELECT MAX(s.last_touched_at) FROM media_item s + WHERE s.parent_id = media_item.id) + WHERE kind = 'show' AND provider_id = ? + """, (provider_id,)) + + def _apply_pre_history(self, provider_id: int, coverage: Coverage | None) -> None: + """Flag items added before watch history began (§4.11). + + On the measured library this is the majority state, not an edge case. + """ + self.db.execute("UPDATE media_item SET pre_history = 0 WHERE provider_id = ?", + (provider_id,)) + if not coverage or not coverage.earliest_event_at: + return + self.db.execute( + "UPDATE media_item SET pre_history = 1 " + "WHERE provider_id = ? AND added_at IS NOT NULL AND added_at > 0 AND added_at < ?", + (provider_id, coverage.earliest_event_at), + ) + + def _mark_missing(self, provider_id: int, scan_id: int, result: ScanResult) -> None: + """Items not seen this full sweep become 'missing', never deleted (§5.4).""" + cur = self.db.execute( + "UPDATE media_item SET status='missing' " + "WHERE provider_id=? AND status='present' " + "AND (last_seen_scan_id IS NULL OR last_seen_scan_id != ?)", + (provider_id, scan_id), + ) + result.items_missing = cur.rowcount if cur.rowcount and cur.rowcount > 0 else 0 + self.db.execute( + "UPDATE episode SET status='missing' " + "WHERE status='present' AND (last_seen_scan_id IS NULL OR last_seen_scan_id != ?)", + (scan_id,), + ) diff --git a/mediashelf/keeps.py b/mediashelf/keeps.py new file mode 100644 index 0000000..3818e1d --- /dev/null +++ b/mediashelf/keeps.py @@ -0,0 +1,328 @@ +"""Keep marks — human judgements the score must not override (§6.6). + +The critical property of this module is that marks survive Plex reassigning +rating keys. They are stored against content identity: + + movie / show : (library_id, guid) + season : (library_id, show_guid, season_number) + +provider_item_id is stored for convenience and linking, and is allowed to go +stale. It is never what a mark is matched on. If that invariant is ever broken, +the failure mode in v2 is deleting content someone explicitly protected — which +is why test_keeps.py reassigns every rating key in the library and asserts the +marks still resolve. +""" + +from __future__ import annotations + +import json +import time + +SCOPES = ("show", "season", "movie") +MODES = ("keep", "exclude") + + +class KeepError(ValueError): + pass + + +# ── resolution ─────────────────────────────────────────────────────────── + +RESOLVE_SQL = """ +UPDATE media_item SET + kept = 0, + kept_via = NULL, + kept_mark_id = NULL +""" + + +def resolve_all(db) -> dict: + """Recompute kept/kept_via for every item. Most specific mark wins. + + Precedence (§6.6): + movie/season explicit 'exclude' -> NOT kept + movie/season explicit 'keep' -> kept + show 'keep' -> kept + library keep_all -> kept + otherwise -> not kept + """ + conn = db.conn + conn.execute("UPDATE media_item SET kept = 0, kept_via = NULL, kept_mark_id = NULL") + + # 1. library rules (weakest) + conn.execute(""" + UPDATE media_item SET kept = 1, kept_via = 'library' + WHERE library_id IN (SELECT id FROM library WHERE keep_all = 1) + """) + + # 2. show marks -> the show row, its seasons + conn.execute(""" + UPDATE media_item SET kept = 1, kept_via = 'show', + kept_mark_id = ( + SELECT k.id FROM keep_mark k + WHERE k.scope = 'show' AND k.mode = 'keep' + AND k.library_id = media_item.library_id + AND k.guid = media_item.guid) + WHERE kind = 'show' AND guid IS NOT NULL AND EXISTS ( + SELECT 1 FROM keep_mark k + WHERE k.scope = 'show' AND k.mode = 'keep' + AND k.library_id = media_item.library_id + AND k.guid = media_item.guid) + """) + conn.execute(""" + UPDATE media_item SET kept = 1, kept_via = 'show', + kept_mark_id = ( + SELECT k.id FROM keep_mark k + WHERE k.scope = 'show' AND k.mode = 'keep' + AND k.library_id = media_item.library_id + AND k.guid = media_item.show_guid) + WHERE kind = 'season' AND show_guid IS NOT NULL AND EXISTS ( + SELECT 1 FROM keep_mark k + WHERE k.scope = 'show' AND k.mode = 'keep' + AND k.library_id = media_item.library_id + AND k.guid = media_item.show_guid) + """) + + # 3. explicit season/movie keeps + conn.execute(""" + UPDATE media_item SET kept = 1, kept_via = 'season', + kept_mark_id = ( + SELECT k.id FROM keep_mark k + WHERE k.scope = 'season' AND k.mode = 'keep' + AND k.library_id = media_item.library_id + AND k.guid = media_item.show_guid + AND k.season_number = media_item.season_number) + WHERE kind = 'season' AND EXISTS ( + SELECT 1 FROM keep_mark k + WHERE k.scope = 'season' AND k.mode = 'keep' + AND k.library_id = media_item.library_id + AND k.guid = media_item.show_guid + AND k.season_number = media_item.season_number) + """) + conn.execute(""" + UPDATE media_item SET kept = 1, kept_via = 'movie', + kept_mark_id = ( + SELECT k.id FROM keep_mark k + WHERE k.scope = 'movie' AND k.mode = 'keep' + AND k.library_id = media_item.library_id + AND k.guid = media_item.guid) + WHERE kind = 'movie' AND guid IS NOT NULL AND EXISTS ( + SELECT 1 FROM keep_mark k + WHERE k.scope = 'movie' AND k.mode = 'keep' + AND k.library_id = media_item.library_id + AND k.guid = media_item.guid) + """) + + # 4. explicit excludes win over everything above + conn.execute(""" + UPDATE media_item SET kept = 0, kept_via = NULL, kept_mark_id = NULL + WHERE kind = 'season' AND EXISTS ( + SELECT 1 FROM keep_mark k + WHERE k.scope = 'season' AND k.mode = 'exclude' + AND k.library_id = media_item.library_id + AND k.guid = media_item.show_guid + AND k.season_number = media_item.season_number) + """) + conn.execute(""" + UPDATE media_item SET kept = 0, kept_via = NULL, kept_mark_id = NULL + WHERE kind = 'movie' AND guid IS NOT NULL AND EXISTS ( + SELECT 1 FROM keep_mark k + WHERE k.scope = 'movie' AND k.mode = 'exclude' + AND k.library_id = media_item.library_id + AND k.guid = media_item.guid) + """) + + kept_items = db.scalar("SELECT COUNT(*) FROM media_item WHERE kept = 1 AND kind != 'show'") or 0 + kept_bytes = db.scalar( + "SELECT COALESCE(SUM(size_bytes),0) FROM media_item WHERE kept = 1 AND kind != 'show'" + ) or 0 + return {"kept_items": kept_items, "kept_bytes": kept_bytes} + + +def mark_matches(db) -> dict[int, int]: + """How many items each mark currently resolves to. 0 => orphaned.""" + counts: dict[int, int] = {} + for row in db.query("SELECT id FROM keep_mark"): + counts[row["id"]] = 0 + for row in db.query( + "SELECT kept_mark_id AS mid, COUNT(*) AS n FROM media_item " + "WHERE kept_mark_id IS NOT NULL GROUP BY kept_mark_id" + ): + counts[row["mid"]] = row["n"] + + # Excludes never set kept_mark_id, so count them separately. + for row in db.query("SELECT * FROM keep_mark WHERE mode = 'exclude'"): + if row["scope"] == "movie": + n = db.scalar( + "SELECT COUNT(*) FROM media_item WHERE kind='movie' " + "AND library_id=? AND guid=?", + (row["library_id"], row["guid"]), + ) + else: + n = db.scalar( + "SELECT COUNT(*) FROM media_item WHERE kind='season' " + "AND library_id=? AND show_guid=? AND season_number=?", + (row["library_id"], row["guid"], row["season_number"]), + ) + counts[row["id"]] = n or 0 + return counts + + +def stamp_matches(db, scan_id: int | None) -> int: + """Record which marks matched something this scan. Returns orphan count.""" + counts = mark_matches(db) + orphans = 0 + for mark_id, n in counts.items(): + if n > 0: + db.execute("UPDATE keep_mark SET last_matched_scan_id = ? WHERE id = ?", + (scan_id, mark_id)) + else: + orphans += 1 + return orphans + + +# ── mutation ───────────────────────────────────────────────────────────── + +def create_from_item(db, item_id: int, mode: str = "keep", note: str | None = None) -> int: + """Create a mark from a grid row, resolving it to the durable key.""" + if mode not in MODES: + raise KeepError("mode must be 'keep' or 'exclude'") + row = db.one( + "SELECT i.*, lib.title AS library_title FROM media_item i " + "JOIN library lib ON lib.id = i.library_id WHERE i.id = ?", + (item_id,), + ) + if row is None: + raise KeepError("no such item") + + kind = row["kind"] + if kind == "movie": + scope, guid, season = "movie", row["guid"], None + label = row["title"] + (" (%s)" % row["year"] if row["year"] else "") + elif kind == "show": + scope, guid, season = "show", row["guid"], None + label = row["title"] + elif kind == "season": + scope, guid, season = "season", row["show_guid"], row["season_number"] + parent = db.one("SELECT title FROM media_item WHERE id = ?", (row["parent_id"],)) + show_title = parent["title"] if parent else "(unknown show)" + label = "%s — Season %s" % (show_title, row["season_number"]) + else: + raise KeepError("cannot keep a %s" % kind) + + if not guid: + raise KeepError( + "this item has no Plex GUID, so a keep could not survive a library " + "rebuild; refusing to create one that would silently detach" + ) + return upsert(db, scope, mode, row["library_id"], guid, season, + label=label, note=note, provider_item_id=row["provider_item_id"]) + + +def upsert(db, scope: str, mode: str, library_id: int, guid: str, + season_number: int | None, *, label: str, note: str | None = None, + provider_item_id: str | None = None) -> int: + if scope not in SCOPES: + raise KeepError("scope must be one of %s" % (SCOPES,)) + if mode not in MODES: + raise KeepError("mode must be 'keep' or 'exclude'") + if scope == "season" and season_number is None: + raise KeepError("a season mark needs a season_number") + if scope != "season": + season_number = None + + now = int(time.time()) + existing = db.one( + "SELECT id FROM keep_mark WHERE scope=? AND library_id=? AND guid=? " + "AND season_number IS ?", + (scope, library_id, guid, season_number), + ) + if existing: + db.execute( + "UPDATE keep_mark SET mode=?, note=COALESCE(?, note), label=?, " + "provider_item_id=?, updated_at=? WHERE id=?", + (mode, note, label, provider_item_id, now, existing["id"]), + ) + return existing["id"] + + cur = db.execute( + "INSERT INTO keep_mark (scope, mode, library_id, guid, season_number, " + "provider_item_id, label, note, created_at, updated_at) " + "VALUES (?,?,?,?,?,?,?,?,?,?)", + (scope, mode, library_id, guid, season_number, provider_item_id, + label, note, now, now), + ) + return cur.lastrowid + + +def delete(db, mark_id: int) -> bool: + cur = db.execute("DELETE FROM keep_mark WHERE id = ?", (mark_id,)) + return cur.rowcount > 0 + + +def set_library_keep_all(db, library_id: int, keep_all: bool) -> None: + db.execute("UPDATE library SET keep_all = ? WHERE id = ?", + (1 if keep_all else 0, library_id)) + + +# ── export / import (§11.5) ────────────────────────────────────────────── + +def export(db) -> dict: + """Human-readable, durable-key export. The only irreplaceable data here.""" + libs = {r["id"]: r["title"] for r in db.query("SELECT id, title FROM library")} + marks = [] + for r in db.query("SELECT * FROM keep_mark ORDER BY id"): + marks.append({ + "scope": r["scope"], + "mode": r["mode"], + "library": libs.get(r["library_id"], str(r["library_id"])), + "guid": r["guid"], + "season_number": r["season_number"], + "label": r["label"], + "note": r["note"], + "created_at": r["created_at"], + }) + library_rules = [r["title"] for r in + db.query("SELECT title FROM library WHERE keep_all = 1 ORDER BY title")] + return { + "version": 1, + "exported_at": int(time.time()), + "library_keep_all": library_rules, + "marks": marks, + } + + +def import_(db, payload: dict) -> dict: + """Restore an export. Matches libraries by title, since ids are local.""" + if not isinstance(payload, dict) or payload.get("version") != 1: + raise KeepError("unrecognised keep export") + + by_title = {r["title"]: r["id"] for r in db.query("SELECT id, title FROM library")} + added = skipped = 0 + + for title in payload.get("library_keep_all") or []: + lid = by_title.get(title) + if lid: + set_library_keep_all(db, lid, True) + else: + skipped += 1 + + for m in payload.get("marks") or []: + lid = by_title.get(m.get("library")) + if not lid or not m.get("guid"): + skipped += 1 + continue + try: + upsert(db, m["scope"], m.get("mode", "keep"), lid, m["guid"], + m.get("season_number"), label=m.get("label") or m["guid"], + note=m.get("note")) + added += 1 + except KeepError: + skipped += 1 + resolve_all(db) + return {"imported": added, "skipped": skipped} + + +def write_export_file(db, path: str) -> None: + with open(path, "w") as fh: + json.dump(export(db), fh, indent=2) diff --git a/mediashelf/migrations/001_initial.sql b/mediashelf/migrations/001_initial.sql new file mode 100644 index 0000000..21e07ba --- /dev/null +++ b/mediashelf/migrations/001_initial.sql @@ -0,0 +1,245 @@ +-- MediaShelf initial schema. See docs/design.md §5. + +CREATE TABLE provider ( + id INTEGER PRIMARY KEY, + kind TEXT NOT NULL, + name TEXT NOT NULL, + base_url TEXT NOT NULL, + server_id TEXT, + version TEXT, + last_scan_id INTEGER, + created_at INTEGER NOT NULL, + UNIQUE (kind, base_url) +); + +CREATE TABLE library ( + id INTEGER PRIMARY KEY, + provider_id INTEGER NOT NULL REFERENCES provider(id), + provider_key TEXT NOT NULL, + title TEXT NOT NULL, + kind TEXT NOT NULL, -- 'movie' | 'show' + locations TEXT, -- JSON array + keep_all INTEGER NOT NULL DEFAULT 0, + scanned_at INTEGER, + UNIQUE (provider_id, provider_key) +); + +CREATE TABLE media_item ( + id INTEGER PRIMARY KEY, + provider_id INTEGER NOT NULL REFERENCES provider(id), + library_id INTEGER NOT NULL REFERENCES library(id), + provider_item_id TEXT NOT NULL, + kind TEXT NOT NULL, -- 'movie' | 'show' | 'season' + guid TEXT, + show_guid TEXT, -- seasons: the parent show's guid + title TEXT NOT NULL, + sort_title TEXT, + year INTEGER, + parent_id INTEGER REFERENCES media_item(id), + season_number INTEGER, + added_at INTEGER, + updated_at INTEGER, + episode_count INTEGER NOT NULL DEFAULT 0, + size_bytes INTEGER NOT NULL DEFAULT 0, + duration_ms INTEGER NOT NULL DEFAULT 0, + part_count INTEGER NOT NULL DEFAULT 0, + primary_path TEXT, + resolution TEXT, + video_codec TEXT, + watch_count INTEGER NOT NULL DEFAULT 0, + partial_count INTEGER NOT NULL DEFAULT 0, + abandoned_count INTEGER NOT NULL DEFAULT 0, + last_watched_at INTEGER, + last_touched_at INTEGER, + first_watched_at INTEGER, + distinct_watcher_count INTEGER NOT NULL DEFAULT 0, + avg_percent_complete REAL, + pre_history INTEGER NOT NULL DEFAULT 0, + kept INTEGER NOT NULL DEFAULT 0, + kept_via TEXT, + kept_mark_id INTEGER, + provider_view_count INTEGER NOT NULL DEFAULT 0, + status TEXT NOT NULL DEFAULT 'present', + first_seen_scan_id INTEGER, + last_seen_scan_id INTEGER, + UNIQUE (provider_id, provider_item_id) +); + +CREATE TABLE episode ( + id INTEGER PRIMARY KEY, + season_item_id INTEGER NOT NULL REFERENCES media_item(id) ON DELETE CASCADE, + provider_item_id TEXT NOT NULL, + episode_number INTEGER, + title TEXT, + added_at INTEGER, + duration_ms INTEGER NOT NULL DEFAULT 0, + size_bytes INTEGER NOT NULL DEFAULT 0, + part_count INTEGER NOT NULL DEFAULT 0, + watch_count INTEGER NOT NULL DEFAULT 0, + partial_count INTEGER NOT NULL DEFAULT 0, + abandoned_count INTEGER NOT NULL DEFAULT 0, + last_watched_at INTEGER, + last_touched_at INTEGER, + status TEXT NOT NULL DEFAULT 'present', + last_seen_scan_id INTEGER, + UNIQUE (provider_item_id) +); + +CREATE TABLE media_part ( + id INTEGER PRIMARY KEY, + media_item_id INTEGER REFERENCES media_item(id) ON DELETE CASCADE, + episode_id INTEGER REFERENCES episode(id) ON DELETE CASCADE, + provider_part_id TEXT, + file_path TEXT NOT NULL, + size_bytes INTEGER NOT NULL DEFAULT 0, + container TEXT, + resolution TEXT, + video_codec TEXT, + audio_codec TEXT, + bitrate INTEGER, + CHECK ((media_item_id IS NULL) != (episode_id IS NULL)) +); + +CREATE TABLE watch_event ( + id INTEGER PRIMARY KEY, + provider_id INTEGER NOT NULL REFERENCES provider(id), + source TEXT NOT NULL, -- 'tautulli' | 'plex' + source_row_id TEXT NOT NULL, + reference_id TEXT, + provider_item_id TEXT NOT NULL, + account_id TEXT, + viewed_at INTEGER NOT NULL, + stopped_at INTEGER, + play_duration_s INTEGER, + paused_counter_s INTEGER, + percent_complete INTEGER, + watched_status REAL, + disposition TEXT NOT NULL, -- completed | partial | abandoned + session_id TEXT, + media_type TEXT, + platform TEXT, + UNIQUE (provider_id, source, source_row_id) +); + +CREATE TABLE account ( + id INTEGER PRIMARY KEY, + provider_id INTEGER NOT NULL REFERENCES provider(id), + account_id TEXT NOT NULL, + name TEXT, + friendly_name TEXT, + UNIQUE (provider_id, account_id) +); + +CREATE TABLE history_coverage ( + id INTEGER PRIMARY KEY, + provider_id INTEGER NOT NULL REFERENCES provider(id), + source TEXT NOT NULL, + earliest_event_at INTEGER, + latest_event_at INTEGER, + event_count INTEGER NOT NULL DEFAULT 0, + updated_at INTEGER NOT NULL, + UNIQUE (provider_id, source) +); + +-- Keep marks: human judgements the score must not override (§6.6). +-- Keyed on CONTENT identity, never on provider_item_id. +CREATE TABLE keep_mark ( + id INTEGER PRIMARY KEY, + scope TEXT NOT NULL, -- 'show' | 'season' | 'movie' + mode TEXT NOT NULL, -- 'keep' | 'exclude' + library_id INTEGER NOT NULL REFERENCES library(id), + guid TEXT NOT NULL, + season_number INTEGER, + provider_item_id TEXT, + label TEXT NOT NULL, + note TEXT, + created_at INTEGER NOT NULL, + updated_at INTEGER NOT NULL, + last_matched_scan_id INTEGER, + UNIQUE (scope, library_id, guid, season_number) +); + +CREATE TABLE scan ( + id INTEGER PRIMARY KEY, + provider_id INTEGER REFERENCES provider(id), + mode TEXT NOT NULL, -- 'full' | 'incremental' + trigger TEXT NOT NULL, -- 'manual' | 'schedule' | 'startup' + status TEXT NOT NULL, -- 'running' | 'succeeded' | 'failed' + history_source TEXT, + started_at INTEGER NOT NULL, + finished_at INTEGER, + progress TEXT, + items_seen INTEGER NOT NULL DEFAULT 0, + items_added INTEGER NOT NULL DEFAULT 0, + items_updated INTEGER NOT NULL DEFAULT 0, + items_missing INTEGER NOT NULL DEFAULT 0, + events_added INTEGER NOT NULL DEFAULT 0, + warning_count INTEGER NOT NULL DEFAULT 0, + warnings TEXT, + error TEXT +); + +CREATE TABLE scan_lock ( + id INTEGER PRIMARY KEY CHECK (id = 1), + scan_id INTEGER, + holder TEXT, + acquired_at INTEGER +); + +CREATE TABLE saved_view ( + id INTEGER PRIMARY KEY, + name TEXT NOT NULL UNIQUE, + description TEXT, + rules TEXT NOT NULL, + sort TEXT, + columns TEXT, + weights TEXT, + builtin INTEGER NOT NULL DEFAULT 0, + created_at INTEGER NOT NULL, + updated_at INTEGER NOT NULL +); + +CREATE TABLE setting ( + key TEXT PRIMARY KEY, + value TEXT NOT NULL +); + +CREATE INDEX ix_item_library ON media_item(library_id, kind, status); +CREATE INDEX ix_item_added ON media_item(added_at); +CREATE INDEX ix_item_lastwatch ON media_item(last_watched_at); +CREATE INDEX ix_item_size ON media_item(size_bytes); +CREATE INDEX ix_item_watchcount ON media_item(watch_count); +CREATE INDEX ix_item_parent ON media_item(parent_id); +CREATE INDEX ix_item_guid ON media_item(guid); +CREATE INDEX ix_item_showguid ON media_item(show_guid); +CREATE INDEX ix_item_kept ON media_item(kept, library_id); +CREATE INDEX ix_episode_season ON episode(season_item_id); +CREATE INDEX ix_part_item ON media_part(media_item_id); +CREATE INDEX ix_part_episode ON media_part(episode_id); +CREATE INDEX ix_event_item ON watch_event(provider_item_id); +CREATE INDEX ix_event_viewed ON watch_event(viewed_at); +CREATE INDEX ix_event_disp ON watch_event(provider_item_id, disposition); +CREATE INDEX ix_event_account ON watch_event(account_id, viewed_at); +CREATE INDEX ix_keep_lookup ON keep_mark(library_id, guid, season_number); + +-- Full-text search over titles, kept in sync by triggers. +CREATE VIRTUAL TABLE media_item_fts USING fts5( + title, sort_title, content='media_item', content_rowid='id' +); + +CREATE TRIGGER media_item_ai AFTER INSERT ON media_item BEGIN + INSERT INTO media_item_fts(rowid, title, sort_title) + VALUES (new.id, new.title, new.sort_title); +END; + +CREATE TRIGGER media_item_ad AFTER DELETE ON media_item BEGIN + INSERT INTO media_item_fts(media_item_fts, rowid, title, sort_title) + VALUES ('delete', old.id, old.title, old.sort_title); +END; + +CREATE TRIGGER media_item_au AFTER UPDATE OF title, sort_title ON media_item BEGIN + INSERT INTO media_item_fts(media_item_fts, rowid, title, sort_title) + VALUES ('delete', old.id, old.title, old.sort_title); + INSERT INTO media_item_fts(rowid, title, sort_title) + VALUES (new.id, new.title, new.sort_title); +END; diff --git a/mediashelf/providers/__init__.py b/mediashelf/providers/__init__.py new file mode 100644 index 0000000..a6783bd --- /dev/null +++ b/mediashelf/providers/__init__.py @@ -0,0 +1,68 @@ +"""Provider construction and history-source selection.""" + +from __future__ import annotations + +import logging + +from .base import ( # noqa: F401 + Account, + AuthError, + Coverage, + HistoryProvider, + Item, + Library, + MediaProvider, + Part, + ProviderError, + ServerInfo, + WatchEvent, +) +from .plex import PlexClient, PlexHistoryProvider, PlexProvider +from .tautulli import TautulliClient, TautulliHistoryProvider + +log = logging.getLogger(__name__) + + +def build_media_provider(cfg) -> PlexProvider: + client = PlexClient( + cfg.plex_base_url, cfg.plex_token, + timeout=cfg.plex_timeout_s, + verify_ssl=cfg.plex_verify_ssl, + page_size=cfg.plex_page_size, + request_delay_ms=cfg.plex_request_delay_ms, + ) + return PlexProvider(client) + + +def build_history_provider(cfg, media: PlexProvider): + """Pick a history source per HISTORY_SOURCE (§4.11). + + 'auto' prefers Tautulli and falls back to Plex, reporting the reason rather + than degrading silently — a silently degraded score is one that gets trusted + when it shouldn't be. Returns (provider, degraded_reason_or_None). + """ + mode = (cfg.history_source or "auto").lower() + + if mode == "plex": + return PlexHistoryProvider(media.client), None + + if mode in ("auto", "tautulli"): + if not cfg.tautulli_configured: + if mode == "tautulli": + raise ProviderError("HISTORY_SOURCE=tautulli but Tautulli is not configured") + return PlexHistoryProvider(media.client), "Tautulli is not configured" + client = TautulliClient( + cfg.tautulli_base_url, cfg.tautulli_api_key, + timeout=cfg.tautulli_timeout_s, page_size=cfg.tautulli_page_size, + ) + provider = TautulliHistoryProvider(client) + try: + provider.server_info() + return provider, None + except ProviderError as e: + if mode == "tautulli": + raise + log.warning("Tautulli unreachable, falling back to Plex history: %s", e) + return PlexHistoryProvider(media.client), "Tautulli unreachable: %s" % e + + raise ProviderError("unknown HISTORY_SOURCE %r" % cfg.history_source) diff --git a/mediashelf/providers/base.py b/mediashelf/providers/base.py new file mode 100644 index 0000000..105d4ec --- /dev/null +++ b/mediashelf/providers/base.py @@ -0,0 +1,139 @@ +"""Provider protocols and the normalized types everything above this layer speaks. + +Two protocols, deliberately separate (§3.1): MediaProvider knows what exists, +HistoryProvider knows what was watched. On this network those are two different +machines — Plex on Loki, Tautulli on Isis — and Emby/Jellyfin later will have no +Tautulli equivalent at all. +""" + +from __future__ import annotations + +from dataclasses import dataclass, field +from typing import Iterator, Protocol, runtime_checkable + + +class ProviderError(RuntimeError): + """Any failure talking to a source. Carries a message safe to display.""" + + +class AuthError(ProviderError): + """Credentials rejected. Never retried — a retry loop won't fix a bad token.""" + + +@dataclass +class ServerInfo: + kind: str + name: str + version: str = "" + server_id: str = "" # Plex machineIdentifier / Tautulli pms_identifier + base_url: str = "" + + +@dataclass +class Library: + provider_key: str + title: str + kind: str # 'movie' | 'show' + locations: list[str] = field(default_factory=list) + + +@dataclass +class Part: + file_path: str + size_bytes: int = 0 + provider_part_id: str | None = None + container: str | None = None + resolution: str | None = None + video_codec: str | None = None + audio_codec: str | None = None + bitrate: int | None = None + + +@dataclass +class Item: + """A movie or an episode as the source reports it. + + Episodes carry their season and show identity so the ingest can roll them up + without a second pass over the API. + """ + + provider_item_id: str + kind: str # 'movie' | 'episode' + title: str + library_key: str + guid: str | None = None + sort_title: str | None = None + year: int | None = None + added_at: int | None = None + updated_at: int | None = None + duration_ms: int = 0 + view_count: int = 0 + last_viewed_at: int | None = None + resolution: str | None = None + video_codec: str | None = None + parts: list[Part] = field(default_factory=list) + # episode-only + show_id: str | None = None + show_title: str | None = None + show_guid: str | None = None + season_id: str | None = None + season_number: int | None = None + episode_number: int | None = None + + @property + def size_bytes(self) -> int: + return sum(p.size_bytes for p in self.parts) + + +@dataclass +class WatchEvent: + """One playback event, normalized across sources.""" + + source: str # 'tautulli' | 'plex' + source_row_id: str + provider_item_id: str + viewed_at: int + account_id: str | None = None + reference_id: str | None = None + stopped_at: int | None = None + play_duration_s: int | None = None + paused_counter_s: int | None = None + percent_complete: int | None = None + watched_status: float | None = None + media_type: str | None = None + platform: str | None = None + + +@dataclass +class Account: + account_id: str + name: str | None = None + friendly_name: str | None = None + + +@dataclass +class Coverage: + earliest_event_at: int | None + latest_event_at: int | None + event_count: int + + +@runtime_checkable +class MediaProvider(Protocol): + def server_info(self) -> ServerInfo: ... + def libraries(self) -> list[Library]: ... + def items(self, library: Library) -> Iterator[Item]: ... + def refresh_library(self, library: Library) -> None: ... # v2 only + + +@runtime_checkable +class HistoryProvider(Protocol): + name: str + + def server_info(self) -> ServerInfo: ... + def accounts(self) -> list[Account]: ... + def watch_events(self, since: int | None = None) -> Iterator[WatchEvent]: ... + def coverage(self) -> Coverage: ... + + @property + def has_completion_data(self) -> bool: ... diff --git a/mediashelf/providers/plex.py b/mediashelf/providers/plex.py new file mode 100644 index 0000000..527d4f1 --- /dev/null +++ b/mediashelf/providers/plex.py @@ -0,0 +1,294 @@ +"""Plex Media Server provider. + +Talks directly to the server on the LAN with a server token — no plex.tv round +trip, so it works whether or not Plex's cloud is up. JSON throughout via the +Accept header, which avoids XML parsing entirely (§4.1). +""" + +from __future__ import annotations + +import logging +import time +from typing import Iterator + +import requests + +from .base import ( + Account, + AuthError, + Coverage, + Item, + Library, + Part, + ProviderError, + ServerInfo, + WatchEvent, +) + +log = logging.getLogger(__name__) + +LIBTYPE_MOVIE = 1 +LIBTYPE_SHOW = 2 +LIBTYPE_SEASON = 3 +LIBTYPE_EPISODE = 4 + + +def _int(v, default=0): + try: + return int(v) + except (TypeError, ValueError): + return default + + +def _opt_int(v): + try: + return int(v) + except (TypeError, ValueError): + return None + + +class PlexClient: + def __init__(self, base_url: str, token: str, *, timeout: int = 30, + verify_ssl: bool = True, page_size: int = 500, + request_delay_ms: int = 0, client_id: str = "mediashelf"): + if not base_url or not token: + raise ProviderError("Plex is not configured (PLEX_BASE_URL / PLEX_TOKEN)") + self.base_url = base_url.rstrip("/") + self.timeout = timeout + self.verify_ssl = verify_ssl + self.page_size = page_size + self.delay = request_delay_ms / 1000.0 + self.session = requests.Session() + self.session.headers.update({ + "X-Plex-Token": token, + "Accept": "application/json", + "X-Plex-Product": "MediaShelf", + "X-Plex-Client-Identifier": client_id, + }) + + def get(self, path: str, params: dict | None = None, *, headers: dict | None = None) -> dict: + url = self.base_url + path + last: Exception | None = None + for attempt in range(3): + try: + if self.delay: + time.sleep(self.delay) + r = self.session.get(url, params=params, headers=headers, + timeout=self.timeout, verify=self.verify_ssl) + if r.status_code in (401, 403): + raise AuthError("Plex rejected the token (HTTP %s)" % r.status_code) + r.raise_for_status() + return r.json().get("MediaContainer", {}) or {} + except AuthError: + raise # never retried: a retry loop will not fix a bad token + except Exception as e: # noqa: BLE001 + last = e + if attempt < 2: + time.sleep(2 ** attempt) + raise ProviderError("Plex request failed: %s (%s)" % (path, last)) + + def paged(self, path: str, params: dict | None = None) -> Iterator[dict]: + """Yield Metadata rows, paging with X-Plex-Container-* headers.""" + start = 0 + while True: + headers = { + "X-Plex-Container-Start": str(start), + "X-Plex-Container-Size": str(self.page_size), + } + mc = self.get(path, params, headers=headers) + batch = mc.get("Metadata") or [] + if not batch: + return + for row in batch: + yield row + start += len(batch) + total = _int(mc.get("totalSize") or mc.get("size") or 0) + if start >= total: + return + + +class PlexProvider: + """MediaProvider implementation.""" + + kind = "plex" + + def __init__(self, client: PlexClient): + self.client = client + + # ── identity ───────────────────────────────────────────────────────── + + def server_info(self) -> ServerInfo: + mc = self.client.get("/identity") + return ServerInfo( + kind="plex", + name=mc.get("friendlyName") or "Plex", + version=mc.get("version") or "", + server_id=mc.get("machineIdentifier") or "", + base_url=self.client.base_url, + ) + + # ── libraries ──────────────────────────────────────────────────────── + + def libraries(self) -> list[Library]: + mc = self.client.get("/library/sections") + out = [] + for d in mc.get("Directory") or []: + kind = d.get("type") + if kind not in ("movie", "show"): + continue # music/photo/other are out of scope (§1.2) + out.append(Library( + provider_key=str(d.get("key")), + title=(d.get("title") or "").strip(), + kind=kind, + locations=[loc.get("path") for loc in (d.get("Location") or []) + if loc.get("path")], + )) + return out + + # ── items ──────────────────────────────────────────────────────────── + + def items(self, library: Library) -> Iterator[Item]: + libtype = LIBTYPE_MOVIE if library.kind == "movie" else LIBTYPE_EPISODE + path = "/library/sections/%s/all" % library.provider_key + for row in self.client.paged(path, {"type": libtype}): + item = self._to_item(row, library) + if item is not None: + yield item + + def show_guids(self, library: Library) -> dict[str, str]: + """Map show ratingKey -> show guid. + + Episodes carry grandparentRatingKey but not the show's guid, and keep + marks for seasons are keyed on the SHOW's guid (§6.6), so this mapping + has to be fetched separately — one cheap request per show library. + """ + out: dict[str, str] = {} + path = "/library/sections/%s/all" % library.provider_key + for row in self.client.paged(path, {"type": LIBTYPE_SHOW}): + rk = row.get("ratingKey") + if rk is not None: + out[str(rk)] = row.get("guid") or "" + return out + + @staticmethod + def _to_item(row: dict, library: Library) -> Item | None: + rk = row.get("ratingKey") + if rk is None: + return None + + parts: list[Part] = [] + resolution = video_codec = None + for media in row.get("Media") or []: + if resolution is None: + resolution = media.get("videoResolution") + video_codec = media.get("videoCodec") + for p in media.get("Part") or []: + if not p.get("file"): + continue + parts.append(Part( + file_path=p["file"], + size_bytes=_int(p.get("size")), + provider_part_id=str(p.get("id")) if p.get("id") is not None else None, + container=p.get("container") or media.get("container"), + resolution=media.get("videoResolution"), + video_codec=media.get("videoCodec"), + audio_codec=media.get("audioCodec"), + bitrate=_opt_int(media.get("bitrate")), + )) + + kind = "movie" if library.kind == "movie" else "episode" + return Item( + provider_item_id=str(rk), + kind=kind, + title=row.get("title") or "(untitled)", + library_key=library.provider_key, + guid=row.get("guid"), + sort_title=row.get("titleSort"), + year=_opt_int(row.get("year")), + added_at=_opt_int(row.get("addedAt")), + updated_at=_opt_int(row.get("updatedAt")), + duration_ms=_int(row.get("duration")), + view_count=_int(row.get("viewCount")), + last_viewed_at=_opt_int(row.get("lastViewedAt")), + resolution=resolution, + video_codec=video_codec, + parts=parts, + show_id=str(row["grandparentRatingKey"]) if row.get("grandparentRatingKey") is not None else None, + show_title=row.get("grandparentTitle"), + season_id=str(row["parentRatingKey"]) if row.get("parentRatingKey") is not None else None, + season_number=_opt_int(row.get("parentIndex")), + episode_number=_opt_int(row.get("index")), + ) + + def refresh_library(self, library: Library) -> None: + raise NotImplementedError("v1 never writes to Plex (§1.2)") + + +class PlexHistoryProvider: + """Fallback HistoryProvider using /status/sessions/history/all. + + Server-wide across accounts, which solves the token-scoping problem — but it + records only that a play happened, never how far it got. Everything it + returns is therefore treated as a completed view and the 'rejection' score + component drops out of the weighting (§4.11). + """ + + name = "plex" + + def __init__(self, client: PlexClient): + self.client = client + + @property + def has_completion_data(self) -> bool: + return False + + def server_info(self) -> ServerInfo: + mc = self.client.get("/identity") + return ServerInfo( + kind="plex", + name=mc.get("friendlyName") or "Plex", + version=mc.get("version") or "", + server_id=mc.get("machineIdentifier") or "", + base_url=self.client.base_url, + ) + + def accounts(self) -> list[Account]: + try: + mc = self.client.get("/accounts") + except ProviderError: + return [] + out = [] + for a in mc.get("Account") or []: + if a.get("id") is None: + continue + out.append(Account(account_id=str(a["id"]), name=a.get("name"))) + return out + + def watch_events(self, since: int | None = None) -> Iterator[WatchEvent]: + params: dict = {"sort": "viewedAt:desc"} + for row in self.client.paged("/status/sessions/history/all", params): + viewed = _opt_int(row.get("viewedAt")) + rk = row.get("ratingKey") + if viewed is None or rk is None: + continue + if since is not None and viewed <= since: + return # sorted desc: the watermark ends the walk + yield WatchEvent( + source="plex", + source_row_id=str(row.get("historyKey") or f"{rk}:{viewed}"), + provider_item_id=str(rk), + viewed_at=viewed, + account_id=str(row["accountID"]) if row.get("accountID") is not None else None, + media_type=row.get("type"), + ) + + def coverage(self) -> Coverage: + earliest = latest = None + count = 0 + for ev in self.watch_events(): + count += 1 + if latest is None or ev.viewed_at > latest: + latest = ev.viewed_at + if earliest is None or ev.viewed_at < earliest: + earliest = ev.viewed_at + return Coverage(earliest, latest, count) diff --git a/mediashelf/providers/tautulli.py b/mediashelf/providers/tautulli.py new file mode 100644 index 0000000..952eb91 --- /dev/null +++ b/mediashelf/providers/tautulli.py @@ -0,0 +1,220 @@ +"""Tautulli history provider — the primary watch-data source (§4.8). + +Tautulli's database is independent of Plex's: clearing or pruning Plex history +does not touch it, and unlike Plex it records how far into an item each play +actually got. That last fact is the whole reason this is the primary source. + +Two things this module deliberately does NOT do: + + * It never calls get_library_media_info for a show section. Measured against + the live server, that returns show-level rows with file_size 0 regardless of + the section_type parameter, so it cannot supply TV sizes (§4.11). Plex is the + only size authority. + * It never trusts an HTTP 200 as success. Tautulli returns 200 with + result:"error" in the body as its normal failure mode. +""" + +from __future__ import annotations + +import logging +import time +from typing import Iterator + +import requests + +from .base import ( + Account, + AuthError, + Coverage, + ProviderError, + ServerInfo, + WatchEvent, +) + +log = logging.getLogger(__name__) + + +def _int(v, default=0): + try: + return int(v) + except (TypeError, ValueError): + return default + + +def _opt_int(v): + try: + return int(v) + except (TypeError, ValueError): + return None + + +class TautulliClient: + def __init__(self, base_url: str, api_key: str, *, timeout: int = 30, + page_size: int = 1000): + if not base_url or not api_key: + raise ProviderError("Tautulli is not configured") + self.base_url = base_url.rstrip("/") + "/api/v2" + self.api_key = api_key + self.timeout = timeout + self.page_size = page_size + self.session = requests.Session() + + def cmd(self, command: str, **params): + params["apikey"] = self.api_key + params["cmd"] = command + last: Exception | None = None + for attempt in range(3): + try: + r = self.session.get(self.base_url, params=params, timeout=self.timeout) + if r.status_code in (401, 403): + raise AuthError("Tautulli rejected the API key (HTTP %s)" % r.status_code) + r.raise_for_status() + body = r.json() + resp = (body or {}).get("response") or {} + result = resp.get("result") + if result != "success": + msg = resp.get("message") or "unknown error" + # An invalid key surfaces here as a 200 with result:error. + if "apikey" in str(msg).lower() or "auth" in str(msg).lower(): + raise AuthError("Tautulli: %s" % msg) + raise ProviderError("Tautulli cmd=%s failed: %s" % (command, msg)) + return resp.get("data") + except (AuthError, ProviderError): + raise + except Exception as e: # noqa: BLE001 + last = e + if attempt < 2: + time.sleep(2 ** attempt) + raise ProviderError("Tautulli request failed: cmd=%s (%s)" % (command, last)) + + +class TautulliHistoryProvider: + """HistoryProvider implementation.""" + + name = "tautulli" + + def __init__(self, client: TautulliClient): + self.client = client + + @property + def has_completion_data(self) -> bool: + return True + + def server_info(self) -> ServerInfo: + d = self.client.cmd("get_server_info") or {} + return ServerInfo( + kind="tautulli", + name=d.get("pms_name") or "Tautulli", + version="", + server_id=d.get("pms_identifier") or "", + base_url=self.client.base_url, + ) + + def accounts(self) -> list[Account]: + data = self.client.cmd("get_users") or [] + out = [] + for u in data: + uid = u.get("user_id") + if uid is None: + continue + out.append(Account( + account_id=str(uid), + name=u.get("username"), + friendly_name=u.get("friendly_name") or u.get("username"), + )) + return out + + def _history_page(self, start: int, length: int, order_dir: str = "desc") -> tuple[list[dict], int]: + d = self.client.cmd( + "get_history", + grouping=0, # raw events; MediaShelf does its own merging (§4.9) + order_column="date", + order_dir=order_dir, + start=start, + length=length, + ) or {} + rows = d.get("data") or [] + total = _int(d.get("recordsFiltered") or d.get("recordsTotal") or 0) + return rows, total + + def watch_events(self, since: int | None = None) -> Iterator[WatchEvent]: + """Walk history newest-first, stopping at the watermark.""" + start = 0 + while True: + rows, total = self._history_page(start, self.client.page_size) + if not rows: + return + for row in rows: + viewed = _opt_int(row.get("date")) + rk = row.get("rating_key") + if viewed is None or rk in (None, ""): + continue + if since is not None and viewed <= since: + return + yield self._to_event(row, viewed, rk) + start += len(rows) + if start >= total: + return + + @staticmethod + def _to_event(row: dict, viewed: int, rk) -> WatchEvent: + row_id = row.get("row_id") + return WatchEvent( + source="tautulli", + source_row_id=str(row_id if row_id is not None else f"{rk}:{viewed}"), + provider_item_id=str(rk), + viewed_at=viewed, + account_id=str(row["user_id"]) if row.get("user_id") is not None else None, + reference_id=str(row["reference_id"]) if row.get("reference_id") is not None else None, + stopped_at=_opt_int(row.get("stopped")), + play_duration_s=_opt_int(row.get("play_duration")), + paused_counter_s=_opt_int(row.get("paused_counter")), + percent_complete=_opt_int(row.get("percent_complete")), + watched_status=_float_or_none(row.get("watched_status")), + media_type=row.get("media_type"), + platform=row.get("platform"), + ) + + def coverage(self) -> Coverage: + """Cheap: one row from each end plus the reported total.""" + newest, total = self._history_page(0, 1, "desc") + oldest, _ = self._history_page(0, 1, "asc") + latest = _opt_int(newest[0].get("date")) if newest else None + earliest = _opt_int(oldest[0].get("date")) if oldest else None + return Coverage(earliest, latest, total) + + # ── movie-only cross-check (§4.11) ─────────────────────────────────── + + def movie_media_info(self, section_id: str) -> list[dict]: + """Per-item size/play data for a MOVIE section only. + + Never call this for a show section: it returns show-level rows with + file_size 0 whatever section_type is passed. + """ + out: list[dict] = [] + start = 0 + while True: + d = self.client.cmd( + "get_library_media_info", + section_id=section_id, + start=start, + length=self.client.page_size, + order_column="file_size", + order_dir="desc", + ) or {} + rows = d.get("data") or [] + if not rows: + break + out.extend(rows) + total = _int(d.get("recordsFiltered") or d.get("recordsTotal") or 0) + start += len(rows) + if start >= total: + break + return out + + +def _float_or_none(v): + try: + return float(v) + except (TypeError, ValueError): + return None diff --git a/mediashelf/queries.py b/mediashelf/queries.py new file mode 100644 index 0000000..5f21310 --- /dev/null +++ b/mediashelf/queries.py @@ -0,0 +1,358 @@ +"""The item query: filters + rules + live score + paging + aggregates. + +The score is computed in the SELECT rather than stored, because weights change on +every slider drag and storing it would mean rewriting thousands of rows per drag +(§6.4). At this row count the whole query runs in single-digit milliseconds. +""" + +from __future__ import annotations + +import json +import time + +from . import rules as rules_mod +from . import scoring + + +def _score_ctx(db, cfg, has_completion_data: bool) -> scoring.ScoreContext: + maxsize = db.scalar( + "SELECT MAX(size_bytes) FROM media_item WHERE kind IN ('movie','season')" + ) or 1 + s = cfg.score + return scoring.ScoreContext( + now=int(time.time()), + max_size_bytes=maxsize, + stale_horizon_days=s.stale_horizon_days, + age_horizon_days=s.age_horizon_days, + popular_at=s.popular_at, + rejected_at=s.rejected_at, + solitude_at=s.solitude_at, + grace_days=s.grace_days, + recent_days=s.recent_days, + has_completion_data=has_completion_data, + ) + + +def history_has_completion(db) -> bool: + """True when the active history source records percent_complete (§4.11).""" + row = db.one( + "SELECT source FROM history_coverage ORDER BY event_count DESC LIMIT 1" + ) + return bool(row and row["source"] == "tautulli") + + +BASE_COLUMNS = """ + i.id, i.kind, i.title, i.sort_title, i.year, i.guid, i.show_guid, + i.library_id, lib.title AS library_title, + i.size_bytes, i.added_at, i.updated_at, i.duration_ms, + i.episode_count, i.part_count, i.primary_path, i.resolution, i.video_codec, + i.watch_count, i.partial_count, i.abandoned_count, i.avg_percent_complete, + i.last_watched_at, i.last_touched_at, i.first_watched_at, + i.distinct_watcher_count, i.pre_history, i.kept, i.kept_via, i.kept_mark_id, + i.provider_view_count, i.status, i.parent_id, i.season_number, + parent.title AS show_title +""" + +FROM_CLAUSE = """ + FROM media_item i + JOIN library lib ON lib.id = i.library_id + LEFT JOIN media_item parent ON parent.id = i.parent_id + LEFT JOIN ( + SELECT guid, COUNT(*) AS dupe_count + FROM media_item WHERE kind = 'movie' AND guid IS NOT NULL AND status = 'present' + GROUP BY guid HAVING COUNT(*) > 1 + ) d ON d.guid = i.guid + CROSS JOIN (SELECT MAX(size_bytes) AS maxsize FROM media_item + WHERE kind IN ('movie','season')) s +""" + + +class Query: + def __init__(self, db, cfg): + self.db = db + self.cfg = cfg + self.has_cd = history_has_completion(db) + self.ctx = _score_ctx(db, cfg, self.has_cd) + + def build(self, *, library_ids=None, kinds=None, q=None, rule_group=None, + include_missing=False, include_kept=False, include_shows=False, + weights=None, extra_where=None): + where = [] + params: dict = {} + + if not include_missing: + where.append("i.status = 'present'") + if not include_kept: + where.append("i.kept = 0") + if not include_shows: + # Shows are containers; seasons are the unit of analysis for TV (§5.3) + where.append("i.kind != 'show'") + + if kinds: + keys = [] + for n, k in enumerate(kinds): + key = "k%d" % n + params[key] = k + keys.append(":" + key) + where.append("i.kind IN (%s)" % ", ".join(keys)) + + if library_ids: + keys = [] + for n, lid in enumerate(library_ids): + key = "lib%d" % n + params[key] = int(lid) + keys.append(":" + key) + where.append("i.library_id IN (%s)" % ", ".join(keys)) + + if q: + params["q"] = _fts_query(q) + where.append( + "i.id IN (SELECT rowid FROM media_item_fts WHERE media_item_fts MATCH :q)" + ) + + if rule_group: + frag, rp = rules_mod.compile_rules(rule_group, self.ctx.now) + if frag: + where.append(frag) + params.update(rp) + + if extra_where: + where.append(extra_where) + + expr, sp = scoring.sql_expression(self.ctx, weights) + params.update(sp) + params["now"] = self.ctx.now + + where_sql = " AND ".join(where) if where else "1=1" + return expr, where_sql, params + + def page(self, *, sort=None, page=1, page_size=100, **kw): + expr, where_sql, params = self.build(**kw) + order = rules_mod.compile_sort(sort) + + total = self.db.scalar( + "SELECT COUNT(*) " + FROM_CLAUSE + " WHERE " + where_sql, params + ) or 0 + agg = self.db.one( + "SELECT COALESCE(SUM(i.size_bytes),0) AS total_size, COUNT(*) AS n " + + FROM_CLAUSE + " WHERE " + where_sql, params + ) + + page = max(1, int(page)) + page_size = max(1, min(int(page_size), 500)) + params["_limit"] = page_size + params["_offset"] = (page - 1) * page_size + + sql = ( + "SELECT " + BASE_COLUMNS + ", " + expr + " AS reclaim_score, " + + scoring.GRACE_SQL + " AS grace, COALESCE(d.dupe_count, 1) AS duplicate_count " + + FROM_CLAUSE + " WHERE " + where_sql + + " ORDER BY " + order + " LIMIT :_limit OFFSET :_offset" + ) + rows = self.db.query(sql, params) + return { + "total": total, + "page": page, + "page_size": page_size, + "aggregate": { + "total_size_bytes": agg["total_size"] or 0, + "item_count": agg["n"] or 0, + }, + "items": [self.serialize(r, weights=kw.get("weights")) for r in rows], + } + + def iter_all(self, *, sort=None, **kw): + """Stream every matching row, for CSV export.""" + expr, where_sql, params = self.build(**kw) + order = rules_mod.compile_sort(sort) + sql = ( + "SELECT " + BASE_COLUMNS + ", " + expr + " AS reclaim_score, " + + scoring.GRACE_SQL + " AS grace, COALESCE(d.dupe_count, 1) AS duplicate_count " + + FROM_CLAUSE + " WHERE " + where_sql + " ORDER BY " + order + ) + for row in self.db.conn.execute(sql, params): + yield row + + def serialize(self, row, weights=None) -> dict: + d = dict(row) + comps = scoring.components(d, self.ctx) + flags = [] + if d.get("pre_history"): + flags.append("pre_history") + if (d.get("abandoned_count") or 0) >= self.ctx.rejected_at and not d.get("watch_count"): + flags.append("rejected") + if (d.get("duplicate_count") or 1) > 1: + flags.append("duplicate") + if (d.get("provider_view_count") or 0) > 0 and not d.get("watch_count"): + flags.append("history_gap") + if (d.get("part_count") or 0) > 1: + flags.append("multi_part") + + return { + "id": d["id"], + "kind": d["kind"], + "title": d["title"], + "show_title": d.get("show_title"), + "season_number": d.get("season_number"), + "year": d.get("year"), + "guid": d.get("guid"), + "library": {"id": d["library_id"], "title": d["library_title"]}, + "size_bytes": d.get("size_bytes") or 0, + "added_at": d.get("added_at"), + "last_watched_at": d.get("last_watched_at"), + "last_touched_at": d.get("last_touched_at"), + "watch_count": d.get("watch_count") or 0, + "partial_count": d.get("partial_count") or 0, + "abandoned_count": d.get("abandoned_count") or 0, + "avg_percent_complete": d.get("avg_percent_complete"), + "distinct_watcher_count": d.get("distinct_watcher_count") or 0, + "episode_count": d.get("episode_count") if d["kind"] == "season" else None, + "primary_path": d.get("primary_path"), + "part_count": d.get("part_count") or 0, + "resolution": d.get("resolution"), + "duplicate_count": d.get("duplicate_count") or 1, + "pre_history": bool(d.get("pre_history")), + "kept": bool(d.get("kept")), + "kept_via": d.get("kept_via"), + "status": d.get("status"), + "reclaim_score": d.get("reclaim_score"), + "reclaim_components": {k: (None if v is None else round(v, 4)) + for k, v in comps.items()}, + "grace": d.get("grace"), + "flags": flags, + } + + +def _fts_query(q: str) -> str: + """Turn user text into a safe FTS5 prefix query. + + FTS5 has its own operator syntax; passing raw user input straight through + lets a stray quote or NEAR() produce an error or a surprising match, so each + token is quoted and turned into a prefix term. + """ + tokens = [t for t in "".join(c if c.isalnum() else " " for c in q).split() if t] + if not tokens: + return '""' + return " ".join('"%s"*' % t for t in tokens[:10]) + + +# ── aggregate stats for the dashboard ──────────────────────────────────── + +def overview(db, cfg) -> dict: + def s(sql, params=()): + return db.scalar(sql, params) or 0 + + unit = "kind IN ('movie','season') AND status='present'" + total_bytes = s(f"SELECT SUM(size_bytes) FROM media_item WHERE {unit}") + total_items = s(f"SELECT COUNT(*) FROM media_item WHERE {unit}") + never_bytes = s(f"SELECT SUM(size_bytes) FROM media_item WHERE {unit} AND watch_count=0") + kept_bytes = s(f"SELECT SUM(size_bytes) FROM media_item WHERE {unit} AND kept=1") + # The three-way split that keeps the keep list honest (§6.6) + never_kept = s(f"SELECT SUM(size_bytes) FROM media_item WHERE {unit} AND watch_count=0 AND kept=1") + confident = s(f"SELECT SUM(size_bytes) FROM media_item WHERE {unit} " + "AND watch_count=0 AND kept=0 AND pre_history=0") + uncertain = s(f"SELECT SUM(size_bytes) FROM media_item WHERE {unit} " + "AND watch_count=0 AND kept=0 AND pre_history=1") + cold = s(f"SELECT SUM(size_bytes) FROM media_item WHERE {unit} AND kept=0 AND " + "(last_watched_at IS NULL OR last_watched_at < strftime('%s','now') - 2*365*86400)") + + cov = db.one("SELECT * FROM history_coverage ORDER BY event_count DESC LIMIT 1") + last_scan = db.one("SELECT * FROM scan WHERE status='succeeded' ORDER BY id DESC LIMIT 1") + + return { + "total_bytes": total_bytes, + "total_items": total_items, + "never_played_bytes": never_bytes, + "kept_bytes": kept_bytes, + "never_played_kept_bytes": never_kept, + "available_bytes": max(never_bytes - never_kept, 0), + "confident_bytes": confident, + "uncertain_bytes": uncertain, + "cold_bytes": cold, + "libraries": s("SELECT COUNT(*) FROM library"), + "episodes": s("SELECT COUNT(*) FROM episode WHERE status='present'"), + "watch_events": s("SELECT COUNT(*) FROM watch_event"), + "accounts": s("SELECT COUNT(*) FROM account"), + "history_source": cov["source"] if cov else None, + "history_since": cov["earliest_event_at"] if cov else None, + "history_until": cov["latest_event_at"] if cov else None, + "has_completion_data": history_has_completion(db), + "last_scan_at": last_scan["finished_at"] if last_scan else None, + "keep_marks": s("SELECT COUNT(*) FROM keep_mark"), + } + + +def size_by_library(db) -> list[dict]: + return [dict(r) for r in db.query(""" + SELECT lib.id, lib.title, lib.kind, lib.keep_all, + COUNT(*) AS items, + COALESCE(SUM(i.size_bytes),0) AS size_bytes, + COALESCE(SUM(CASE WHEN i.watch_count=0 THEN i.size_bytes ELSE 0 END),0) AS never_bytes, + COALESCE(SUM(CASE WHEN i.kept=1 THEN i.size_bytes ELSE 0 END),0) AS kept_bytes + FROM media_item i JOIN library lib ON lib.id = i.library_id + WHERE i.kind IN ('movie','season') AND i.status='present' + GROUP BY lib.id ORDER BY size_bytes DESC + """)] + + +def added_over_time(db, bucket="month") -> list[dict]: + fmt = "%Y-%m" if bucket == "month" else "%Y" + return [dict(r) for r in db.query(f""" + SELECT strftime('{fmt}', added_at, 'unixepoch') AS period, + COUNT(*) AS items, COALESCE(SUM(size_bytes),0) AS size_bytes + FROM media_item + WHERE kind IN ('movie','season') AND status='present' + AND added_at IS NOT NULL AND added_at > 0 + GROUP BY period ORDER BY period + """)] + + +def completion_split(db) -> dict: + """finished / abandoned / never-opened by size — invisible without Tautulli.""" + row = db.one(""" + SELECT + COALESCE(SUM(CASE WHEN watch_count > 0 THEN size_bytes ELSE 0 END),0) AS finished, + COALESCE(SUM(CASE WHEN watch_count = 0 AND (abandoned_count > 0 OR partial_count > 0) + THEN size_bytes ELSE 0 END),0) AS started, + COALESCE(SUM(CASE WHEN watch_count = 0 AND abandoned_count = 0 AND partial_count = 0 + THEN size_bytes ELSE 0 END),0) AS never + FROM media_item WHERE kind IN ('movie','season') AND status='present' + """) + return dict(row) + + +def size_vs_lastwatched(db, limit=3000) -> list[dict]: + return [dict(r) for r in db.query(""" + SELECT id, title, kind, size_bytes, last_watched_at, added_at, kept, + watch_count, pre_history + FROM media_item + WHERE kind IN ('movie','season') AND status='present' AND size_bytes > 0 + ORDER BY size_bytes DESC LIMIT ? + """, (limit,))] + + +def duplicate_groups(db) -> list[dict]: + """Same content held more than once — usually Movies vs 4K Movies (§6.5).""" + rows = db.query(""" + SELECT i.guid, i.id, i.title, i.year, i.size_bytes, i.resolution, + i.watch_count, i.last_watched_at, i.kept, lib.title AS library_title + FROM media_item i JOIN library lib ON lib.id = i.library_id + WHERE i.kind='movie' AND i.status='present' AND i.guid IS NOT NULL + AND i.guid IN (SELECT guid FROM media_item WHERE kind='movie' + AND status='present' AND guid IS NOT NULL + GROUP BY guid HAVING COUNT(*) > 1) + ORDER BY i.guid, i.size_bytes DESC + """) + groups: dict[str, dict] = {} + for r in rows: + g = groups.setdefault(r["guid"], {"guid": r["guid"], "title": r["title"], + "year": r["year"], "copies": []}) + g["copies"].append(dict(r)) + out = [] + for g in groups.values(): + sizes = [c["size_bytes"] or 0 for c in g["copies"]] + g["total_bytes"] = sum(sizes) + g["redundant_bytes"] = sum(sizes) - max(sizes) if sizes else 0 + out.append(g) + out.sort(key=lambda g: -g["redundant_bytes"]) + return out diff --git a/mediashelf/rules.py b/mediashelf/rules.py new file mode 100644 index 0000000..aaf90e9 --- /dev/null +++ b/mediashelf/rules.py @@ -0,0 +1,224 @@ +"""Saved-view rule grammar → parameterized SQL (§7.1). + +Everything here is whitelist-driven. Field names map to columns through a dict; +operators come from a fixed set; values are always bound parameters. There is no +string interpolation of user input anywhere in this module, and anything outside +the whitelist is a RuleError (surfaced as HTTP 400), never a best-effort guess. +""" + +from __future__ import annotations + +from typing import Any + +DAY = 86400 + + +class RuleError(ValueError): + """Malformed or non-whitelisted rule. Always a 400, never a 500.""" + + +# field name -> (sql column, type) +FIELDS: dict[str, tuple[str, str]] = { + "library_id": ("i.library_id", "int"), + "kind": ("i.kind", "str"), + "title": ("i.title", "str"), + "year": ("i.year", "int"), + "size_bytes": ("i.size_bytes", "int"), + "added_at": ("i.added_at", "ts"), + "updated_at": ("i.updated_at", "ts"), + "episode_count": ("i.episode_count", "int"), + "part_count": ("i.part_count", "int"), + "duration_ms": ("i.duration_ms", "int"), + "primary_path": ("i.primary_path", "str"), + "resolution": ("i.resolution", "str"), + "video_codec": ("i.video_codec", "str"), + "watch_count": ("i.watch_count", "int"), + "partial_count": ("i.partial_count", "int"), + "abandoned_count": ("i.abandoned_count", "int"), + "avg_percent_complete": ("i.avg_percent_complete", "float"), + "last_watched_at": ("i.last_watched_at", "ts"), + "last_touched_at": ("i.last_touched_at", "ts"), + "first_watched_at": ("i.first_watched_at", "ts"), + "distinct_watcher_count": ("i.distinct_watcher_count", "int"), + "pre_history": ("i.pre_history", "bool"), + "kept": ("i.kept", "bool"), + "kept_via": ("i.kept_via", "str"), + "status": ("i.status", "str"), + "guid": ("i.guid", "str"), + # derived, provided by the query builder as a correlated expression + "watch_ratio": ("(CASE WHEN i.episode_count > 0 " + "THEN CAST(i.watch_count AS REAL) / i.episode_count " + "ELSE CAST(i.watch_count AS REAL) END)", "float"), + "duplicate_count": ("COALESCE(d.dupe_count, 1)", "int"), +} + +SIMPLE_OPS = { + "eq": "=", "ne": "!=", "lt": "<", "lte": "<=", "gt": ">", "gte": ">=", +} +LIST_OPS = {"in": "IN", "not_in": "NOT IN"} +LIKE_OPS = {"contains": "%{}%", "starts_with": "{}%", "ends_with": "%{}"} +NULL_OPS = {"is_null": "IS NULL", "is_not_null": "IS NOT NULL"} +REL_OPS = {"older_than_days", "newer_than_days"} +SPECIAL_OPS = {"never"} + +ALL_OPS = (set(SIMPLE_OPS) | set(LIST_OPS) | set(LIKE_OPS) + | set(NULL_OPS) | REL_OPS | SPECIAL_OPS) + + +class _Builder: + def __init__(self, now: int): + self.now = now + self.params: dict[str, Any] = {} + self._n = 0 + + def bind(self, value) -> str: + self._n += 1 + key = "r%d" % self._n + self.params[key] = value + return ":" + key + + def coerce(self, value, ftype: str): + if ftype == "int": + try: + return int(value) + except (TypeError, ValueError): + raise RuleError("expected an integer, got %r" % (value,)) + if ftype == "float": + try: + return float(value) + except (TypeError, ValueError): + raise RuleError("expected a number, got %r" % (value,)) + if ftype == "bool": + if isinstance(value, bool): + return 1 if value else 0 + if str(value).lower() in ("1", "true", "yes"): + return 1 + if str(value).lower() in ("0", "false", "no"): + return 0 + raise RuleError("expected a boolean, got %r" % (value,)) + if ftype == "ts": + try: + return int(value) + except (TypeError, ValueError): + raise RuleError("expected a timestamp, got %r" % (value,)) + return str(value) + + def group(self, node: dict, depth: int = 0) -> str: + if depth > 8: + raise RuleError("rule nesting too deep") + if not isinstance(node, dict): + raise RuleError("rule node must be an object") + + if "rules" in node: + op = str(node.get("op", "and")).lower() + if op not in ("and", "or"): + raise RuleError("group operator must be 'and' or 'or'") + rules = node.get("rules") or [] + if not isinstance(rules, list): + raise RuleError("'rules' must be a list") + parts = [self.group(r, depth + 1) for r in rules] + parts = [p for p in parts if p] + if not parts: + return "" + joiner = " AND " if op == "and" else " OR " + return "(" + joiner.join(parts) + ")" + + return self.condition(node) + + def condition(self, node: dict) -> str: + field = node.get("field") + op = str(node.get("op", "")).lower() + if field not in FIELDS: + raise RuleError("unknown field %r" % (field,)) + if op not in ALL_OPS: + raise RuleError("unknown operator %r" % (op,)) + + col, ftype = FIELDS[field] + value = node.get("value") + + if op in NULL_OPS: + return "%s %s" % (col, NULL_OPS[op]) + + if op == "never": + # "never watched" is null-or-zero, which is not the same as IS NULL + return "(%s IS NULL OR %s = 0)" % (col, col) + + if op in REL_OPS: + days = self.coerce(value, "int") + cutoff = self.bind(self.now - days * DAY) + if op == "older_than_days": + return "(%s IS NOT NULL AND %s > 0 AND %s < %s)" % (col, col, col, cutoff) + return "(%s IS NOT NULL AND %s >= %s)" % (col, col, cutoff) + + if op in LIST_OPS: + if not isinstance(value, (list, tuple)) or not value: + raise RuleError("operator %r needs a non-empty list" % op) + if len(value) > 500: + raise RuleError("list too long") + placeholders = ", ".join(self.bind(self.coerce(v, ftype)) for v in value) + return "%s %s (%s)" % (col, LIST_OPS[op], placeholders) + + if op in LIKE_OPS: + if ftype != "str": + raise RuleError("operator %r only applies to text fields" % op) + pattern = LIKE_OPS[op].format(_escape_like(str(value))) + return "%s LIKE %s ESCAPE '\\'" % (col, self.bind(pattern)) + + return "%s %s %s" % (col, SIMPLE_OPS[op], self.bind(self.coerce(value, ftype))) + + +def _escape_like(s: str) -> str: + return s.replace("\\", "\\\\").replace("%", "\\%").replace("_", "\\_") + + +def compile_rules(rules: dict | None, now: int) -> tuple[str, dict]: + """Compile a rule group to (sql_fragment, params). Empty rules -> ('', {}).""" + if not rules: + return "", {} + b = _Builder(now) + sql = b.group(rules) + return sql, b.params + + +SORTABLE = { + "title": "i.sort_title, i.title", + "year": "i.year", + "size_bytes": "i.size_bytes", + "added_at": "i.added_at", + "updated_at": "i.updated_at", + "last_watched_at": "i.last_watched_at", + "last_touched_at": "i.last_touched_at", + "watch_count": "i.watch_count", + "partial_count": "i.partial_count", + "abandoned_count": "i.abandoned_count", + "distinct_watcher_count": "i.distinct_watcher_count", + "episode_count": "i.episode_count", + "avg_percent_complete": "i.avg_percent_complete", + "reclaim_score": "reclaim_score", + "library": "lib.title", + "kind": "i.kind", +} + + +def compile_sort(spec: str | None) -> str: + """'reclaim_score:desc,size_bytes:desc' -> ORDER BY clause. Whitelisted.""" + if not spec: + return "reclaim_score DESC, i.size_bytes DESC" + out = [] + for part in str(spec).split(","): + part = part.strip() + if not part: + continue + if ":" in part: + name, direction = part.split(":", 1) + else: + name, direction = part, "asc" + name = name.strip() + direction = "DESC" if direction.strip().lower() == "desc" else "ASC" + if name not in SORTABLE: + raise RuleError("cannot sort by %r" % name) + out.append("%s %s" % (SORTABLE[name], direction)) + if not out: + return "reclaim_score DESC, i.size_bytes DESC" + out.append("i.id ASC") # stable tiebreak so paging never repeats a row + return ", ".join(out) diff --git a/mediashelf/scanner.py b/mediashelf/scanner.py new file mode 100644 index 0000000..9499ab6 --- /dev/null +++ b/mediashelf/scanner.py @@ -0,0 +1,152 @@ +"""Background scan runner and the scheduler. + +The scheduler runs in-process. With gunicorn --workers 2 both workers would +start one and two scans would race for the SQLite write lock, so the scheduler +only starts in the worker that wins an exclusive flock on /data. The scan_lock +table is the second line of defence; both are needed, since either alone leaves +a race window (§11.4). +""" + +from __future__ import annotations + +import json +import logging +import os +import threading +import time + +log = logging.getLogger(__name__) + +_scan_thread: threading.Thread | None = None +_scan_lock = threading.Lock() +_scheduler = None +_flock_handle = None + + +def _run_scan(app, mode: str, trigger: str) -> None: + from . import ingest, keeps, providers + + with app.app_context(): + ext = app.extensions["mediashelf"] + cfg = ext["config"] + db = ext["db"] + try: + media = providers.build_media_provider(cfg) + history, degraded = providers.build_history_provider(cfg, media) + if degraded: + log.warning("history running degraded: %s", degraded) + ext["degraded_reason"] = degraded + else: + ext["degraded_reason"] = None + result = ingest.Ingest(db, cfg, media, history).run(mode, trigger) + log.info("scan %s %s: seen=%d added=%d events=%d", + result.scan_id, result.status, result.items_seen, + result.items_added, result.events_added) + if result.status == "succeeded": + _export_backups(db, cfg) + except Exception: # noqa: BLE001 + log.exception("background scan failed") + finally: + db.close() + + +def _export_backups(db, cfg) -> None: + """Keeps and saved views are the only data not reconstructible (§11.5).""" + from . import keeps + + try: + data_dir = os.path.dirname(os.path.abspath(cfg.database_path)) or "." + keeps.write_export_file(db, os.path.join(data_dir, "keeps.json")) + views = [dict(r) for r in db.query("SELECT * FROM saved_view WHERE builtin = 0")] + with open(os.path.join(data_dir, "views.json"), "w") as fh: + json.dump({"version": 1, "exported_at": int(time.time()), "views": views}, + fh, indent=2) + except OSError as e: + log.warning("could not write keep/view export: %s", e) + + +def start_background_scan(app, mode: str = "incremental", trigger: str = "manual") -> bool: + """Returns False if a scan is already running in this process.""" + global _scan_thread + with _scan_lock: + if _scan_thread is not None and _scan_thread.is_alive(): + return False + real_app = app._get_current_object() if hasattr(app, "_get_current_object") else app + _scan_thread = threading.Thread( + target=_run_scan, args=(real_app, mode, trigger), + name="mediashelf-scan", daemon=True) + _scan_thread.start() + return True + + +def scan_running() -> bool: + return _scan_thread is not None and _scan_thread.is_alive() + + +# ── scheduler ──────────────────────────────────────────────────────────── + +def _acquire_process_lock(data_dir: str) -> bool: + """Exclusive flock, so only one gunicorn worker schedules anything.""" + global _flock_handle + import fcntl + + try: + os.makedirs(data_dir, exist_ok=True) + fh = open(os.path.join(data_dir, ".scheduler.lock"), "w") + fcntl.flock(fh.fileno(), fcntl.LOCK_EX | fcntl.LOCK_NB) + fh.write(str(os.getpid())) + fh.flush() + _flock_handle = fh # held for the life of the process + return True + except (OSError, BlockingIOError): + return False + + +def _cron_kwargs(expr: str) -> dict: + parts = (expr or "").split() + if len(parts) != 5: + raise ValueError("cron expression must have 5 fields, got %r" % expr) + minute, hour, dom, month, dow = parts + return {"minute": minute, "hour": hour, "day": dom, + "month": month, "day_of_week": dow} + + +def start_scheduler(app) -> bool: + """Start the nightly jobs in exactly one process. Returns True if started.""" + global _scheduler + cfg = app.extensions["mediashelf"]["config"] + if not cfg.scheduler_enabled: + log.info("scheduler disabled by config") + return False + + data_dir = os.path.dirname(os.path.abspath(cfg.database_path)) or "." + if not _acquire_process_lock(data_dir): + log.info("another worker holds the scheduler lock; not scheduling here") + return False + + from apscheduler.schedulers.background import BackgroundScheduler + from apscheduler.triggers.cron import CronTrigger + + real_app = app._get_current_object() if hasattr(app, "_get_current_object") else app + _scheduler = BackgroundScheduler(timezone=cfg.tz) + + try: + _scheduler.add_job( + lambda: start_background_scan(real_app, "incremental", "schedule"), + CronTrigger(**_cron_kwargs(cfg.scan_schedule_cron), timezone=cfg.tz), + id="incremental", replace_existing=True, max_instances=1) + _scheduler.add_job( + lambda: start_background_scan(real_app, "full", "schedule"), + CronTrigger(**_cron_kwargs(cfg.scan_full_sweep_cron), timezone=cfg.tz), + id="full-sweep", replace_existing=True, max_instances=1) + except ValueError as e: + log.error("bad cron configuration, scheduler not started: %s", e) + return False + + _scheduler.start() + log.info("scheduler started (incremental %r, full sweep %r, tz %s)", + cfg.scan_schedule_cron, cfg.scan_full_sweep_cron, cfg.tz) + + if cfg.scan_on_startup: + start_background_scan(real_app, "full", "startup") + return True diff --git a/mediashelf/scoring.py b/mediashelf/scoring.py new file mode 100644 index 0000000..07df676 --- /dev/null +++ b/mediashelf/scoring.py @@ -0,0 +1,295 @@ +"""The reclaim score (§6). + +Implemented twice on purpose: once as a SQL expression, because weights change +on every slider drag and storing the score would mean recomputing 5,700 rows per +drag; and once in Python, for CSV export and for tests. Two implementations of +one formula is a real risk, so test_scoring.py asserts they agree to within 0.01 +on a generated corpus. + +Components are each normalized to [0, 1] where 1 means "better deletion +candidate". The weighted mean uses only the components AVAILABLE for a row, so a +missing component renormalizes instead of dragging every score down. +""" + +from __future__ import annotations + +import math +from dataclasses import dataclass + +DAY = 86400.0 + +COMPONENTS = ("size", "staleness", "unpopularity", "solitude", "age", "rejection") + +DEFAULT_WEIGHTS = { + "size": 0.28, + "staleness": 0.24, + "unpopularity": 0.22, + "solitude": 0.10, + "age": 0.10, + "rejection": 0.06, +} + + +@dataclass +class ScoreContext: + """Per-request constants. max_size is computed once in a CTE, not per row.""" + + now: int + max_size_bytes: int + stale_horizon_days: int = 730 + age_horizon_days: int = 1095 + popular_at: int = 3 + rejected_at: int = 2 + solitude_at: int = 3 + grace_days: int = 30 + recent_days: int = 90 + has_completion_data: bool = True + + +def normalize_weights(weights: dict | None, *, has_completion_data: bool = True) -> dict: + w = dict(DEFAULT_WEIGHTS) + if weights: + for k, v in weights.items(): + if k in DEFAULT_WEIGHTS: + try: + w[k] = max(0.0, float(v)) + except (TypeError, ValueError): + pass + if not has_completion_data: + # No percent_complete anywhere => rejection is not merely zero, it is + # unavailable, and must not dilute the remaining components (§4.11). + w["rejection"] = 0.0 + if sum(w.values()) <= 0: + w = dict(DEFAULT_WEIGHTS) + return w + + +# ─────────────────────────────── python ────────────────────────────────── + + +def components(row: dict, ctx: ScoreContext) -> dict[str, float | None]: + """Component values for one row. None means 'not available for this row'.""" + size = int(row.get("size_bytes") or 0) + added_at = row.get("added_at") + last_watched = row.get("last_watched_at") + watch_count = int(row.get("watch_count") or 0) + abandoned = int(row.get("abandoned_count") or 0) + watchers = int(row.get("distinct_watcher_count") or 0) + episodes = int(row.get("episode_count") or 0) + kind = row.get("kind") or "movie" + pre_history = bool(row.get("pre_history")) + + # size — log scale: 2GB vs 4GB matters more than 60GB vs 62GB + max_size = max(int(ctx.max_size_bytes or 0), 1) + c_size = math.log10(1 + size) / math.log10(1 + max_size) if size > 0 else 0.0 + c_size = min(max(c_size, 0.0), 1.0) + + # staleness — never watched scores 1.0 + if last_watched: + days = max(0.0, (ctx.now - last_watched) / DAY) + c_stale = min(days / max(ctx.stale_horizon_days, 1), 1.0) + else: + c_stale = 1.0 + if pre_history and not last_watched: + # Might have been watched before history coverage began; capping keeps a + # 2009 film that was watched in 2015 from scoring as never-watched (§4.11) + c_stale = min(c_stale, 0.75) + + # unpopularity — TV normalized per-episode so it compares to a movie + if kind == "season" and episodes > 0: + normalized_watches = watch_count / episodes + else: + normalized_watches = float(watch_count) + c_unpop = 1.0 - min(normalized_watches / max(ctx.popular_at, 1), 1.0) + + # age + if added_at: + days = max(0.0, (ctx.now - added_at) / DAY) + c_age = min(days / max(ctx.age_horizon_days, 1), 1.0) + else: + c_age = 0.0 + + # solitude + c_sol = 1.0 - min(watchers / max(ctx.solitude_at, 1), 1.0) + + # rejection — zeroed the moment anyone completes a view + if not ctx.has_completion_data: + c_rej = None + elif watch_count > 0: + c_rej = 0.0 + else: + c_rej = min(abandoned / max(ctx.rejected_at, 1), 1.0) + + return { + "size": c_size, + "staleness": c_stale, + "unpopularity": c_unpop, + "solitude": c_sol, + "age": c_age, + "rejection": c_rej, + } + + +def round_half_up(x: float, places: int = 2) -> float: + """Match SQLite's ROUND(), which rounds half away from zero. + + Python's built-in round() uses banker's rounding, so the two disagree at + exactly x.xx5 — the SQL grid would show 53.13 where the CSV export showed + 53.12. Harmless, but the kind of inconsistency that costs an afternoon when + someone notices the two disagree and assumes the formula differs. + """ + factor = 10 ** places + return math.floor(abs(x) * factor + 0.5) / factor * (1 if x >= 0 else -1) + + +def score_row(row: dict, ctx: ScoreContext, weights: dict | None = None) -> dict: + """Returns {score, score_raw, components, grace}. + + score_raw is unrounded — tests compare it against the SQL expression so a + real formula divergence is not masked by rounding, and vice versa. + """ + w = normalize_weights(weights, has_completion_data=ctx.has_completion_data) + comps = components(row, ctx) + + num = 0.0 + den = 0.0 + for name in COMPONENTS: + value = comps.get(name) + weight = w.get(name, 0.0) + if value is None or weight <= 0: + continue + num += weight * value + den += weight + score = 100.0 * (num / den) if den > 0 else 0.0 + + grace = None + now = ctx.now + added_at = row.get("added_at") + last_watched = row.get("last_watched_at") + if added_at and (now - added_at) < ctx.grace_days * DAY: + score = 0.0 + grace = "new" + elif last_watched and (now - last_watched) < ctx.recent_days * DAY: + score = min(score, 25.0) + grace = "recent" + + return { + "score": round_half_up(score, 2), + "score_raw": score, + "components": {k: (None if v is None else round(v, 4)) for k, v in comps.items()}, + "grace": grace, + } + + +# ──────────────────────────────── sql ──────────────────────────────────── + + +def sql_expression(ctx: ScoreContext, weights: dict | None = None, + *, rounded: bool = True) -> tuple[str, dict]: + """Return (expression, params) computing the score for media_item rows. + + Expects a CTE or join providing `maxsize` as the library-wide max size. + Mirrors components()/score_row() exactly — see test_scoring.py. Pass + rounded=False to compare the raw formula without rounding in the way. + """ + w = normalize_weights(weights, has_completion_data=ctx.has_completion_data) + + p = { + "now": ctx.now, + "stale_h": max(ctx.stale_horizon_days, 1), + "age_h": max(ctx.age_horizon_days, 1), + "popular_at": max(ctx.popular_at, 1), + "rejected_at": max(ctx.rejected_at, 1), + "solitude_at": max(ctx.solitude_at, 1), + "grace_s": ctx.grace_days * 86400, + "recent_s": ctx.recent_days * 86400, + "w_size": w["size"], + "w_stale": w["staleness"], + "w_unpop": w["unpopularity"], + "w_sol": w["solitude"], + "w_age": w["age"], + "w_rej": w["rejection"], + } + + c_size = ( + "MIN(MAX(CASE WHEN i.size_bytes > 0 THEN " + " (LOG(1 + i.size_bytes) / LOG(1 + MAX(s.maxsize, 1))) ELSE 0.0 END, 0.0), 1.0)" + ) + + c_stale_raw = ( + "CASE WHEN i.last_watched_at IS NOT NULL AND i.last_watched_at > 0" + " THEN MIN(MAX(CAST(:now - i.last_watched_at AS REAL) / 86400.0, 0.0)" + " / :stale_h, 1.0)" + " ELSE 1.0 END" + ) + # pre_history cap applies only when there is no recorded watch at all + c_stale = ( + "CASE WHEN i.pre_history = 1 AND (i.last_watched_at IS NULL OR i.last_watched_at = 0)" + " THEN MIN(%s, 0.75) ELSE %s END" % (c_stale_raw, c_stale_raw) + ) + + normalized_watches = ( + "CASE WHEN i.kind = 'season' AND i.episode_count > 0" + " THEN CAST(i.watch_count AS REAL) / i.episode_count" + " ELSE CAST(i.watch_count AS REAL) END" + ) + c_unpop = "(1.0 - MIN((%s) / :popular_at, 1.0))" % normalized_watches + + c_age = ( + "CASE WHEN i.added_at IS NOT NULL AND i.added_at > 0" + " THEN MIN(MAX(CAST(:now - i.added_at AS REAL) / 86400.0, 0.0) / :age_h, 1.0)" + " ELSE 0.0 END" + ) + + c_sol = "(1.0 - MIN(CAST(i.distinct_watcher_count AS REAL) / :solitude_at, 1.0))" + + if ctx.has_completion_data: + c_rej = ( + "CASE WHEN i.watch_count > 0 THEN 0.0" + " ELSE MIN(CAST(i.abandoned_count AS REAL) / :rejected_at, 1.0) END" + ) + rej_num = ":w_rej * (%s)" % c_rej + rej_den = ":w_rej" + else: + rej_num = "0.0" + rej_den = "0.0" + + numerator = ( + f":w_size * ({c_size}) + :w_stale * ({c_stale}) + :w_unpop * ({c_unpop})" + f" + :w_sol * ({c_sol}) + :w_age * ({c_age}) + {rej_num}" + ) + denominator = f"(:w_size + :w_stale + :w_unpop + :w_sol + :w_age + {rej_den})" + + raw = f"CASE WHEN {denominator} > 0 THEN 100.0 * ({numerator}) / {denominator} ELSE 0.0 END" + + graced = ( + "CASE" + " WHEN i.added_at IS NOT NULL AND i.added_at > 0" + " AND (:now - i.added_at) < :grace_s THEN 0.0" + " WHEN i.last_watched_at IS NOT NULL AND i.last_watched_at > 0" + " AND (:now - i.last_watched_at) < :recent_s THEN MIN(%s, 25.0)" + " ELSE %s END" % (raw, raw) + ) + return ("ROUND(%s, 2)" % graced if rounded else graced), p + + +GRACE_SQL = ( + "CASE" + " WHEN i.added_at IS NOT NULL AND i.added_at > 0" + " AND (:now - i.added_at) < :grace_s THEN 'new'" + " WHEN i.last_watched_at IS NOT NULL AND i.last_watched_at > 0" + " AND (:now - i.last_watched_at) < :recent_s THEN 'recent'" + " ELSE NULL END" +) + + +def register_sqlite_functions(conn) -> None: + """SQLite has no LOG() by default; add it (and a MIN/MAX-safe guard).""" + def _log10(x): + try: + x = float(x) + except (TypeError, ValueError): + return None + return math.log10(x) if x > 0 else 0.0 + + conn.create_function("LOG", 1, _log10, deterministic=True) diff --git a/mediashelf/static/app.css b/mediashelf/static/app.css new file mode 100644 index 0000000..7dc7c1d --- /dev/null +++ b/mediashelf/static/app.css @@ -0,0 +1,194 @@ +:root { + --bg: #14161a; + --panel: #1b1e24; + --panel-2: #22262e; + --line: #2e333c; + --text: #e6e9ef; + --dim: #99a1b0; + --faint: #6b7280; + --accent: #6ea8fe; + --warm: #f0b357; + --hot: #e8705f; + --good: #5dc98a; + --kept: #b48ce8; + --radius: 8px; +} + +@media (prefers-color-scheme: light) { + :root { + --bg: #f6f7f9; --panel: #ffffff; --panel-2: #f0f2f5; --line: #dfe3e9; + --text: #1a1d23; --dim: #5b6472; --faint: #8b93a1; --accent: #2f6fd0; + } +} + +* { box-sizing: border-box; } +html, body { margin: 0; padding: 0; } +body { + background: var(--bg); color: var(--text); + font: 14px/1.5 ui-sans-serif, system-ui, -apple-system, "Segoe UI", Roboto, sans-serif; +} +h2, h3 { margin: 0 0 .5rem; font-weight: 600; } +h3 { font-size: .95rem; } +a { color: var(--accent); } + +/* ── topbar ─────────────────────────────────────────────────────── */ +.topbar { + display: flex; align-items: center; gap: 1.5rem; + padding: .6rem 1rem; background: var(--panel); + border-bottom: 1px solid var(--line); position: sticky; top: 0; z-index: 20; +} +.brand { display: flex; align-items: center; gap: .5rem; font-weight: 650; } +.logo { color: var(--accent); font-size: 1.2rem; } +.readonly-badge { + font-size: .68rem; font-weight: 600; letter-spacing: .04em; text-transform: uppercase; + color: var(--warm); border: 1px solid var(--warm); border-radius: 999px; + padding: .1rem .5rem; opacity: .9; +} +.tabs { display: flex; gap: .15rem; flex: 1; } +.tab { + background: none; border: 0; color: var(--dim); padding: .4rem .75rem; + border-radius: var(--radius); cursor: pointer; font: inherit; +} +.tab:hover { background: var(--panel-2); color: var(--text); } +.tab.active { background: var(--panel-2); color: var(--text); font-weight: 600; } +.topbar-right { display: flex; align-items: center; gap: .6rem; } +.source-badge { font-size: .78rem; color: var(--dim); } + +/* ── buttons ────────────────────────────────────────────────────── */ +.btn { + background: var(--panel-2); color: var(--text); border: 1px solid var(--line); + border-radius: var(--radius); padding: .38rem .75rem; cursor: pointer; font: inherit; +} +.btn:hover:not(:disabled) { border-color: var(--accent); } +.btn:disabled { opacity: .4; cursor: not-allowed; } +.btn.small { padding: .22rem .5rem; font-size: .82rem; } +.btn.ghost { background: none; } +.btn.danger:not(:disabled) { color: var(--hot); border-color: var(--hot); } + +/* ── layout ─────────────────────────────────────────────────────── */ +main { padding: 1rem; max-width: 1800px; margin: 0 auto; } +.view { display: none; } +.view.active { display: block; } +.card { + background: var(--panel); border: 1px solid var(--line); + border-radius: var(--radius); padding: .9rem; margin-bottom: 1rem; +} +.hint { color: var(--dim); font-size: .82rem; margin: .1rem 0 .7rem; } + +/* ── dashboard ──────────────────────────────────────────────────── */ +.split-hero { + display: grid; grid-template-columns: repeat(3, 1fr); gap: 1px; + background: var(--line); border: 1px solid var(--line); + border-radius: var(--radius); overflow: hidden; margin-bottom: 1rem; +} +.split-cell { background: var(--panel); padding: 1rem 1.1rem; } +.split-cell .label { + font-size: .7rem; letter-spacing: .07em; text-transform: uppercase; color: var(--faint); +} +.split-cell .value { font-size: 1.9rem; font-weight: 650; margin-top: .2rem; } +.split-cell .sub { font-size: .78rem; color: var(--dim); } +.split-cell.kept .value { color: var(--kept); } +.split-cell.available .value { color: var(--good); } + +.tiles { display: grid; grid-template-columns: repeat(auto-fit, minmax(160px, 1fr)); gap: .6rem; margin-bottom: 1rem; } +.tile { background: var(--panel); border: 1px solid var(--line); border-radius: var(--radius); padding: .7rem .8rem; } +.tile .label { font-size: .7rem; letter-spacing: .06em; text-transform: uppercase; color: var(--faint); } +.tile .value { font-size: 1.2rem; font-weight: 600; margin-top: .15rem; } +.tile .sub { font-size: .75rem; color: var(--dim); } + +.chart-grid { display: grid; grid-template-columns: 1fr 1fr; gap: 1rem; } +.card.wide { grid-column: 1 / -1; } +.chart { min-height: 40px; } + +.bar-row { display: grid; grid-template-columns: 150px 1fr 90px; gap: .5rem; align-items: center; margin-bottom: .3rem; font-size: .82rem; } +.bar-track { background: var(--panel-2); border-radius: 3px; height: 15px; overflow: hidden; display: flex; } +.bar-fill { height: 100%; background: var(--accent); } +.bar-fill.never { background: var(--warm); } +.bar-fill.kept { background: var(--kept); } +.bar-num { text-align: right; color: var(--dim); font-variant-numeric: tabular-nums; } +.bar-label { overflow: hidden; text-overflow: ellipsis; white-space: nowrap; } + +.stack { display: flex; height: 30px; border-radius: 4px; overflow: hidden; margin-bottom: .5rem; } +.stack > div { display: flex; align-items: center; justify-content: center; font-size: .72rem; color: #10131a; font-weight: 600; } +.legend { display: flex; gap: 1rem; font-size: .8rem; color: var(--dim); flex-wrap: wrap; } +.legend i { display: inline-block; width: 10px; height: 10px; border-radius: 2px; margin-right: .3rem; } + +/* ── grid ───────────────────────────────────────────────────────── */ +.grid-layout { display: grid; grid-template-columns: 250px 1fr; gap: 1rem; align-items: start; } +.filter-rail { + background: var(--panel); border: 1px solid var(--line); border-radius: var(--radius); + padding: .8rem; position: sticky; top: 60px; max-height: calc(100vh - 76px); overflow: auto; +} +.rail-section { margin-bottom: .9rem; } +.rail-label { display: block; font-size: .7rem; letter-spacing: .06em; text-transform: uppercase; color: var(--faint); margin-bottom: .3rem; } +.filter-rail input[type=search], .filter-rail select { + width: 100%; background: var(--panel-2); color: var(--text); + border: 1px solid var(--line); border-radius: 6px; padding: .32rem .45rem; font: inherit; +} +.checklist { max-height: 190px; overflow: auto; font-size: .84rem; } +.checklist label, .switch { display: flex; align-items: center; gap: .4rem; padding: .1rem 0; cursor: pointer; } +.weights label { display: block; font-size: .78rem; color: var(--dim); margin-top: .35rem; } +.weights input[type=range] { width: 100%; } + +.grid-main { min-width: 0; } +.grid-toolbar { display: flex; justify-content: space-between; align-items: center; margin-bottom: .5rem; gap: 1rem; flex-wrap: wrap; } +.grid-summary { font-size: .85rem; color: var(--dim); } +.grid-summary b { color: var(--text); } +.grid-actions { display: flex; gap: .4rem; } + +.table-wrap { overflow-x: auto; background: var(--panel); border: 1px solid var(--line); border-radius: var(--radius); } +table { border-collapse: collapse; width: 100%; font-size: .84rem; } +th, td { padding: .38rem .55rem; text-align: left; white-space: nowrap; border-bottom: 1px solid var(--line); } +th { position: sticky; top: 0; background: var(--panel-2); font-weight: 600; font-size: .76rem; letter-spacing: .03em; text-transform: uppercase; color: var(--dim); cursor: pointer; user-select: none; z-index: 1; } +th.sorted { color: var(--accent); } +tbody tr:hover { background: var(--panel-2); } +tbody tr.kept-row { opacity: .72; } +td.num { text-align: right; font-variant-numeric: tabular-nums; } +td.title-cell { max-width: 380px; overflow: hidden; text-overflow: ellipsis; } +.title-link { cursor: pointer; } +.title-link:hover { text-decoration: underline; } +.sub-title { color: var(--faint); font-size: .78rem; } + +.score-pill { display: inline-block; min-width: 44px; text-align: center; padding: .08rem .35rem; border-radius: 4px; font-variant-numeric: tabular-nums; font-weight: 600; } +.flag { font-size: .66rem; padding: .04rem .28rem; border-radius: 3px; margin-left: .25rem; border: 1px solid; } +.flag.pre_history { color: var(--warm); border-color: var(--warm); } +.flag.rejected { color: var(--hot); border-color: var(--hot); } +.flag.duplicate { color: var(--accent); border-color: var(--accent); } +.flag.kept { color: var(--kept); border-color: var(--kept); } +.flag.multi_part, .flag.history_gap { color: var(--faint); border-color: var(--faint); } + +.pager { display: flex; gap: .6rem; align-items: center; justify-content: center; padding: .7rem; color: var(--dim); font-size: .85rem; } + +/* ── drawer ─────────────────────────────────────────────────────── */ +.scrim { position: fixed; inset: 0; background: rgba(0,0,0,.45); z-index: 40; } +.drawer { + position: fixed; top: 0; right: 0; bottom: 0; width: min(680px, 92vw); + background: var(--panel); border-left: 1px solid var(--line); + z-index: 50; overflow: auto; padding: 1rem; +} +.drawer-head { display: flex; justify-content: space-between; align-items: flex-start; gap: 1rem; margin-bottom: .8rem; } +.kv { display: grid; grid-template-columns: 190px 1fr; gap: .2rem .8rem; font-size: .85rem; margin-bottom: 1rem; } +.kv dt { color: var(--dim); } +.kv dd { margin: 0; word-break: break-all; } +.comp-row { display: grid; grid-template-columns: 110px 1fr 46px; gap: .5rem; align-items: center; font-size: .8rem; margin-bottom: .2rem; } +.comp-track { background: var(--panel-2); height: 8px; border-radius: 3px; overflow: hidden; } +.comp-fill { height: 100%; background: var(--accent); } + +/* ── misc ───────────────────────────────────────────────────────── */ +.banner { padding: .55rem 1rem; font-size: .85rem; border-bottom: 1px solid; } +.banner.warn { background: rgba(240,179,87,.12); border-color: var(--warm); color: var(--warm); } +.banner.info { background: rgba(110,168,254,.1); border-color: var(--accent); color: var(--accent); } +.lib-rules label { display: flex; align-items: center; gap: .5rem; padding: .25rem 0; font-size: .87rem; } +.mark { display: grid; grid-template-columns: 1fr auto; gap: .5rem; padding: .5rem 0; border-bottom: 1px solid var(--line); align-items: center; } +.mark .meta { font-size: .78rem; color: var(--dim); } +.empty { color: var(--faint); font-size: .87rem; padding: .6rem 0; } +.dupe-group { border-bottom: 1px solid var(--line); padding: .55rem 0; } +.dupe-copies { display: flex; gap: 1rem; flex-wrap: wrap; font-size: .82rem; color: var(--dim); margin-top: .2rem; } +.scatter { width: 100%; height: 260px; } +.scatter circle { cursor: pointer; } +@media (max-width: 1000px) { + .chart-grid { grid-template-columns: 1fr; } + .grid-layout { grid-template-columns: 1fr; } + .filter-rail { position: static; max-height: none; } + .split-hero { grid-template-columns: 1fr; } +} diff --git a/mediashelf/static/app.js b/mediashelf/static/app.js new file mode 100644 index 0000000..b11edda --- /dev/null +++ b/mediashelf/static/app.js @@ -0,0 +1,753 @@ +'use strict'; +/* MediaShelf UI. Vanilla JS, no build step, no vendored framework (§9). */ + +const $ = (s, r = document) => r.querySelector(s); +const $$ = (s, r = document) => Array.from(r.querySelectorAll(s)); + +const state = { + view: 'dashboard', + page: 1, + pageSize: 100, + sort: 'reclaim_score:desc', + selected: new Set(), + weights: null, + defaultWeights: { + size: 0.28, staleness: 0.24, unpopularity: 0.22, + solitude: 0.10, age: 0.10, rejection: 0.06, + }, + libraries: [], + views: [], + lastPage: null, + hasCompletion: true, +}; + +/* ── helpers ──────────────────────────────────────────────────────── */ + +function bytes(n) { + n = Number(n || 0); + const u = ['B', 'KB', 'MB', 'GB', 'TB', 'PB']; + let i = 0; + while (Math.abs(n) >= 1024 && i < u.length - 1) { n /= 1024; i++; } + return `${n.toFixed(n >= 100 || i === 0 ? 0 : 1)} ${u[i]}`; +} +function exact(n) { return Number(n || 0).toLocaleString() + ' bytes'; } +function date(ts) { + if (!ts) return '—'; + return new Date(ts * 1000).toISOString().slice(0, 10); +} +function ago(ts) { + if (!ts) return 'never'; + const d = Math.floor((Date.now() / 1000 - ts) / 86400); + if (d < 1) return 'today'; + if (d < 60) return `${d}d`; + if (d < 730) return `${Math.floor(d / 30)}mo`; + return `${(d / 365).toFixed(1)}y`; +} +/* "today" and "3 days ago" read differently — don't blindly suffix " ago". */ +function agoPhrase(ts) { + if (!ts) return 'never'; + const a = ago(ts); + return a === 'today' ? 'today' : a + ' ago'; +} +function el(tag, attrs = {}, ...kids) { + const n = document.createElement(tag); + for (const [k, v] of Object.entries(attrs)) { + if (v === null || v === undefined || v === false) continue; + if (k === 'class') n.className = v; + else if (k === 'text') n.textContent = v; + else if (k === 'html') n.innerHTML = v; + else if (k.startsWith('on')) n.addEventListener(k.slice(2), v); + else n.setAttribute(k, v); + } + for (const kid of kids.flat()) { + if (kid === null || kid === undefined || kid === false) continue; + n.append(kid.nodeType ? kid : document.createTextNode(kid)); + } + return n; +} +async function api(path, opts) { + const r = await fetch('/api/v1' + path, opts); + if (!r.ok) { + let msg = r.statusText; + try { msg = (await r.json()).message || msg; } catch (_) {} + throw new Error(msg); + } + return r.json(); +} +function scoreColor(s) { + if (s === null || s === undefined) return 'var(--faint)'; + if (s >= 75) return 'var(--hot)'; + if (s >= 50) return 'var(--warm)'; + if (s >= 25) return 'var(--dim)'; + return 'var(--good)'; +} +function banner(kind, text) { + $('#banner-area').append(el('div', { class: `banner ${kind}`, text })); +} + +/* ── navigation ───────────────────────────────────────────────────── */ + +$$('.tab').forEach(t => t.addEventListener('click', () => show(t.dataset.view))); + +const VIEWS = ['dashboard', 'grid', 'keeps', 'views', 'duplicates', 'scans']; + +function show(name) { + if (!VIEWS.includes(name)) name = 'dashboard'; + state.view = name; + $$('.tab').forEach(t => t.classList.toggle('active', t.dataset.view === name)); + $$('.view').forEach(v => v.classList.toggle('active', v.id === 'view-' + name)); + const loader = { dashboard: loadDashboard, grid: loadGrid, keeps: loadKeeps, + views: loadViews, duplicates: loadDuplicates, scans: loadScans }[name]; + if (loader) loader(); + if (location.hash.replace('#', '') !== name) { + const url = new URL(location); + url.hash = name; + history.replaceState(null, '', url); + } +} + +/* A hash-only change does not reload the document, so without this a pasted + #grid link (or the back button) leaves the dashboard on screen. */ +window.addEventListener('hashchange', () => { + const name = location.hash.replace('#', '') || 'dashboard'; + if (name !== state.view) show(name); +}); + +/* ── dashboard ────────────────────────────────────────────────────── */ + +async function loadDashboard() { + const [ov, libs, comp, added, scatter] = await Promise.all([ + api('/stats/overview'), api('/stats/size-by-library'), + api('/stats/completion'), api('/stats/added-over-time'), + api('/stats/size-vs-lastwatched'), + ]); + state.hasCompletion = ov.has_completion_data; + + // The three-way split that keeps the keep list honest (§6.6) + $('#reclaim-split').replaceChildren( + el('div', { class: 'split-cell' }, + el('div', { class: 'label', text: 'Never played' }), + el('div', { class: 'value', text: bytes(ov.never_played_bytes) }), + el('div', { class: 'sub', text: `${bytes(ov.confident_bytes)} confident · ${bytes(ov.uncertain_bytes)} uncertain` })), + el('div', { class: 'split-cell kept' }, + el('div', { class: 'label', text: 'Kept' }), + el('div', { class: 'value', text: bytes(ov.never_played_kept_bytes) }), + el('div', { class: 'sub', text: `${ov.keep_marks} mark(s) · ${bytes(ov.kept_bytes)} kept overall` })), + el('div', { class: 'split-cell available' }, + el('div', { class: 'label', text: 'Available' }), + el('div', { class: 'value', text: bytes(ov.available_bytes) }), + el('div', { class: 'sub', text: 'never played and not kept' })), + ); + + const pct = ov.never_played_bytes ? (ov.never_played_kept_bytes / ov.never_played_bytes) : 0; + if (pct > 0.5) { + banner('warn', `${Math.round(pct * 100)}% of never-played content is marked keep — ` + + `the reclaim report is mostly reporting on things you have decided to keep.`); + } + + $('#tiles').replaceChildren( + tile('Library size', bytes(ov.total_bytes), `${ov.total_items.toLocaleString()} rows · ${ov.libraries} libraries`), + tile('Episodes', ov.episodes.toLocaleString(), 'rolled up into seasons'), + tile('Cold 2+ years', bytes(ov.cold_bytes), 'not played in two years'), + tile('Watch history', ov.watch_events.toLocaleString() + ' plays', + ov.history_since ? `${ov.accounts} users since ${date(ov.history_since)}` : 'no history'), + tile('Last scan', agoPhrase(ov.last_scan_at), + ov.history_source ? 'via ' + ov.history_source : ''), + ); + + if (ov.history_since) { + const badge = $('#source-badge'); + badge.textContent = `${ov.history_source || 'no history'} · since ${date(ov.history_since)}`; + badge.title = ov.has_completion_data + ? 'Completion data available — the rejection component is active' + : 'No completion data — running degraded, rejection component disabled'; + } + if (!ov.has_completion_data && ov.watch_events > 0) { + banner('warn', 'Watch history has no completion data (Plex fallback). ' + + 'The score is running degraded: the “rejection” component is disabled.'); + } + + // size by library + const maxL = Math.max(...libs.libraries.map(l => l.size_bytes), 1); + $('#chart-libraries').replaceChildren(...libs.libraries.map(l => + el('div', { class: 'bar-row' }, + el('div', { class: 'bar-label', text: l.title, title: l.title }), + el('div', { class: 'bar-track' }, + el('div', { class: 'bar-fill', style: `width:${(l.size_bytes - l.never_bytes) / maxL * 100}%` }), + el('div', { class: 'bar-fill never', style: `width:${l.never_bytes / maxL * 100}%` })), + el('div', { class: 'bar-num', text: bytes(l.size_bytes), title: exact(l.size_bytes) })))); + $('#chart-libraries').append(el('div', { class: 'legend' }, + el('span', {}, el('i', { style: 'background:var(--accent)' }), 'played'), + el('span', {}, el('i', { style: 'background:var(--warm)' }), 'never played'))); + + // completion split + const total = comp.finished + comp.started + comp.never || 1; + const seg = (v, c, label) => v / total > 0.04 + ? el('div', { style: `width:${v / total * 100}%;background:${c}`, text: label, title: bytes(v) }) + : el('div', { style: `width:${v / total * 100}%;background:${c}`, title: `${label}: ${bytes(v)}` }); + $('#chart-completion').replaceChildren( + el('div', { class: 'stack' }, + seg(comp.finished, 'var(--good)', bytes(comp.finished)), + seg(comp.started, 'var(--warm)', bytes(comp.started)), + seg(comp.never, 'var(--hot)', bytes(comp.never))), + el('div', { class: 'legend' }, + el('span', {}, el('i', { style: 'background:var(--good)' }), 'finished'), + el('span', {}, el('i', { style: 'background:var(--warm)' }), 'started, never finished'), + el('span', {}, el('i', { style: 'background:var(--hot)' }), 'never opened'))); + + // added over time + const maxA = Math.max(...added.buckets.map(b => b.size_bytes), 1); + const recent = added.buckets.slice(-48); + $('#chart-added').replaceChildren(el('div', { + style: 'display:flex;align-items:flex-end;gap:2px;height:120px', + }, ...recent.map(b => el('div', { + style: `flex:1;min-width:3px;background:var(--accent);height:${Math.max(b.size_bytes / maxA * 100, 1)}%`, + title: `${b.period}: ${bytes(b.size_bytes)} (${b.items} items)`, + })))); + if (recent.length) { + $('#chart-added').append(el('div', { class: 'legend' }, + el('span', { text: recent[0].period }), el('span', { text: '→' }), + el('span', { text: recent[recent.length - 1].period }))); + } + + drawScatter(scatter.points); +} + +function tile(label, value, sub) { + return el('div', { class: 'tile' }, + el('div', { class: 'label', text: label }), + el('div', { class: 'value', text: value }), + sub ? el('div', { class: 'sub', text: sub }) : null); +} + +function drawScatter(points) { + const W = 900, H = 260, PAD = 34; + const now = Date.now() / 1000; + const maxSize = Math.max(...points.map(p => p.size_bytes), 1); + const svg = document.createElementNS('http://www.w3.org/2000/svg', 'svg'); + svg.setAttribute('viewBox', `0 0 ${W} ${H}`); + svg.setAttribute('class', 'scatter'); + svg.setAttribute('preserveAspectRatio', 'none'); + + const mk = (t, a) => { + const n = document.createElementNS('http://www.w3.org/2000/svg', t); + for (const [k, v] of Object.entries(a)) n.setAttribute(k, v); + return n; + }; + // axes + svg.append(mk('line', { x1: PAD, y1: H - PAD, x2: W - 4, y2: H - PAD, stroke: 'var(--line)' })); + svg.append(mk('line', { x1: PAD, y1: 4, x2: PAD, y2: H - PAD, stroke: 'var(--line)' })); + + const maxDays = 3650; + for (const p of points) { + const days = p.last_watched_at ? Math.min((now - p.last_watched_at) / 86400, maxDays) : maxDays; + const x = PAD + (days / maxDays) * (W - PAD - 8); + const y = (H - PAD) - (Math.log10(1 + p.size_bytes) / Math.log10(1 + maxSize)) * (H - PAD - 8); + const c = mk('circle', { + cx: x.toFixed(1), cy: y.toFixed(1), r: 2.4, + fill: p.kept ? 'var(--kept)' : (p.watch_count ? 'var(--accent)' : 'var(--warm)'), + 'fill-opacity': .55, + }); + const title = document.createElementNS('http://www.w3.org/2000/svg', 'title'); + title.textContent = `${p.title} — ${bytes(p.size_bytes)}, ` + + (p.last_watched_at ? `last played ${agoPhrase(p.last_watched_at)}` : 'never played'); + c.append(title); + c.addEventListener('click', () => openItem(p.id)); + svg.append(c); + } + const t1 = mk('text', { x: PAD, y: H - 8, fill: 'var(--faint)', 'font-size': 10 }); + t1.textContent = 'recently played'; + const t2 = mk('text', { x: W - 90, y: H - 8, fill: 'var(--faint)', 'font-size': 10 }); + t2.textContent = 'never / 10y+'; + svg.append(t1, t2); + $('#chart-scatter').replaceChildren(svg); +} + +/* ── grid ─────────────────────────────────────────────────────────── */ + +const COLUMNS = [ + { key: '_sel', label: '', sortable: false }, + { key: 'title', label: 'Title' }, + { key: 'library', label: 'Library' }, + { key: 'size_bytes', label: 'Size', num: true }, + { key: 'added_at', label: 'Added', num: true }, + { key: 'last_watched_at', label: 'Last played', num: true }, + { key: 'watch_count', label: 'Plays', num: true }, + { key: 'abandoned_count', label: 'Bailed', num: true }, + { key: 'distinct_watcher_count', label: 'Viewers', num: true }, + { key: 'reclaim_score', label: 'Reclaim', num: true }, + { key: '_keep', label: '', sortable: false }, +]; + +function gridParams() { + const p = new URLSearchParams(); + const q = $('#f-q').value.trim(); + if (q) p.set('q', q); + $$('#f-libraries input:checked').forEach(i => p.append('library_id', i.value)); + $$('.f-kind:checked').forEach(i => p.append('kind', i.value)); + if ($('#f-kept').checked) p.set('include_kept', '1'); + if ($('#f-missing').checked) p.set('include_missing', '1'); + + const rules = { op: 'and', rules: [] }; + const minSize = Number($('#f-minsize').value); + if (minSize > 0) rules.rules.push({ field: 'size_bytes', op: 'gte', value: minSize }); + + switch ($('#f-watch').value) { + case 'never': rules.rules.push({ field: 'watch_count', op: 'eq', value: 0 }); break; + case 'confident': + rules.rules.push({ field: 'watch_count', op: 'eq', value: 0 }, + { field: 'pre_history', op: 'eq', value: false }); break; + case 'uncertain': + rules.rules.push({ field: 'watch_count', op: 'eq', value: 0 }, + { field: 'pre_history', op: 'eq', value: true }); break; + case 'rejected': + rules.rules.push({ field: 'abandoned_count', op: 'gte', value: 1 }, + { field: 'watch_count', op: 'eq', value: 0 }); break; + case 'watched': rules.rules.push({ field: 'watch_count', op: 'gte', value: 1 }); break; + } + const viewId = $('#f-view').value; + if (viewId) p.set('view_id', viewId); + if (rules.rules.length) p.set('rules', JSON.stringify(rules)); + if (state.weights) p.set('weights', JSON.stringify(state.weights)); + p.set('sort', state.sort); + p.set('page', state.page); + p.set('page_size', state.pageSize); + return p; +} + +async function loadGrid() { + if (!state.libraries.length) await loadFilterOptions(); + let data; + try { + data = await api('/items?' + gridParams()); + } catch (e) { + $('#grid-body').replaceChildren(el('tr', {}, el('td', { colspan: COLUMNS.length, class: 'empty', text: 'Query failed: ' + e.message }))); + return; + } + state.lastPage = data; + renderHead(); + renderRows(data); + + renderSummary(data); + $('#page-info').textContent = + `page ${data.page} of ${Math.max(1, Math.ceil(data.total / data.page_size))}`; + $('#page-prev').disabled = data.page <= 1; + $('#page-next').disabled = data.page * data.page_size >= data.total; +} + +function selectedBytes() { + if (!state.lastPage) return 0; + return state.lastPage.items + .filter(i => state.selected.has(i.id)) + .reduce((s, i) => s + (i.size_bytes || 0), 0); +} + +function renderHead() { + $('#grid-head').replaceChildren(...COLUMNS.map(c => { + if (c.key === '_sel') { + return el('th', {}, el('input', { + type: 'checkbox', title: 'select all on this page', + onchange: (e) => { + (state.lastPage?.items || []).forEach(i => + e.target.checked ? state.selected.add(i.id) : state.selected.delete(i.id)); + renderRows(state.lastPage); loadGridSummaryOnly(); + }, + })); + } + const active = state.sort.startsWith(c.key + ':'); + const th = el('th', { + class: active ? 'sorted' : '', + text: c.label + (active ? (state.sort.endsWith('desc') ? ' ↓' : ' ↑') : ''), + }); + if (c.sortable !== false) { + th.addEventListener('click', () => { + const dir = state.sort === c.key + ':desc' ? 'asc' : 'desc'; + state.sort = c.key + ':' + dir; + state.page = 1; + loadGrid(); + }); + } + return th; + })); +} + +/* replaceChildren() stringifies null into the literal text "null" rather than + skipping it, so the optional selection span is filtered out, not passed in. */ +function renderSummary(data) { + const parts = [el('span', { + html: `${data.total.toLocaleString()} rows · ${bytes(data.aggregate.total_size_bytes)}`, + })]; + if (state.selected.size) { + parts.push(el('span', { + html: ` — ${state.selected.size} selected · ${bytes(selectedBytes())}`, + })); + } + $('#grid-summary').replaceChildren(...parts); + $('#btn-keep').disabled = state.selected.size === 0; +} + +function loadGridSummaryOnly() { + renderSummary(state.lastPage); +} + +function renderRows(data) { + if (!data.items.length) { + $('#grid-body').replaceChildren(el('tr', {}, el('td', { + colspan: COLUMNS.length, class: 'empty', + text: 'Nothing matches. If everything here is kept, tick “Kept items” to see it.', + }))); + return; + } + $('#grid-body').replaceChildren(...data.items.map(it => { + const label = it.kind === 'season' + ? `${it.show_title || '?'} — Season ${it.season_number ?? '?'}` + : it.title + (it.year ? ` (${it.year})` : ''); + const flags = it.flags.map(f => el('span', { class: 'flag ' + f, text: f.replace('_', ' ') })); + if (it.kept) flags.unshift(el('span', { class: 'flag kept', text: 'kept · ' + it.kept_via })); + + return el('tr', { class: it.kept ? 'kept-row' : '' }, + el('td', {}, el('input', { + type: 'checkbox', checked: state.selected.has(it.id), + onchange: (e) => { + e.target.checked ? state.selected.add(it.id) : state.selected.delete(it.id); + loadGridSummaryOnly(); + }, + })), + el('td', { class: 'title-cell' }, + el('span', { class: 'title-link', text: label, onclick: () => openItem(it.id) }), + it.kind === 'season' ? el('span', { class: 'sub-title', text: ` · ${it.episode_count} eps` }) : null, + ...flags), + el('td', { text: it.library.title }), + el('td', { class: 'num', text: bytes(it.size_bytes), title: exact(it.size_bytes) }), + el('td', { class: 'num', text: date(it.added_at) }), + el('td', { class: 'num', text: it.last_watched_at ? ago(it.last_watched_at) : 'never' }), + el('td', { class: 'num', text: it.watch_count }), + el('td', { class: 'num', text: it.abandoned_count || '' }), + el('td', { class: 'num', text: it.distinct_watcher_count || '' }), + el('td', { class: 'num' }, el('span', { + class: 'score-pill', + style: `color:${scoreColor(it.reclaim_score)}`, + text: it.kept ? '—' : (it.reclaim_score ?? '—'), + title: componentTooltip(it), + })), + el('td', {}, el('button', { + class: 'btn small ghost', text: it.kept ? 'Un-keep' : 'Keep', + onclick: () => toggleKeep(it), + }))); + })); +} + +function componentTooltip(it) { + const c = it.reclaim_components || {}; + const lines = Object.entries(c) + .map(([k, v]) => `${k}: ${v === null ? 'n/a' : v.toFixed(2)}`); + if (it.grace) lines.push(`grace: ${it.grace}`); + return lines.join('\n'); +} + +async function toggleKeep(it) { + try { + if (it.kept) { + const marks = await api('/keeps'); + const mine = marks.marks.find(m => m.id === it.kept_mark_id); + if (!mine) { + alert('This is kept by a library rule — turn that off on the Kept tab.'); + return; + } + await api('/keeps/' + mine.id, { method: 'DELETE' }); + } else { + await api('/keeps', { + method: 'POST', headers: { 'Content-Type': 'application/json' }, + body: JSON.stringify({ item_id: it.id, mode: 'keep' }), + }); + } + loadGrid(); + } catch (e) { alert(e.message); } +} + +$('#btn-keep').addEventListener('click', async () => { + const ids = Array.from(state.selected); + const note = prompt(`Keep ${ids.length} item(s). Optional note — why are you keeping these?`, ''); + if (note === null) return; + try { + const r = await api('/keeps/bulk', { + method: 'POST', headers: { 'Content-Type': 'application/json' }, + body: JSON.stringify({ item_ids: ids, mode: 'keep', note: note || null }), + }); + state.selected.clear(); + if (r.failed.length) { + alert(`${r.created} kept. ${r.failed.length} could not be: ${r.failed[0].reason}`); + } + loadGrid(); + } catch (e) { alert(e.message); } +}); + +$('#btn-export').addEventListener('click', () => { + location.href = '/api/v1/export.csv?' + gridParams(); +}); +$('#page-prev').addEventListener('click', () => { state.page--; loadGrid(); }); +$('#page-next').addEventListener('click', () => { state.page++; loadGrid(); }); + +['#f-q', '#f-minsize', '#f-watch', '#f-view', '#f-kept', '#f-missing'].forEach(sel => { + const node = $(sel); + const ev = node.tagName === 'INPUT' && node.type === 'search' ? 'input' : 'change'; + let t; + node.addEventListener(ev, () => { + clearTimeout(t); + t = setTimeout(() => { state.page = 1; loadGrid(); }, ev === 'input' ? 300 : 0); + }); +}); + +async function loadFilterOptions() { + const [libs, views] = await Promise.all([api('/libraries'), api('/views')]); + state.libraries = libs.libraries; + state.views = views.views; + $('#f-libraries').replaceChildren(...libs.libraries.map(l => + el('label', {}, el('input', { + type: 'checkbox', value: l.id, + onchange: () => { state.page = 1; loadGrid(); }, + }), `${l.title} (${bytes(l.size_bytes)})`))); + $('#f-view').replaceChildren(el('option', { value: '', text: '— none —' }), + ...views.views.map(v => el('option', { value: v.id, text: v.name, title: v.description || '' }))); + buildWeightSliders(); + $$('.f-kind').forEach(i => i.addEventListener('change', () => { state.page = 1; loadGrid(); })); +} + +function buildWeightSliders() { + const w = state.weights || state.defaultWeights; + $('#weight-sliders').replaceChildren(...Object.keys(state.defaultWeights).map(k => + el('label', {}, + `${k} `, el('span', { id: 'wv-' + k, text: (w[k] ?? 0).toFixed(2) }), + el('input', { + type: 'range', min: 0, max: 1, step: 0.02, value: w[k] ?? 0, + oninput: (e) => { + state.weights = { ...(state.weights || state.defaultWeights) }; + state.weights[k] = Number(e.target.value); + $('#wv-' + k).textContent = Number(e.target.value).toFixed(2); + clearTimeout(buildWeightSliders._t); + buildWeightSliders._t = setTimeout(loadGrid, 250); + }, + })))); +} +$('#weights-reset').addEventListener('click', () => { + state.weights = null; buildWeightSliders(); loadGrid(); +}); + +/* ── item drawer ──────────────────────────────────────────────────── */ + +async function openItem(id) { + const it = await api('/items/' + id); + $('#drawer-title').textContent = it.kind === 'season' + ? `${it.show_title} — Season ${it.season_number}` + : it.title + (it.year ? ` (${it.year})` : ''); + + const body = $('#drawer-body'); + body.replaceChildren(); + + const dl = el('dl', { class: 'kv' }); + const add = (k, v) => { dl.append(el('dt', { text: k }), el('dd', { text: v })); }; + add('Library', it.library.title); + add('Size', `${bytes(it.size_bytes)} (${exact(it.size_bytes)})`); + add('Added', date(it.added_at)); + add('Last played', it.last_watched_at ? `${date(it.last_watched_at)} (${agoPhrase(it.last_watched_at)})` : 'never'); + add('Plays', `${it.watch_count} finished · ${it.partial_count} partial · ${it.abandoned_count} abandoned`); + add('Distinct viewers', it.distinct_watcher_count); + if (it.avg_percent_complete !== null && it.avg_percent_complete !== undefined) { + add('Average completion', it.avg_percent_complete.toFixed(0) + '%'); + } + if (it.kind === 'season') add('Episodes', it.episode_count); + add('Files', it.part_count); + if (it.pre_history) { + add('Note', 'Added before watch history began — “never played” is unproven here.'); + } + add('Kept', it.kept ? `yes (via ${it.kept_via})` : 'no'); + body.append(dl); + + body.append(el('h3', { text: 'Reclaim score' })); + body.append(el('div', { class: 'hint', text: it.kept + ? 'Not scored for deletion: this item is kept.' + : `Score ${it.reclaim_score}${it.grace ? ` (clamped: ${it.grace})` : ''}` })); + for (const [k, v] of Object.entries(it.reclaim_components || {})) { + body.append(el('div', { class: 'comp-row' }, + el('span', { text: k }), + el('div', { class: 'comp-track' }, + el('div', { class: 'comp-fill', style: `width:${(v ?? 0) * 100}%` })), + el('span', { text: v === null ? 'n/a' : v.toFixed(2) }))); + } + + if (it.duplicates?.length) { + body.append(el('h3', { text: 'Other copies' })); + it.duplicates.forEach(d => body.append(el('div', { class: 'hint' }, + `${d.library_title}: ${bytes(d.size_bytes)} ${d.resolution || ''} · ${d.watch_count} plays`))); + } + + body.append(el('h3', { text: 'Files' })); + if (it.parts?.length) { + it.parts.forEach(p => body.append(el('div', { class: 'hint', text: `${bytes(p.size_bytes)} — ${p.file_path}` }))); + } else if (it.episodes?.length) { + body.append(el('div', { class: 'hint', text: `${it.episodes.length} episodes` })); + it.episodes.forEach(e => body.append(el('div', { class: 'hint' }, + `E${String(e.episode_number).padStart(2, '0')} · ${bytes(e.size_bytes)} · ` + + `${e.watch_count} plays · added ${date(e.added_at)}`))); + } + + if (it.watch_history?.length) { + body.append(el('h3', { text: 'Watch history' })); + it.watch_history.slice(0, 40).forEach(h => body.append(el('div', { class: 'hint' }, + `${date(h.viewed_at)} · ${h.who || 'unknown'} · ` + + `${h.percent_complete === null ? 'completion unknown' : h.percent_complete + '%'} (${h.disposition})`))); + } + + $('#drawer').hidden = false; + $('#drawer-scrim').hidden = false; +} +const closeDrawer = () => { $('#drawer').hidden = true; $('#drawer-scrim').hidden = true; }; +$('#drawer-close').addEventListener('click', closeDrawer); +$('#drawer-scrim').addEventListener('click', closeDrawer); +document.addEventListener('keydown', e => { if (e.key === 'Escape') closeDrawer(); }); + +/* ── kept ─────────────────────────────────────────────────────────── */ + +async function loadKeeps() { + const [data, libs] = await Promise.all([api('/keeps'), api('/libraries')]); + + $('#keep-libraries').replaceChildren(...libs.libraries.map(l => + el('label', {}, el('input', { + type: 'checkbox', checked: !!l.keep_all, + onchange: async (e) => { + await api(`/libraries/${l.id}/keep_all`, { + method: 'PUT', headers: { 'Content-Type': 'application/json' }, + body: JSON.stringify({ keep_all: e.target.checked }), + }); + loadKeeps(); + }, + }), `${l.title} — ${bytes(l.size_bytes)}`))); + + const live = data.marks.filter(m => !m.orphaned); + const orphans = data.marks.filter(m => m.orphaned); + + $('#keep-marks').replaceChildren( + el('div', { class: 'hint', text: + `${data.kept_items.toLocaleString()} items kept, totalling ${bytes(data.kept_bytes)}.` }), + ...(live.length ? live.map(markRow) : [el('div', { class: 'empty', + text: 'Nothing marked yet. Keep something from the Library tab.' })])); + + $('#keep-orphans-card').hidden = orphans.length === 0; + $('#keep-orphans').replaceChildren(...orphans.map(markRow)); +} + +function markRow(m) { + return el('div', { class: 'mark' }, + el('div', {}, + el('div', { text: m.label }), + el('div', { class: 'meta', text: + `${m.scope} · ${m.library_title} · ${m.mode}` + + ` · ${m.resolved_items} item(s), ${bytes(m.resolved_bytes)}` + + (m.note ? ` · “${m.note}”` : '') + + ` · marked ${date(m.created_at)}` })), + el('button', { + class: 'btn small ghost', text: 'Remove', + onclick: async () => { + if (!confirm(`Remove the keep on “${m.label}”?`)) return; + await api('/keeps/' + m.id, { method: 'DELETE' }); + loadKeeps(); + }, + })); +} + +/* ── views / duplicates / scans ───────────────────────────────────── */ + +async function loadViews() { + const data = await api('/views'); + $('#views-list').replaceChildren(...data.views.map(v => + el('div', { class: 'mark' }, + el('div', {}, + el('div', { text: v.name + (v.builtin ? ' · built in' : '') }), + el('div', { class: 'meta', text: v.description || '' })), + el('button', { + class: 'btn small', text: 'Open', + onclick: () => { $('#f-view').value = v.id; state.page = 1; show('grid'); }, + })))); +} + +async function loadDuplicates() { + const data = await api('/duplicates'); + $('#dupes-summary').textContent = + `${data.group_count} group(s) · ${bytes(data.total_redundant_bytes)} redundant.`; + $('#dupes-list').replaceChildren(...(data.groups.length ? data.groups.slice(0, 200).map(g => + el('div', { class: 'dupe-group' }, + el('div', {}, el('b', { text: g.title }), g.year ? ` (${g.year})` : '', + ` — ${bytes(g.redundant_bytes)} redundant`), + el('div', { class: 'dupe-copies' }, ...g.copies.map(c => + el('span', { class: 'title-link', onclick: () => openItem(c.id) }, + `${c.library_title}: ${bytes(c.size_bytes)} ${c.resolution || ''} · ${c.watch_count} plays`))))) + : [el('div', { class: 'empty', text: 'No duplicate GUIDs across libraries.' })])); +} + +async function loadScans() { + const [scans, sources] = await Promise.all([api('/scans'), api('/sources')]); + const cov = sources.coverage[0]; + const info = [ + el('div', { class: 'hint', text: `Plex: ${sources.plex.base_url || 'not configured'}` }), + el('div', { class: 'hint', text: `Tautulli: ${sources.tautulli.base_url || 'not configured'}` }), + el('div', { class: 'hint', text: `History source: ${sources.history_source || 'none'}` + + (sources.has_completion_data ? ' (completion data available)' : ' (no completion data — degraded)') }), + ]; + if (cov) { + info.push(el('div', { class: 'hint', text: + `Coverage: ${date(cov.earliest_event_at)} → ${date(cov.latest_event_at)}, ` + + `${cov.event_count.toLocaleString()} events` })); + } + $('#sources-info').replaceChildren(...info); + + $('#scans-list').replaceChildren(...(scans.scans.length ? scans.scans.map(s => + el('div', { class: 'mark' }, + el('div', {}, + el('div', { text: `${s.mode} · ${s.status}` + (s.error ? ` — ${s.error}` : '') }), + el('div', { class: 'meta', text: + `${date(s.started_at)} · seen ${s.items_seen} · added ${s.items_added} · ` + + `events ${s.events_added} · missing ${s.items_missing} · warnings ${s.warning_count}` + + (s.finished_at ? ` · took ${s.finished_at - s.started_at}s` : '') })), + el('span', { class: 'meta', text: s.history_source || '' }))) + : [el('div', { class: 'empty', text: 'No scans yet.' })])); +} + +/* ── scan control ─────────────────────────────────────────────────── */ + +$('#scan-now').addEventListener('click', async () => { + try { + await api('/scans', { + method: 'POST', headers: { 'Content-Type': 'application/json' }, + body: JSON.stringify({ mode: 'incremental' }), + }); + pollScan(); + } catch (e) { alert(e.message); } +}); + +async function pollScan() { + const btn = $('#scan-now'); + btn.disabled = true; + const tick = async () => { + const s = await api('/scans/current'); + if (s) { + btn.textContent = s.progress ? `Scanning: ${s.progress}` : 'Scanning…'; + setTimeout(tick, 2000); + } else { + btn.textContent = 'Scan now'; + btn.disabled = false; + $('#banner-area').replaceChildren(); + show(state.view); + } + }; + tick(); +} + +/* ── boot ─────────────────────────────────────────────────────────── */ + +(async function boot() { + show(location.hash.replace('#', '') || 'dashboard'); + try { + const s = await api('/scans/current'); + if (s) pollScan(); + } catch (_) {} +})(); diff --git a/mediashelf/templates/index.html b/mediashelf/templates/index.html new file mode 100644 index 0000000..780d353 --- /dev/null +++ b/mediashelf/templates/index.html @@ -0,0 +1,200 @@ + + + + + +MediaShelf + + + + + +
+
+ + MediaShelf + report only +
+ +
+ + +
+
+ + + +
+ +
+
+
+
+
+

Size by library

+
+
+
+

Finished · started & abandoned · never opened

+

The middle band is invisible without completion data.

+
+
+
+

Added over time

+
+
+
+

Size vs. time since last watched

+

Upper right is the reclaim target. Click a point to open it.

+
+
+
+
+ + +
+
+ + +
+
+
+
+ + + +
+
+
+ + + +
+
+
+ + + +
+
+
+
+ + +
+
+

Library rules

+

A whole library kept by rule — anything added to it later is kept too. + Nothing is set by default.

+
+
+
+

Explicit marks

+
+
+ +
+ + +
+

Saved views

+
+ + +
+
+

Duplicate groups

+

The same film held more than once, matched on Plex GUID. Neither copy + looks like a candidate on its own because plays are split between them. + This is a report — which copy supersedes which depends on what plays back well + on your devices, which MediaShelf has no way to know.

+
+
+
+
+ + +
+

Sources

+

Scan history

+
+
+ + + + + + + diff --git a/mediashelf/web.py b/mediashelf/web.py new file mode 100644 index 0000000..4f813e7 --- /dev/null +++ b/mediashelf/web.py @@ -0,0 +1,13 @@ +"""Server-rendered shell. All data comes from /api/v1 (§9).""" + +from __future__ import annotations + +from flask import Blueprint, current_app, render_template + +bp = Blueprint("web", __name__) + + +@bp.get("/") +def index(): + cfg = current_app.extensions["mediashelf"]["config"] + return render_template("index.html", tz=cfg.tz) diff --git a/pytest.ini b/pytest.ini new file mode 100644 index 0000000..505a8a0 --- /dev/null +++ b/pytest.ini @@ -0,0 +1,4 @@ +[pytest] +testpaths = tests +filterwarnings = + ignore::DeprecationWarning diff --git a/requirements-dev.txt b/requirements-dev.txt new file mode 100644 index 0000000..8c146eb --- /dev/null +++ b/requirements-dev.txt @@ -0,0 +1,3 @@ +-r requirements.txt +pytest==8.3.3 +responses==0.25.3 diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 0000000..3793640 --- /dev/null +++ b/requirements.txt @@ -0,0 +1,4 @@ +Flask==3.0.3 +gunicorn==22.0.0 +APScheduler==3.10.4 +requests==2.32.3 diff --git a/tests/__init__.py b/tests/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/tests/conftest.py b/tests/conftest.py new file mode 100644 index 0000000..7877d2e --- /dev/null +++ b/tests/conftest.py @@ -0,0 +1,79 @@ +import os +import sys +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parent.parent +sys.path.insert(0, str(ROOT)) +sys.path.insert(0, str(ROOT / "tools")) + +import mockserver # noqa: E402 + +from mediashelf import ingest, providers # noqa: E402 +from mediashelf.app import create_app # noqa: E402 +from mediashelf.config import Config # noqa: E402 +from mediashelf.db import Database # noqa: E402 + + +@pytest.fixture(scope="session") +def fake_server(): + with mockserver.FakeServer() as srv: + yield srv + + +@pytest.fixture +def cfg(fake_server, tmp_path, monkeypatch): + env = { + "PLEX_BASE_URL": fake_server.url, + "PLEX_TOKEN": "testtoken", + "TAUTULLI_BASE_URL": fake_server.url, + "TAUTULLI_API_KEY": "testkey", + "DATABASE_PATH": str(tmp_path / "test.db"), + "PLEX_PAGE_SIZE": "25", + "TAUTULLI_PAGE_SIZE": "300", + "PLEX_TIMEOUT_S": "10", + "TAUTULLI_TIMEOUT_S": "10", + "SCHEDULER_ENABLED": "0", + "KEEP_ALL_LIBRARIES": "", + } + for k, v in env.items(): + monkeypatch.setenv(k, v) + return Config.from_env() + + +@pytest.fixture +def db(cfg): + d = Database(cfg.database_path) + d.migrate() + from mediashelf.scoring import register_sqlite_functions + register_sqlite_functions(d.conn) + yield d + d.close() + + +@pytest.fixture +def scanned(db, cfg): + """A database with one completed full scan against the fake server.""" + media = providers.build_media_provider(cfg) + history, _ = providers.build_history_provider(cfg, media) + result = ingest.Ingest(db, cfg, media, history).run("full", "manual") + assert result.status == "succeeded", result.error + return db + + +@pytest.fixture +def rescan(db, cfg): + def _go(mode="full"): + media = providers.build_media_provider(cfg) + history, _ = providers.build_history_provider(cfg, media) + return ingest.Ingest(db, cfg, media, history).run(mode, "manual") + return _go + + +@pytest.fixture +def client(cfg, scanned): + app = create_app(cfg, start_scheduler=False) + app.config["TESTING"] = True + with app.test_client() as c: + yield c diff --git a/tests/test_ingest.py b/tests/test_ingest.py new file mode 100644 index 0000000..573343c --- /dev/null +++ b/tests/test_ingest.py @@ -0,0 +1,121 @@ +"""Ingest correctness. The headline property is idempotency: a scanner that +double-counts produces a report that looks plausible and is wrong, which is worse +than one that crashes.""" + +import mockserver + + +def snapshot(db): + return { + "items": db.scalar("SELECT COUNT(*) FROM media_item"), + "unit_bytes": db.scalar( + "SELECT COALESCE(SUM(size_bytes),0) FROM media_item " + "WHERE kind IN ('movie','season')"), + "parts": db.scalar("SELECT COUNT(*) FROM media_part"), + "part_bytes": db.scalar("SELECT COALESCE(SUM(size_bytes),0) FROM media_part"), + "episodes": db.scalar("SELECT COUNT(*) FROM episode"), + "events": db.scalar("SELECT COUNT(*) FROM watch_event"), + "watch_sum": db.scalar( + "SELECT COALESCE(SUM(watch_count),0) FROM media_item WHERE kind='movie'"), + "missing": db.scalar("SELECT COUNT(*) FROM media_item WHERE status='missing'"), + } + + +def test_scan_succeeds_and_totals_match_source(scanned): + expected = mockserver.stats() + assert scanned.scalar( + "SELECT COALESCE(SUM(size_bytes),0) FROM media_item WHERE kind IN ('movie','season')" + ) == expected["total_bytes"] + assert scanned.scalar("SELECT COUNT(*) FROM episode") == expected["episodes"] + assert scanned.scalar("SELECT COUNT(*) FROM watch_event") == expected["history"] + + +def test_scan_is_idempotent(scanned, rescan): + first = snapshot(scanned) + rescan("full") + second = snapshot(scanned) + rescan("full") + third = snapshot(scanned) + assert first == second == third, "a repeated scan changed the data" + + +def test_incremental_adds_no_duplicate_events(scanned, rescan): + before = scanned.scalar("SELECT COUNT(*) FROM watch_event") + r = rescan("incremental") + assert r.status == "succeeded" + assert scanned.scalar("SELECT COUNT(*) FROM watch_event") == before + + +def test_seasons_are_the_unit_for_tv(scanned): + seasons = scanned.query( + "SELECT * FROM media_item WHERE kind='season' AND episode_count > 0") + assert seasons, "no seasons were built" + for s in seasons: + rolled = scanned.scalar( + "SELECT COALESCE(SUM(size_bytes),0) FROM episode " + "WHERE season_item_id=? AND status='present'", (s["id"],)) + assert s["size_bytes"] == rolled + assert s["parent_id"] is not None, "season is not linked to its show" + + +def test_multi_part_items_sum_all_parts(scanned): + """A movie held as two files must report the total, not the first part.""" + rows = scanned.query( + "SELECT id, size_bytes, part_count FROM media_item " + "WHERE kind='movie' AND part_count > 1") + assert rows, "fixture has no multi-part movies" + for r in rows: + total = scanned.scalar( + "SELECT SUM(size_bytes) FROM media_part WHERE media_item_id=?", (r["id"],)) + assert r["size_bytes"] == total + + +def test_dispositions_are_classified(scanned, cfg): + got = {r[0]: r[1] for r in scanned.query( + "SELECT disposition, COUNT(*) FROM watch_event GROUP BY 1")} + assert set(got) <= {"completed", "partial", "abandoned"} + assert got.get("abandoned", 0) > 0, "fixture should produce abandoned plays" + bad = scanned.scalar( + "SELECT COUNT(*) FROM watch_event WHERE disposition='completed' " + "AND percent_complete IS NOT NULL AND percent_complete < ?", + (cfg.completion_threshold,)) + assert bad == 0 + + +def test_pre_history_flag_tracks_coverage(scanned): + cov = scanned.one("SELECT * FROM history_coverage ORDER BY event_count DESC LIMIT 1") + assert cov and cov["earliest_event_at"] + wrong = scanned.scalar( + "SELECT COUNT(*) FROM media_item WHERE pre_history=1 AND added_at >= ?", + (cov["earliest_event_at"],)) + assert wrong == 0 + assert scanned.scalar("SELECT COUNT(*) FROM media_item WHERE pre_history=1") > 0 + + +def test_missing_items_are_flagged_not_deleted(scanned, rescan, monkeypatch): + victim = scanned.one("SELECT * FROM media_item WHERE kind='movie' LIMIT 1") + removed = [m for m in mockserver.MOVIES + if m["ratingKey"] == victim["provider_item_id"]] + assert removed + monkeypatch.setattr(mockserver, "MOVIES", + [m for m in mockserver.MOVIES + if m["ratingKey"] != victim["provider_item_id"]]) + rescan("full") + row = scanned.one("SELECT * FROM media_item WHERE id=?", (victim["id"],)) + assert row is not None, "a vanished item was deleted rather than flagged" + assert row["status"] == "missing" + + +def test_refuses_history_from_a_different_plex_server(db, cfg, monkeypatch): + """Joining another server's history would produce confident nonsense (§4.11).""" + from mediashelf import ingest, providers + from mediashelf.providers.base import ProviderError, ServerInfo + + media = providers.build_media_provider(cfg) + history, _ = providers.build_history_provider(cfg, media) + monkeypatch.setattr(history, "server_info", + lambda: ServerInfo(kind="tautulli", name="Other", + server_id="TOTALLY-DIFFERENT")) + result = ingest.Ingest(db, cfg, media, history).run("full", "manual") + assert result.status == "failed" + assert "different plex server" in (result.error or "").lower() diff --git a/tests/test_keeps.py b/tests/test_keeps.py new file mode 100644 index 0000000..363a9a4 --- /dev/null +++ b/tests/test_keeps.py @@ -0,0 +1,164 @@ +"""Keep marks (§6.6). + +The single most important test in the suite is +test_marks_survive_every_rating_key_being_reassigned. Its failure mode in v2 is +deleting content someone explicitly protected. +""" + +import pytest + +from mediashelf import keeps +from mediashelf.keeps import KeepError + + +def kept_ids(db): + return {r["id"] for r in db.query("SELECT id FROM media_item WHERE kept=1")} + + +def a_movie(db): + return db.one("SELECT * FROM media_item WHERE kind='movie' " + "AND guid IS NOT NULL ORDER BY size_bytes DESC LIMIT 1") + + +def a_season(db): + return db.one("SELECT * FROM media_item WHERE kind='season' " + "AND show_guid IS NOT NULL ORDER BY size_bytes DESC LIMIT 1") + + +def test_marks_survive_every_rating_key_being_reassigned(scanned): + """Plex reassigns ratingKeys on library rebuilds. Marks must not detach.""" + movie, season = a_movie(scanned), a_season(scanned) + keeps.create_from_item(scanned, movie["id"], "keep", "expensive to re-acquire") + keeps.create_from_item(scanned, season["id"], "keep", "might watch someday") + keeps.resolve_all(scanned) + before = kept_ids(scanned) + assert len(before) == 2 + + for table in ("media_item", "episode", "watch_event"): + scanned.execute( + f"UPDATE {table} SET provider_item_id = 'REBUILD-' || provider_item_id") + + keeps.resolve_all(scanned) + assert kept_ids(scanned) == before, "keeps detached when rating keys changed" + + +def test_same_guid_in_two_libraries_marks_independently(scanned): + """Movies and 4K Movies share GUIDs; an unscoped mark would keep both.""" + dupe_guid = scanned.scalar( + "SELECT guid FROM media_item WHERE kind='movie' AND guid IS NOT NULL " + "GROUP BY guid HAVING COUNT(*) > 1 LIMIT 1") + assert dupe_guid, "fixture has no cross-library duplicates" + copies = scanned.query( + "SELECT * FROM media_item WHERE guid=? AND kind='movie' ORDER BY library_id", + (dupe_guid,)) + assert len(copies) == 2 + + keeps.create_from_item(scanned, copies[0]["id"], "keep") + keeps.resolve_all(scanned) + + assert scanned.scalar("SELECT kept FROM media_item WHERE id=?", (copies[0]["id"],)) == 1 + assert scanned.scalar("SELECT kept FROM media_item WHERE id=?", (copies[1]["id"],)) == 0 + + +def test_library_rule_keeps_everything_in_it(scanned): + lib_id = scanned.scalar("SELECT id FROM library WHERE kind='movie' LIMIT 1") + n = scanned.scalar("SELECT COUNT(*) FROM media_item WHERE library_id=?", (lib_id,)) + keeps.set_library_keep_all(scanned, lib_id, True) + keeps.resolve_all(scanned) + assert scanned.scalar( + "SELECT COUNT(*) FROM media_item WHERE library_id=? AND kept=1", (lib_id,)) == n + + +def test_explicit_exclude_overrides_a_library_rule(scanned): + """Without this, 'keep all of X except one' is a dead end.""" + lib_id = scanned.scalar("SELECT id FROM library WHERE kind='movie' LIMIT 1") + keeps.set_library_keep_all(scanned, lib_id, True) + keeps.resolve_all(scanned) + victim = scanned.one( + "SELECT * FROM media_item WHERE library_id=? AND kind='movie' " + "AND guid IS NOT NULL LIMIT 1", (lib_id,)) + assert scanned.scalar("SELECT kept FROM media_item WHERE id=?", (victim["id"],)) == 1 + + keeps.create_from_item(scanned, victim["id"], "exclude", "actually don't want this") + keeps.resolve_all(scanned) + assert scanned.scalar("SELECT kept FROM media_item WHERE id=?", (victim["id"],)) == 0 + + +def test_show_mark_keeps_all_its_seasons(scanned): + show = scanned.one( + "SELECT * FROM media_item WHERE kind='show' AND guid IS NOT NULL " + "AND (SELECT COUNT(*) FROM media_item s WHERE s.parent_id = media_item.id) > 1 " + "LIMIT 1") + assert show, "fixture has no multi-season show" + keeps.create_from_item(scanned, show["id"], "keep", "whole series") + keeps.resolve_all(scanned) + unkept = scanned.scalar( + "SELECT COUNT(*) FROM media_item WHERE parent_id=? AND kept=0", (show["id"],)) + assert unkept == 0 + + +def test_season_mark_keeps_only_that_season(scanned): + season = a_season(scanned) + keeps.create_from_item(scanned, season["id"], "keep") + keeps.resolve_all(scanned) + siblings = scanned.query( + "SELECT id, kept FROM media_item WHERE parent_id=? AND id != ?", + (season["parent_id"], season["id"])) + assert scanned.scalar("SELECT kept FROM media_item WHERE id=?", (season["id"],)) == 1 + assert all(s["kept"] == 0 for s in siblings) + + +def test_orphan_detection(scanned): + movie = a_movie(scanned) + keeps.create_from_item(scanned, movie["id"], "keep") + keeps.resolve_all(scanned) + assert keeps.stamp_matches(scanned, 1) == 0 + + # content leaves the library entirely + scanned.execute("DELETE FROM media_item WHERE id=?", (movie["id"],)) + keeps.resolve_all(scanned) + assert keeps.stamp_matches(scanned, 2) == 1, "orphaned mark was not detected" + assert scanned.scalar("SELECT COUNT(*) FROM keep_mark") == 1, \ + "an orphaned mark must never be auto-deleted" + + +def test_export_import_round_trip(scanned): + movie, season = a_movie(scanned), a_season(scanned) + keeps.create_from_item(scanned, movie["id"], "keep", "note one") + keeps.create_from_item(scanned, season["id"], "keep", "note two") + lib_id = scanned.scalar("SELECT id FROM library WHERE kind='movie' LIMIT 1") + keeps.set_library_keep_all(scanned, lib_id, True) + keeps.resolve_all(scanned) + before = kept_ids(scanned) + + payload = keeps.export(scanned) + assert payload["version"] == 1 + assert len(payload["marks"]) == 2 + assert payload["library_keep_all"] + + scanned.execute("DELETE FROM keep_mark") + scanned.execute("UPDATE library SET keep_all=0") + keeps.resolve_all(scanned) + assert kept_ids(scanned) == set() + + keeps.import_(scanned, payload) + assert kept_ids(scanned) == before, "restore did not reproduce the keep state" + + +def test_refuses_to_mark_an_item_with_no_guid(scanned): + """A mark that cannot survive a rebuild is worse than no mark.""" + movie = a_movie(scanned) + scanned.execute("UPDATE media_item SET guid=NULL WHERE id=?", (movie["id"],)) + with pytest.raises(KeepError, match="GUID"): + keeps.create_from_item(scanned, movie["id"], "keep") + + +def test_kept_items_are_hidden_from_the_grid_by_default(client, scanned): + total = client.get("/api/v1/items?page_size=1").get_json()["total"] + movie = a_movie(scanned) + r = client.post("/api/v1/keeps", json={"item_id": movie["id"], "note": "x"}) + assert r.status_code == 201 + + assert client.get("/api/v1/items?page_size=1").get_json()["total"] == total - 1 + assert client.get( + "/api/v1/items?page_size=1&include_kept=1").get_json()["total"] == total diff --git a/tests/test_packaging.py b/tests/test_packaging.py new file mode 100644 index 0000000..57e966b --- /dev/null +++ b/tests/test_packaging.py @@ -0,0 +1,77 @@ +"""Deployment-shape checks. + +A variable set in Portainer's UI but missing from the compose file's +`environment:` block silently does nothing. That cost real debugging time on the +Mythica stack; this test makes it impossible to reintroduce here. +""" + +import re +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parent.parent + +yaml = pytest.importorskip("yaml") + + +def declared_in_compose() -> set[str]: + spec = yaml.safe_load((ROOT / "docker-compose.yml").read_text()) + env = spec["services"]["mediashelf"]["environment"] + return {e.split("=", 1)[0] for e in env} + + +def read_by_config() -> set[str]: + src = (ROOT / "mediashelf" / "config.py").read_text() + return (set(re.findall(r'_[bisf]\("([A-Z_][A-Z0-9_]*)"', src)) + | set(re.findall(r'_csv\("([A-Z_][A-Z0-9_]*)"\)', src))) + + +def test_every_config_var_is_declared_in_the_stack(): + missing = read_by_config() - declared_in_compose() + assert not missing, ( + "these are read by config.py but absent from docker-compose.yml's " + "environment block, so setting them in Portainer would silently do " + f"nothing: {sorted(missing)}") + + +def test_compose_declares_nothing_the_app_ignores(): + extra = declared_in_compose() - read_by_config() + assert not extra, f"compose declares unused variables: {sorted(extra)}" + + +def test_keep_all_libraries_defaults_to_empty(): + """Nothing ships kept (§6.6).""" + spec = yaml.safe_load((ROOT / "docker-compose.yml").read_text()) + env = spec["services"]["mediashelf"]["environment"] + line = next(e for e in env if e.startswith("KEEP_ALL_LIBRARIES=")) + assert line.endswith(":-}") or line.endswith("="), \ + f"a library is pre-kept in the shipped stack: {line}" + + example = (ROOT / ".env.example").read_text() + assert re.search(r"^KEEP_ALL_LIBRARIES=\s*$", example, re.M), \ + ".env.example must not name a library" + + +def test_image_is_not_pinned_to_bare_latest(): + spec = yaml.safe_load((ROOT / "docker-compose.yml").read_text()) + image = spec["services"]["mediashelf"]["image"] + assert "latest" not in image, ( + "a bare :latest tag is what forced the Mythica stack to be recreated " + "when a PUT update kept serving old code (§11.2)") + + +def test_dockerfile_runs_as_non_root(): + df = (ROOT / "Dockerfile").read_text() + user_lines = [l for l in df.splitlines() if l.startswith("USER ")] + assert user_lines and not user_lines[-1].strip().endswith("root") + + +def test_no_credentials_committed(): + """Cheap guard against the SOAP-password-in-git-history problem.""" + suspicious = re.compile( + r"(PLEX_TOKEN|TAUTULLI_API_KEY|MEDIASHELF_SECRET_KEY)\s*=\s*['\"]?[A-Za-z0-9]{16,}") + for path in list(ROOT.glob("*.yml")) + list(ROOT.glob("*.py")) + \ + list((ROOT / "mediashelf").rglob("*.py")) + [ROOT / ".env.example"]: + text = path.read_text() + assert not suspicious.search(text), f"possible credential in {path.name}" diff --git a/tests/test_rules_and_api.py b/tests/test_rules_and_api.py new file mode 100644 index 0000000..9ce6933 --- /dev/null +++ b/tests/test_rules_and_api.py @@ -0,0 +1,184 @@ +"""Rule grammar (must reject everything outside the whitelist) and API shapes.""" + +import json + +import pytest + +from mediashelf.rules import RuleError, compile_rules, compile_sort + +NOW = 1_757_000_000 + + +def test_basic_conditions_compile_to_bound_parameters(): + sql, params = compile_rules( + {"op": "and", "rules": [ + {"field": "size_bytes", "op": "gte", "value": 1024}, + {"field": "watch_count", "op": "eq", "value": 0}]}, NOW) + assert "i.size_bytes >=" in sql and "i.watch_count =" in sql + assert set(params.values()) == {1024, 0} + assert "1024" not in sql, "value was interpolated instead of bound" + + +def test_nested_groups(): + sql, _ = compile_rules({"op": "or", "rules": [ + {"field": "kind", "op": "eq", "value": "movie"}, + {"op": "and", "rules": [ + {"field": "kind", "op": "eq", "value": "season"}, + {"field": "episode_count", "op": "gt", "value": 5}]}]}, NOW) + assert sql.count("(") >= 2 and " OR " in sql and " AND " in sql + + +@pytest.mark.parametrize("field", [ + "title; DROP TABLE media_item", + "1=1", + "i.size_bytes) OR (1", + "nonexistent_column", + "", +]) +def test_unknown_fields_are_rejected(field): + with pytest.raises(RuleError): + compile_rules({"field": field, "op": "eq", "value": 1}, NOW) + + +@pytest.mark.parametrize("op", ["exec", "'; --", "regexp", ""]) +def test_unknown_operators_are_rejected(op): + with pytest.raises(RuleError): + compile_rules({"field": "title", "op": op, "value": "x"}, NOW) + + +def test_like_wildcards_in_user_input_are_escaped(): + sql, params = compile_rules( + {"field": "title", "op": "contains", "value": "100% _real_"}, NOW) + assert "ESCAPE" in sql + assert list(params.values())[0] == r"%100\% \_real\_%" + + +def test_relative_date_operators(): + sql, params = compile_rules( + {"field": "last_watched_at", "op": "older_than_days", "value": 730}, NOW) + assert "IS NOT NULL" in sql + assert list(params.values())[0] == NOW - 730 * 86400 + + +def test_never_operator_treats_zero_and_null_alike(): + sql, _ = compile_rules({"field": "watch_count", "op": "never"}, NOW) + assert "IS NULL" in sql and "= 0" in sql + + +def test_in_operator_bounds_list_length(): + with pytest.raises(RuleError): + compile_rules({"field": "library_id", "op": "in", + "value": list(range(600))}, NOW) + with pytest.raises(RuleError): + compile_rules({"field": "library_id", "op": "in", "value": []}, NOW) + + +def test_deeply_nested_rules_are_rejected(): + node = {"field": "watch_count", "op": "eq", "value": 0} + for _ in range(12): + node = {"op": "and", "rules": [node]} + with pytest.raises(RuleError): + compile_rules(node, NOW) + + +@pytest.mark.parametrize("spec", ["size_bytes); DROP TABLE x--", "evil:desc", "1"]) +def test_sort_whitelist(spec): + with pytest.raises(RuleError): + compile_sort(spec) + + +def test_sort_has_a_stable_tiebreak(): + assert compile_sort("size_bytes:desc").endswith("i.id ASC") + + +# ── API ────────────────────────────────────────────────────────────────── + +def test_core_endpoints_respond(client): + for path in ["/healthz", "/api/v1/stats/overview", "/api/v1/libraries", + "/api/v1/views", "/api/v1/duplicates", "/api/v1/scans", + "/api/v1/sources", "/api/v1/keeps", "/api/v1/accounts", + "/api/v1/stats/completion", "/api/v1/stats/added-over-time", + "/api/v1/stats/watch-distribution", "/"]: + assert client.get(path).status_code == 200, path + + +def test_items_shape(client): + data = client.get("/api/v1/items?page_size=5").get_json() + assert {"total", "page", "page_size", "aggregate", "items"} <= set(data) + assert data["total"] > 0 + item = data["items"][0] + for key in ("id", "kind", "title", "library", "size_bytes", "reclaim_score", + "reclaim_components", "kept", "flags", "pre_history"): + assert key in item, key + assert set(item["reclaim_components"]) == { + "size", "staleness", "unpopularity", "solitude", "age", "rejection"} + + +def test_shows_are_excluded_from_the_grid_by_default(client): + data = client.get("/api/v1/items?page_size=500").get_json() + assert all(i["kind"] != "show" for i in data["items"]) + + +def test_injection_via_rules_is_a_400_not_a_500(client): + bad = json.dumps({"op": "and", "rules": [ + {"field": "title; DROP TABLE media_item", "op": "eq", "value": "x"}]}) + r = client.get("/api/v1/items?rules=" + bad) + assert r.status_code == 400 + assert r.get_json()["error"] == "invalid_rule" + assert client.get("/api/v1/items?page_size=1").get_json()["total"] > 0 + + +def test_malformed_json_rules_is_a_400(client): + assert client.get("/api/v1/items?rules=notjson").status_code == 400 + + +def test_search_handles_hostile_input(client): + for q in ['"', 'a OR b', 'NEAR(', '*', "'; --"]: + assert client.get("/api/v1/items?q=" + q).status_code == 200, q + + +def test_csv_export_streams_every_row_not_just_a_page(client): + total = client.get("/api/v1/items?page_size=1").get_json()["total"] + body = client.get("/api/v1/export.csv?page_size=1").data.decode() + lines = [l for l in body.splitlines() if l.strip()] + assert len(lines) == total + 1 + + +def test_saved_views_are_seeded_and_runnable(client): + views = client.get("/api/v1/views").get_json()["views"] + names = {v["name"] for v in views} + assert {"Confident reclaim", "Uncertain reclaim", "Kept"} <= names + for v in views: + r = client.get("/api/v1/items?page_size=1&view_id=%d" % v["id"]) + assert r.status_code == 200, v["name"] + + +def test_builtin_views_cannot_be_deleted(client): + v = next(v for v in client.get("/api/v1/views").get_json()["views"] if v["builtin"]) + assert client.delete("/api/v1/views/%d" % v["id"]).get_json()["deleted"] is False + + +def test_settings_never_leak_credentials(client): + body = client.get("/api/v1/settings").get_json() + assert body["plex_token"] == "***" + assert body["tautulli_api_key"] == "***" + assert body["secret_key"] == "***" + assert "testtoken" not in json.dumps(body) + assert "testkey" not in json.dumps(body) + + +def test_overview_reports_the_three_way_split(client): + ov = client.get("/api/v1/stats/overview").get_json() + for key in ("never_played_bytes", "kept_bytes", "available_bytes", + "confident_bytes", "uncertain_bytes", "has_completion_data"): + assert key in ov, key + assert ov["confident_bytes"] + ov["uncertain_bytes"] <= ov["never_played_bytes"] + + +def test_bulk_keep(client, scanned): + ids = [i["id"] for i in + client.get("/api/v1/items?page_size=5").get_json()["items"]] + r = client.post("/api/v1/keeps/bulk", json={"item_ids": ids, "note": "batch"}) + assert r.status_code == 200 + assert r.get_json()["created"] == len(ids) + assert client.get("/api/v1/keeps").get_json()["kept_items"] >= len(ids) diff --git a/tests/test_scoring.py b/tests/test_scoring.py new file mode 100644 index 0000000..1242e98 --- /dev/null +++ b/tests/test_scoring.py @@ -0,0 +1,159 @@ +"""The score is implemented twice — SQL for the live grid, Python for export and +tests. Two implementations of one formula is a real risk; this is the mitigation.""" + +import random +import time + +import pytest + +from mediashelf import scoring + + +def make_rows(n=500, seed=17): + rnd = random.Random(seed) + now = int(time.time()) + rows = [] + for i in range(n): + kind = rnd.choice(["movie", "season"]) + rows.append({ + "provider_item_id": str(i), + "kind": kind, + "title": "T%d" % i, + "size_bytes": rnd.choice([0, 1, 10**6, 2 * 10**9, 60 * 10**9, 213 * 10**9]), + "added_at": rnd.choice([None, now - rnd.randint(1, 5000) * 86400]), + "last_watched_at": rnd.choice([None, None, now - rnd.randint(1, 4000) * 86400]), + "watch_count": rnd.choice([0, 0, 0, 1, 2, 5, 24]), + "abandoned_count": rnd.randint(0, 5), + "distinct_watcher_count": rnd.randint(0, 8), + "episode_count": rnd.randint(1, 30) if kind == "season" else 0, + "pre_history": rnd.choice([0, 1]), + }) + return now, rows + + +WEIGHT_PROFILES = [ + None, + {"size": 1, "staleness": 0, "unpopularity": 0, "solitude": 0, "age": 0, "rejection": 0}, + {"size": 0, "staleness": 0, "unpopularity": 0, "solitude": 0, "age": 0, "rejection": 1}, + {"size": .1, "staleness": .5, "unpopularity": .1, "solitude": .1, "age": .1, "rejection": .1}, + {"size": -5, "staleness": "x", "unpopularity": .2}, # junk must be tolerated +] + + +@pytest.mark.parametrize("has_completion", [True, False]) +@pytest.mark.parametrize("weights", WEIGHT_PROFILES) +def test_sql_and_python_agree(db, has_completion, weights): + now, rows = make_rows() + db.execute("INSERT INTO provider (kind,name,base_url,created_at) " + "VALUES ('plex','L','http://x',1)") + db.execute("INSERT INTO library (provider_id,provider_key,title,kind) " + "VALUES (1,'1','Movies','movie')") + for r in rows: + db.execute( + "INSERT INTO media_item (provider_id,library_id,provider_item_id,kind,title," + "size_bytes,added_at,last_watched_at,watch_count,abandoned_count," + "distinct_watcher_count,episode_count,pre_history) " + "VALUES (1,1,:provider_item_id,:kind,:title,:size_bytes,:added_at," + ":last_watched_at,:watch_count,:abandoned_count,:distinct_watcher_count," + ":episode_count,:pre_history)", r) + + maxsize = db.scalar("SELECT MAX(size_bytes) FROM media_item") or 1 + ctx = scoring.ScoreContext(now=now, max_size_bytes=maxsize, + has_completion_data=has_completion) + + # Compare the raw formula, so a real divergence cannot hide behind rounding. + expr, params = scoring.sql_expression(ctx, weights, rounded=False) + params["now"] = now + raw = {r["pid"]: r["score"] for r in db.query( + "WITH s(maxsize) AS (SELECT MAX(size_bytes) FROM media_item) " + f"SELECT i.provider_item_id AS pid, {expr} AS score FROM media_item i, s", params)} + + rexpr, rparams = scoring.sql_expression(ctx, weights, rounded=True) + rparams["now"] = now + shown = {r["pid"]: r["score"] for r in db.query( + "WITH s(maxsize) AS (SELECT MAX(size_bytes) FROM media_item) " + f"SELECT i.provider_item_id AS pid, {rexpr} AS score FROM media_item i, s", rparams)} + + for r in rows: + out = scoring.score_row(r, ctx, weights) + pid = r["provider_item_id"] + assert abs(out["score_raw"] - raw[pid]) <= 1e-9, ( + f"formula diverges on {pid}: SQL {raw[pid]} vs Python {out['score_raw']}") + # Displayed values must agree to the last displayed digit. Exact equality + # is NOT assertable: SQLite and Python evaluate the same formula in a + # different order, so a raw score sitting exactly on a .xx5 boundary can + # land on either side of it while still agreeing to 1e-9 above. + # Compared in integer hundredths: subtracting two 2dp floats does not + # give exactly 0.01, so a float tolerance here fails on its own rounding. + assert abs(round(out["score"] * 100) - round(shown[pid] * 100)) <= 1, ( + f"rounding diverges on {pid}: SQL {shown[pid]} vs Python {out['score']}") + + +def test_new_arrival_grace_clamps_to_zero(): + now = int(time.time()) + ctx = scoring.ScoreContext(now=now, max_size_bytes=10**12) + row = {"kind": "movie", "size_bytes": 10**12, "added_at": now - 5 * 86400, + "last_watched_at": None, "watch_count": 0, "abandoned_count": 0, + "distinct_watcher_count": 0, "episode_count": 0, "pre_history": 0} + out = scoring.score_row(row, ctx) + assert out["score"] == 0.0 and out["grace"] == "new" + + +def test_recent_watch_grace_caps_the_score(): + now = int(time.time()) + ctx = scoring.ScoreContext(now=now, max_size_bytes=10**12) + row = {"kind": "movie", "size_bytes": 10**12, "added_at": now - 3000 * 86400, + "last_watched_at": now - 5 * 86400, "watch_count": 1, "abandoned_count": 0, + "distinct_watcher_count": 1, "episode_count": 0, "pre_history": 0} + out = scoring.score_row(row, ctx) + assert out["score"] <= 25.0 and out["grace"] == "recent" + + +def test_rejection_is_unavailable_not_zero_without_completion_data(): + """A missing component must renormalize, not drag every score down (§6.2).""" + now = int(time.time()) + row = {"kind": "movie", "size_bytes": 5 * 10**9, "added_at": now - 2000 * 86400, + "last_watched_at": None, "watch_count": 0, "abandoned_count": 0, + "distinct_watcher_count": 0, "episode_count": 0, "pre_history": 0} + with_cd = scoring.score_row( + row, scoring.ScoreContext(now=now, max_size_bytes=10**11, has_completion_data=True)) + without = scoring.score_row( + row, scoring.ScoreContext(now=now, max_size_bytes=10**11, has_completion_data=False)) + assert without["components"]["rejection"] is None + # rejection would have scored 0 here, so dropping it must RAISE the score + assert without["score"] > with_cd["score"] + + +def test_rejection_zeroes_once_something_is_finished(): + now = int(time.time()) + ctx = scoring.ScoreContext(now=now, max_size_bytes=10**11) + base = {"kind": "movie", "size_bytes": 5 * 10**9, "added_at": now - 2000 * 86400, + "last_watched_at": None, "abandoned_count": 4, + "distinct_watcher_count": 2, "episode_count": 0, "pre_history": 0} + assert scoring.components({**base, "watch_count": 0}, ctx)["rejection"] == 1.0 + assert scoring.components({**base, "watch_count": 1}, ctx)["rejection"] == 0.0 + + +def test_tv_watches_are_normalized_per_episode(): + """A 24-episode season watched once must not look 24x more popular.""" + now = int(time.time()) + ctx = scoring.ScoreContext(now=now, max_size_bytes=10**11) + season = {"kind": "season", "size_bytes": 10**10, "added_at": now - 1000 * 86400, + "last_watched_at": now - 500 * 86400, "watch_count": 24, + "abandoned_count": 0, "distinct_watcher_count": 1, + "episode_count": 24, "pre_history": 0} + movie = {**season, "kind": "movie", "watch_count": 1, "episode_count": 0} + assert scoring.components(season, ctx)["unpopularity"] == \ + pytest.approx(scoring.components(movie, ctx)["unpopularity"]) + + +def test_pre_history_never_watched_is_capped_below_truly_never_watched(): + now = int(time.time()) + ctx = scoring.ScoreContext(now=now, max_size_bytes=10**11) + base = {"kind": "movie", "size_bytes": 10**10, "added_at": now - 3000 * 86400, + "last_watched_at": None, "watch_count": 0, "abandoned_count": 0, + "distinct_watcher_count": 0, "episode_count": 0} + known = scoring.components({**base, "pre_history": 0}, ctx)["staleness"] + unknown = scoring.components({**base, "pre_history": 1}, ctx)["staleness"] + assert known == 1.0 + assert unknown < known diff --git a/tools/mockserver.py b/tools/mockserver.py index 66d57cd..8cc2c4b 100644 --- a/tools/mockserver.py +++ b/tools/mockserver.py @@ -1,119 +1,291 @@ #!/usr/bin/env python3 -"""Throwaway mock of the Plex + Tautulli endpoints probe.py uses, for testing it -without a live server. Not part of the application.""" +"""A fake Plex + Tautulli pair, good enough to exercise the whole ingest. + +Used two ways: + * standalone, to try tools/probe.py without touching a live server + * imported by the test suite (tests/conftest.py) as a fixture + +It deliberately reproduces the awkward parts of the real APIs: multi-version +movies, split parts, shows whose episodes carry no show GUID, Tautulli's +200-with-result-error failure mode, and the fact that a library's item count and +its reported totalSize have to agree for paging to terminate. + +Not part of the application. +""" + +from __future__ import annotations + import json import random +import threading import time import urllib.parse -from http.server import BaseHTTPRequestHandler, HTTPServer +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer NOW = int(time.time()) -random.seed(7) +DAY = 86400 -MOVIES = [] -for i in range(120): - parts = [{"file": "/mnt/vault2/movies/Film %d/Film %d.mkv" % (i, i), - "size": random.randint(2, 60) * 10**9, "container": "mkv"}] - if i % 17 == 0: # some multi-version items - parts.append({"file": "/mnt/vault1/movies4k/Film %d/Film %d.4k.mkv" % (i, i), - "size": random.randint(40, 90) * 10**9, "container": "mkv"}) - MOVIES.append({ - "ratingKey": str(1000 + i), "type": "movie", "title": "Film %d" % i, - "year": 1990 + i % 35, "addedAt": NOW - random.randint(30, 3000) * 86400, - "viewCount": random.choice([0, 0, 0, 1, 2]), - "Media": [{"videoResolution": "1080", "Part": parts}], - }) +MACHINE_ID = "abc123def456abc123def456" +COVERAGE_DAYS = 400 # Tautulli has been logging this long -EPISODES = [] -for s in range(8): - for se in range(1, 4): - for ep in range(1, 11): - EPISODES.append({ - "ratingKey": str(50000 + len(EPISODES)), "type": "episode", - "title": "Ep %d" % ep, "index": ep, - "grandparentRatingKey": str(900 + s), "grandparentTitle": "Show %d" % s, - "parentRatingKey": str(9000 + s * 10 + se), "parentIndex": se, - "addedAt": NOW - random.randint(30, 2000) * 86400, - "viewCount": random.choice([0, 0, 1]), - "Media": [{"Part": [{"file": "/mnt/vault2/tv/Show %d/S%02d/E%02d.mkv" % (s, se, ep), - "size": random.randint(1, 4) * 10**9}]}], + +def _build(seed=7): + rnd = random.Random(seed) + movies, episodes, shows, history = [], [], [], [] + + # ── movies, in two libraries that share GUIDs (the duplicate case) ── + for i in range(60): + guid = "plex://movie/%04d" % i + parts = [{ + "id": 9000 + i, "file": "/mnt/titan4/Movies/Film %d/Film %d.mkv" % (i, i), + "size": rnd.randint(2, 40) * 10**9, "container": "mkv", + }] + if i % 12 == 0: # a split file + parts.append({ + "id": 9500 + i, + "file": "/mnt/titan4/Movies/Film %d/Film %d.part2.mkv" % (i, i), + "size": rnd.randint(1, 8) * 10**9, "container": "mkv", }) + movies.append({ + "sectionKey": "1", "ratingKey": str(1000 + i), "guid": guid, "type": "movie", + "title": "Film %d" % i, "titleSort": "Film %04d" % i, "year": 1990 + i % 35, + "addedAt": NOW - rnd.randint(10, 3000) * DAY, + "updatedAt": NOW - rnd.randint(1, 500) * DAY, + "duration": rnd.randint(80, 190) * 60000, + "viewCount": rnd.choice([0, 0, 0, 1, 2]), + "Media": [{"videoResolution": "1080", "videoCodec": "h264", + "audioCodec": "eac3", "bitrate": 8000, "Part": parts}], + }) -HISTORY = [] -for i in range(500): - HISTORY.append({ - "row_id": i, "reference_id": i // 2, "date": NOW - random.randint(1, 900) * 86400, - "rating_key": str(random.choice([m["ratingKey"] for m in MOVIES])), - "user_id": random.choice([1, 2, 3]), "user": "u%d" % random.choice([1, 2, 3]), - "friendly_name": random.choice(["Jess", "Sam", "Guest"]), - "media_type": "movie", - "percent_complete": random.choice([3, 8, 12, 45, 60, 92, 97, 100]), - "watched_status": random.choice([0, 0.5, 1]), - "play_duration": random.randint(60, 9000), "paused_counter": 0, - }) -HISTORY.sort(key=lambda h: h["date"]) + # 4K copies of the first 15 films: same GUID, different library + for i in range(15): + movies.append({ + "sectionKey": "2", "ratingKey": str(2000 + i), "guid": "plex://movie/%04d" % i, + "type": "movie", "title": "Film %d" % i, "year": 1990 + i % 35, + "addedAt": NOW - rnd.randint(10, 900) * DAY, + "duration": rnd.randint(80, 190) * 60000, "viewCount": rnd.choice([0, 1]), + "Media": [{"videoResolution": "4k", "videoCodec": "hevc", "bitrate": 40000, + "Part": [{"id": 7000 + i, + "file": "/mnt/titan4/4K Movies/Film %d/Film %d.4k.mkv" % (i, i), + "size": rnd.randint(40, 90) * 10**9, "container": "mkv"}]}], + }) + + # ── shows / seasons / episodes ───────────────────────────────────── + ep_rk = 50000 + for s in range(6): + show_rk = str(900 + s) + shows.append({ + "sectionKey": "3", "ratingKey": show_rk, "guid": "plex://show/%04d" % s, + "type": "show", "title": "Show %d" % s, + }) + for se in range(1, rnd.randint(2, 4)): + season_rk = str(9000 + s * 10 + se) + for ep in range(1, rnd.randint(6, 13)): + episodes.append({ + "sectionKey": "3", "ratingKey": str(ep_rk), "guid": "plex://episode/%d" % ep_rk, + "type": "episode", "title": "Episode %d" % ep, "index": ep, + "grandparentRatingKey": show_rk, "grandparentTitle": "Show %d" % s, + "parentRatingKey": season_rk, "parentIndex": se, + "addedAt": NOW - rnd.randint(10, 2500) * DAY, + "duration": rnd.randint(20, 55) * 60000, + "viewCount": rnd.choice([0, 0, 1]), + "Media": [{"videoResolution": "1080", "videoCodec": "h264", + "Part": [{"id": 60000 + ep_rk, + "file": "/mnt/titan4/TVShows/Show %d/S%02d/E%02d.mkv" % (s, se, ep), + "size": rnd.randint(1, 5) * 10**9}]}], + }) + ep_rk += 1 + + # ── Tautulli history ─────────────────────────────────────────────── + watchable = [m["ratingKey"] for m in movies] + [e["ratingKey"] for e in episodes] + row_id = 1 + for _ in range(1200): + rk = rnd.choice(watchable) + pc = rnd.choice([2, 6, 11, 30, 55, 70, 88, 95, 99, 100, 100]) + started = NOW - rnd.randint(1, COVERAGE_DAYS) * DAY - rnd.randint(0, 80000) + history.append({ + "row_id": row_id, "reference_id": row_id // 2, + "date": started, "started": started, + "stopped": started + rnd.randint(300, 7200), + "rating_key": rk, + "user_id": rnd.choice([1, 2, 3, 4, 5]), + "user": "user%d" % rnd.choice([1, 2, 3]), + "friendly_name": rnd.choice(["Jess", "Sam", "Alex", "Guest"]), + "media_type": "movie" if rk.startswith(("1", "2")) else "episode", + "percent_complete": pc, + "watched_status": 1 if pc >= 85 else (0.5 if pc >= 15 else 0), + "play_duration": rnd.randint(120, 7000), "paused_counter": 0, + "platform": "Chrome", + }) + row_id += 1 + # guarantee the oldest event is exactly COVERAGE_DAYS old, so pre_history is testable + history[0]["date"] = history[0]["started"] = NOW - COVERAGE_DAYS * DAY + history.sort(key=lambda h: h["date"]) + return movies, episodes, shows, history -def page(items, q): - start = int(q.get("X-Plex-Container-Start", ["0"])[0]) - size = int(q.get("X-Plex-Container-Size", [str(len(items))])[0]) +MOVIES, EPISODES, SHOWS, HISTORY = _build() + +SECTIONS = [ + {"key": "1", "type": "movie", "title": "Movies", + "Location": [{"path": "/mnt/titan4/Movies"}]}, + {"key": "2", "type": "movie", "title": "4K Movies", + "Location": [{"path": "/mnt/titan4/4K Movies"}]}, + {"key": "3", "type": "show", "title": "TV Shows", + "Location": [{"path": "/mnt/titan4/TVShows"}]}, + {"key": "4", "type": "artist", "title": "Music", + "Location": [{"path": "/mnt/titan4/Music"}]}, +] + + +def _page(items, headers, query): + start = int(headers.get("X-Plex-Container-Start") + or query.get("X-Plex-Container-Start", ["0"])[0]) + size = int(headers.get("X-Plex-Container-Size") + or query.get("X-Plex-Container-Size", [str(len(items))])[0]) return items[start:start + size], len(items) -class H(BaseHTTPRequestHandler): +class Handler(BaseHTTPRequestHandler): + protocol_version = "HTTP/1.1" + def log_message(self, *a): pass - def _send(self, obj): + def _send(self, obj, code=200): body = json.dumps(obj).encode() - self.send_response(200) + self.send_response(code) self.send_header("Content-Type", "application/json") self.send_header("Content-Length", str(len(body))) self.end_headers() self.wfile.write(body) - def do_GET(self): + def do_GET(self): # noqa: N802 u = urllib.parse.urlsplit(self.path) q = urllib.parse.parse_qs(u.query) + # ── Plex ─────────────────────────────────────────────────────── if u.path == "/identity": - return self._send({"MediaContainer": {"machineIdentifier": "abc123def456", - "version": "1.41.0.1234"}}) + return self._send({"MediaContainer": { + "machineIdentifier": MACHINE_ID, "version": "1.43.3.10861", + "friendlyName": "Loki"}}) + if u.path == "/library/sections": - return self._send({"MediaContainer": {"Directory": [ - {"key": "1", "type": "movie", "title": "Movies", - "Location": [{"path": "/mnt/vault2/movies"}]}, - {"key": "2", "type": "show", "title": "TV Shows", - "Location": [{"path": "/mnt/vault2/tv"}]}, - {"key": "3", "type": "artist", "title": "Music", - "Location": [{"path": "/mnt/vault2/music"}]}, - ]}}) + return self._send({"MediaContainer": {"Directory": SECTIONS}}) + + if u.path == "/accounts": + return self._send({"MediaContainer": {"Account": [ + {"id": 1, "name": "jess"}, {"id": 2, "name": "sam"}]}}) + if u.path.startswith("/library/sections/") and u.path.endswith("/all"): key = u.path.split("/")[3] - src = MOVIES if key == "1" else EPISODES - batch, total = page(src, q) - return self._send({"MediaContainer": {"Metadata": batch, "totalSize": total, - "size": len(batch)}}) + libtype = q.get("type", ["1"])[0] + if libtype == "1": + src = [m for m in MOVIES if m["sectionKey"] == key] + elif libtype == "4": + src = [e for e in EPISODES if e["sectionKey"] == key] + elif libtype == "2": + src = [s for s in SHOWS if s["sectionKey"] == key] + else: + src = [] + batch, total = _page(src, self.headers, q) + return self._send({"MediaContainer": { + "Metadata": batch, "totalSize": total, "size": len(batch)}}) + + if u.path == "/status/sessions/history/all": + rows = [{"historyKey": "/status/sessions/history/%d" % h["row_id"], + "ratingKey": h["rating_key"], "viewedAt": h["date"], + "accountID": h["user_id"], "type": h["media_type"]} + for h in sorted(HISTORY, key=lambda x: -x["date"])] + batch, total = _page(rows, self.headers, q) + return self._send({"MediaContainer": { + "Metadata": batch, "totalSize": total, "size": len(batch)}}) + + # ── Tautulli ─────────────────────────────────────────────────── if u.path == "/api/v2": cmd = q.get("cmd", [""])[0] + if q.get("apikey", [""])[0] == "BADKEY": + return self._send({"response": {"result": "error", + "message": "Invalid apikey", "data": None}}) if cmd == "get_server_info": + ident = q.get("_force_id", [MACHINE_ID])[0] return self._send({"response": {"result": "success", "data": { - "pms_identifier": "abc123def456", "pms_name": "Loki", + "pms_identifier": ident, "pms_name": "Loki", "pms_ip": "192.168.1.10", "pms_port": 32400}}}) + if cmd == "get_users": + return self._send({"response": {"result": "success", "data": [ + {"user_id": i, "username": "user%d" % i, + "friendly_name": n} for i, n in + enumerate(["Jess", "Sam", "Alex", "Guest", "Kid"], start=1)]}}) if cmd == "get_history": - d = q.get("order_dir", ["desc"])[0] - rows = HISTORY if d == "asc" else list(reversed(HISTORY)) + direction = q.get("order_dir", ["desc"])[0] + rows = sorted(HISTORY, key=lambda h: h["date"], + reverse=(direction != "asc")) + start = int(q.get("start", ["0"])[0]) length = int(q.get("length", ["25"])[0]) return self._send({"response": {"result": "success", "data": { - "data": rows[:length], "recordsFiltered": len(HISTORY), - "recordsTotal": len(HISTORY)}}}) - if cmd == "get_users": - return self._send({"response": {"result": "success", "data": []}}) + "data": rows[start:start + length], + "recordsFiltered": len(rows), "recordsTotal": len(rows)}}}) + if cmd == "get_libraries": + return self._send({"response": {"result": "success", "data": [ + {"section_id": s["key"], "section_name": s["title"], + "section_type": s["type"]} for s in SECTIONS]}}) + if cmd == "get_library_media_info": + # Show sections really do return zero sizes here (§4.11). + key = q.get("section_id", [""])[0] + sec = next((s for s in SECTIONS if s["key"] == key), None) + if sec and sec["type"] == "show": + rows = [{"rating_key": s["ratingKey"], "title": s["title"], + "media_type": "show", "file_size": 0, "play_count": 0} + for s in SHOWS] + else: + rows = [{"rating_key": m["ratingKey"], "title": m["title"], + "media_type": "movie", + "file_size": sum(p["size"] for md in m["Media"] for p in md["Part"]), + "added_at": m["addedAt"], "play_count": m.get("viewCount", 0)} + for m in MOVIES if m["sectionKey"] == key] + return self._send({"response": {"result": "success", "data": { + "data": rows, "recordsFiltered": len(rows), "recordsTotal": len(rows)}}}) return self._send({"response": {"result": "error", "message": "unknown cmd", "data": None}}) + self.send_response(404) + self.send_header("Content-Length", "0") self.end_headers() +class FakeServer: + """Context manager wrapping the handler on an ephemeral port.""" + + def __init__(self, port: int = 0): + self.httpd = ThreadingHTTPServer(("127.0.0.1", port), Handler) + self.httpd.daemon_threads = True + self.port = self.httpd.server_address[1] + self.url = "http://127.0.0.1:%d" % self.port + self._thread: threading.Thread | None = None + + def __enter__(self) -> "FakeServer": + self._thread = threading.Thread(target=self.httpd.serve_forever, daemon=True) + self._thread.start() + return self + + def __exit__(self, *exc): + self.httpd.shutdown() + self.httpd.server_close() + return False + + +def stats() -> dict: + return { + "movies": len(MOVIES), "episodes": len(EPISODES), "shows": len(SHOWS), + "history": len(HISTORY), + "total_bytes": sum(p["size"] for m in MOVIES for md in m["Media"] for p in md["Part"]) + + sum(p["size"] for e in EPISODES for md in e["Media"] for p in md["Part"]), + } + + if __name__ == "__main__": - HTTPServer(("127.0.0.1", 8899), H).serve_forever() + srv = FakeServer(8899) + print("mock Plex+Tautulli on %s" % srv.url) + print(json.dumps(stats(), indent=2)) + srv.httpd.serve_forever() diff --git a/wsgi.py b/wsgi.py new file mode 100644 index 0000000..0d0f00b --- /dev/null +++ b/wsgi.py @@ -0,0 +1,5 @@ +"""gunicorn entry point: gunicorn -w 2 -b 0.0.0.0:8080 wsgi:app""" + +from mediashelf.app import create_app + +app = create_app()