MediaShelf/mediashelf/queries.py
Jess Hallsworth e819548ff2
Stop trusting Plex's Date Added on its own
Jess spotted that dates looked like file dates rather than library-add dates.
He is right, and MediaShelf was not the culprit: it reproduces Plex's addedAt
exactly (verified 500/500 identical to the second). Plex's own field is what
follows the file — replace or re-encode one and Date Added resets while the
item, its ratingKey and its watch history all survive.

Measured on the live library, comparing addedAt against lastViewedAt where both
exist: 55 of 509 movies (10.8%) and 306 of 1,393 TV Show Archive items (22.0%)
were watched BEFORE they were "added" — 19% overall. 2001: A Space Odyssey
reports added 2026-07-31, last watched 2017-08-26.

That is not cosmetic. pre_history is derived from added_at, so an old item whose
file was replaced looks post-coverage and gets promoted into the CONFIDENT
reclaim pool, which is the one pool meant to be trustworthy.

A completed play proves the item already existed, so added_at is now
MIN(provider_added_at, first_watched_at). Plex's raw value is kept in
provider_added_at, added_at_source records which applied, and the item drawer
explains the substitution instead of quietly disagreeing with Plex. Unwatched
items keep Plex's value since nothing contradicts it. first_seen_at is also
recorded now and is authoritative for anything added from here on.

Plex's API has no better field; the true insert time is only in Plex's own
metadata_items.created_at on Loki.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01GVbG48GAXfCZatcmX123Ra
2026-09-10 17:35:48 +00:00

383 lines
16 KiB
Python

"""The item query: filters + rules + live score + paging + aggregates.
The score is computed in the SELECT rather than stored, because weights change on
every slider drag and storing it would mean rewriting thousands of rows per drag
(§6.4). At this row count the whole query runs in single-digit milliseconds.
"""
from __future__ import annotations
import json
import time
from . import rules as rules_mod
from . import scoring
def _score_ctx(db, cfg, has_completion_data: bool) -> scoring.ScoreContext:
maxsize = db.scalar(
"SELECT MAX(size_bytes) FROM media_item WHERE kind IN ('movie','season')"
) or 1
s = cfg.score
return scoring.ScoreContext(
now=int(time.time()),
max_size_bytes=maxsize,
stale_horizon_days=s.stale_horizon_days,
age_horizon_days=s.age_horizon_days,
popular_at=s.popular_at,
rejected_at=s.rejected_at,
solitude_at=s.solitude_at,
grace_days=s.grace_days,
recent_days=s.recent_days,
has_completion_data=has_completion_data,
)
def history_has_completion(db) -> bool:
"""True when the ingested history actually carries percent_complete (§4.11).
Read from the events themselves, not from history_coverage: that row is only
written at the END of an ingest, so during the first scan the coverage table
is empty while watch_event is already full of Tautulli rows. Deriving the
flag from the table made the UI announce a Plex fallback that had not
happened, mid-scan, on every fresh database.
"""
return bool(db.scalar(
"SELECT 1 FROM watch_event WHERE percent_complete IS NOT NULL LIMIT 1"))
def active_history_source(db) -> str | None:
"""Which source history came from — coverage if it exists, else the scan."""
row = db.one("SELECT source FROM history_coverage ORDER BY event_count DESC LIMIT 1")
if row:
return row["source"]
row = db.one("SELECT history_source FROM scan WHERE history_source IS NOT NULL "
"ORDER BY id DESC LIMIT 1")
return row["history_source"] if row else None
BASE_COLUMNS = """
i.id, i.kind, i.title, i.sort_title, i.year, i.guid, i.show_guid,
i.library_id, lib.title AS library_title,
i.size_bytes, i.added_at, i.updated_at, i.duration_ms,
i.episode_count, i.part_count, i.primary_path, i.resolution, i.video_codec,
i.watch_count, i.partial_count, i.abandoned_count, i.avg_percent_complete,
i.last_watched_at, i.last_touched_at, i.first_watched_at,
i.distinct_watcher_count, i.pre_history, i.kept, i.kept_via, i.kept_mark_id,
i.provider_view_count, i.status, i.parent_id, i.season_number,
i.provider_added_at, i.added_at_source, i.first_seen_at,
parent.title AS show_title
"""
FROM_CLAUSE = """
FROM media_item i
JOIN library lib ON lib.id = i.library_id
LEFT JOIN media_item parent ON parent.id = i.parent_id
LEFT JOIN (
SELECT guid, COUNT(*) AS dupe_count
FROM media_item WHERE kind = 'movie' AND guid IS NOT NULL AND status = 'present'
GROUP BY guid HAVING COUNT(*) > 1
) d ON d.guid = i.guid
CROSS JOIN (SELECT MAX(size_bytes) AS maxsize FROM media_item
WHERE kind IN ('movie','season')) s
"""
class Query:
def __init__(self, db, cfg):
self.db = db
self.cfg = cfg
self.has_cd = history_has_completion(db)
self.ctx = _score_ctx(db, cfg, self.has_cd)
def build(self, *, library_ids=None, kinds=None, q=None, rule_group=None,
include_missing=False, include_kept=False, include_shows=False,
weights=None, extra_where=None):
where = []
params: dict = {}
if not include_missing:
where.append("i.status = 'present'")
if not include_kept:
where.append("i.kept = 0")
if not include_shows:
# Shows are containers; seasons are the unit of analysis for TV (§5.3)
where.append("i.kind != 'show'")
if kinds:
keys = []
for n, k in enumerate(kinds):
key = "k%d" % n
params[key] = k
keys.append(":" + key)
where.append("i.kind IN (%s)" % ", ".join(keys))
if library_ids:
keys = []
for n, lid in enumerate(library_ids):
key = "lib%d" % n
params[key] = int(lid)
keys.append(":" + key)
where.append("i.library_id IN (%s)" % ", ".join(keys))
if q:
params["q"] = _fts_query(q)
where.append(
"i.id IN (SELECT rowid FROM media_item_fts WHERE media_item_fts MATCH :q)"
)
if rule_group:
frag, rp = rules_mod.compile_rules(rule_group, self.ctx.now)
if frag:
where.append(frag)
params.update(rp)
if extra_where:
where.append(extra_where)
expr, sp = scoring.sql_expression(self.ctx, weights)
params.update(sp)
params["now"] = self.ctx.now
where_sql = " AND ".join(where) if where else "1=1"
return expr, where_sql, params
def page(self, *, sort=None, page=1, page_size=100, **kw):
expr, where_sql, params = self.build(**kw)
order = rules_mod.compile_sort(sort)
total = self.db.scalar(
"SELECT COUNT(*) " + FROM_CLAUSE + " WHERE " + where_sql, params
) or 0
agg = self.db.one(
"SELECT COALESCE(SUM(i.size_bytes),0) AS total_size, COUNT(*) AS n "
+ FROM_CLAUSE + " WHERE " + where_sql, params
)
page = max(1, int(page))
page_size = max(1, min(int(page_size), 500))
params["_limit"] = page_size
params["_offset"] = (page - 1) * page_size
sql = (
"SELECT " + BASE_COLUMNS + ", " + expr + " AS reclaim_score, "
+ scoring.GRACE_SQL + " AS grace, COALESCE(d.dupe_count, 1) AS duplicate_count "
+ FROM_CLAUSE + " WHERE " + where_sql
+ " ORDER BY " + order + " LIMIT :_limit OFFSET :_offset"
)
rows = self.db.query(sql, params)
return {
"total": total,
"page": page,
"page_size": page_size,
"aggregate": {
"total_size_bytes": agg["total_size"] or 0,
"item_count": agg["n"] or 0,
},
"items": [self.serialize(r, weights=kw.get("weights")) for r in rows],
}
def iter_all(self, *, sort=None, **kw):
"""Stream every matching row, for CSV export."""
expr, where_sql, params = self.build(**kw)
order = rules_mod.compile_sort(sort)
sql = (
"SELECT " + BASE_COLUMNS + ", " + expr + " AS reclaim_score, "
+ scoring.GRACE_SQL + " AS grace, COALESCE(d.dupe_count, 1) AS duplicate_count "
+ FROM_CLAUSE + " WHERE " + where_sql + " ORDER BY " + order
)
for row in self.db.conn.execute(sql, params):
yield row
def serialize(self, row, weights=None) -> dict:
d = dict(row)
comps = scoring.components(d, self.ctx)
flags = []
if d.get("pre_history"):
flags.append("pre_history")
if (d.get("abandoned_count") or 0) >= self.ctx.rejected_at and not d.get("watch_count"):
flags.append("rejected")
if (d.get("duplicate_count") or 1) > 1:
flags.append("duplicate")
if (d.get("provider_view_count") or 0) > 0 and not d.get("watch_count"):
flags.append("history_gap")
if d.get("added_at_source") == "first_watch":
flags.append("added_corrected")
# A season has one part per episode, so part_count > 1 is normal there;
# only flag genuinely extra files (a movie held twice, or a split episode).
parts = d.get("part_count") or 0
if (d["kind"] == "movie" and parts > 1) or \
(d["kind"] == "season" and parts > (d.get("episode_count") or 0)):
flags.append("multi_part")
return {
"id": d["id"],
"kind": d["kind"],
"title": d["title"],
"show_title": d.get("show_title"),
"season_number": d.get("season_number"),
"year": d.get("year"),
"guid": d.get("guid"),
"library": {"id": d["library_id"], "title": d["library_title"]},
"size_bytes": d.get("size_bytes") or 0,
"added_at": d.get("added_at"),
"provider_added_at": d.get("provider_added_at"),
"added_at_source": d.get("added_at_source"),
"first_seen_at": d.get("first_seen_at"),
"last_watched_at": d.get("last_watched_at"),
"last_touched_at": d.get("last_touched_at"),
"watch_count": d.get("watch_count") or 0,
"partial_count": d.get("partial_count") or 0,
"abandoned_count": d.get("abandoned_count") or 0,
"avg_percent_complete": d.get("avg_percent_complete"),
"distinct_watcher_count": d.get("distinct_watcher_count") or 0,
"episode_count": d.get("episode_count") if d["kind"] == "season" else None,
"primary_path": d.get("primary_path"),
"part_count": d.get("part_count") or 0,
"resolution": d.get("resolution"),
"duplicate_count": d.get("duplicate_count") or 1,
"pre_history": bool(d.get("pre_history")),
"kept": bool(d.get("kept")),
"kept_via": d.get("kept_via"),
"status": d.get("status"),
"reclaim_score": d.get("reclaim_score"),
"reclaim_components": {k: (None if v is None else round(v, 4))
for k, v in comps.items()},
"grace": d.get("grace"),
"flags": flags,
}
def _fts_query(q: str) -> str:
"""Turn user text into a safe FTS5 prefix query.
FTS5 has its own operator syntax; passing raw user input straight through
lets a stray quote or NEAR() produce an error or a surprising match, so each
token is quoted and turned into a prefix term.
"""
tokens = [t for t in "".join(c if c.isalnum() else " " for c in q).split() if t]
if not tokens:
return '""'
return " ".join('"%s"*' % t for t in tokens[:10])
# ── aggregate stats for the dashboard ────────────────────────────────────
def overview(db, cfg) -> dict:
def s(sql, params=()):
return db.scalar(sql, params) or 0
unit = "kind IN ('movie','season') AND status='present'"
total_bytes = s(f"SELECT SUM(size_bytes) FROM media_item WHERE {unit}")
total_items = s(f"SELECT COUNT(*) FROM media_item WHERE {unit}")
never_bytes = s(f"SELECT SUM(size_bytes) FROM media_item WHERE {unit} AND watch_count=0")
kept_bytes = s(f"SELECT SUM(size_bytes) FROM media_item WHERE {unit} AND kept=1")
# The three-way split that keeps the keep list honest (§6.6)
never_kept = s(f"SELECT SUM(size_bytes) FROM media_item WHERE {unit} AND watch_count=0 AND kept=1")
confident = s(f"SELECT SUM(size_bytes) FROM media_item WHERE {unit} "
"AND watch_count=0 AND kept=0 AND pre_history=0")
uncertain = s(f"SELECT SUM(size_bytes) FROM media_item WHERE {unit} "
"AND watch_count=0 AND kept=0 AND pre_history=1")
cold = s(f"SELECT SUM(size_bytes) FROM media_item WHERE {unit} AND kept=0 AND "
"(last_watched_at IS NULL OR last_watched_at < strftime('%s','now') - 2*365*86400)")
cov = db.one("SELECT * FROM history_coverage ORDER BY event_count DESC LIMIT 1")
last_scan = db.one("SELECT * FROM scan WHERE status='succeeded' ORDER BY id DESC LIMIT 1")
return {
"total_bytes": total_bytes,
"total_items": total_items,
"never_played_bytes": never_bytes,
"kept_bytes": kept_bytes,
"never_played_kept_bytes": never_kept,
"available_bytes": max(never_bytes - never_kept, 0),
"confident_bytes": confident,
"uncertain_bytes": uncertain,
"cold_bytes": cold,
"libraries": s("SELECT COUNT(*) FROM library"),
"episodes": s("SELECT COUNT(*) FROM episode WHERE status='present'"),
"watch_events": s("SELECT COUNT(*) FROM watch_event"),
"accounts": s("SELECT COUNT(*) FROM account"),
"history_source": active_history_source(db),
"history_since": cov["earliest_event_at"] if cov else None,
"history_until": cov["latest_event_at"] if cov else None,
"has_completion_data": history_has_completion(db),
"last_scan_at": last_scan["finished_at"] if last_scan else None,
"keep_marks": s("SELECT COUNT(*) FROM keep_mark"),
}
def size_by_library(db) -> list[dict]:
return [dict(r) for r in db.query("""
SELECT lib.id, lib.title, lib.kind, lib.keep_all,
COUNT(*) AS items,
COALESCE(SUM(i.size_bytes),0) AS size_bytes,
COALESCE(SUM(CASE WHEN i.watch_count=0 THEN i.size_bytes ELSE 0 END),0) AS never_bytes,
COALESCE(SUM(CASE WHEN i.kept=1 THEN i.size_bytes ELSE 0 END),0) AS kept_bytes
FROM media_item i JOIN library lib ON lib.id = i.library_id
WHERE i.kind IN ('movie','season') AND i.status='present'
GROUP BY lib.id ORDER BY size_bytes DESC
""")]
def added_over_time(db, bucket="month") -> list[dict]:
fmt = "%Y-%m" if bucket == "month" else "%Y"
return [dict(r) for r in db.query(f"""
SELECT strftime('{fmt}', added_at, 'unixepoch') AS period,
COUNT(*) AS items, COALESCE(SUM(size_bytes),0) AS size_bytes
FROM media_item
WHERE kind IN ('movie','season') AND status='present'
AND added_at IS NOT NULL AND added_at > 0
GROUP BY period ORDER BY period
""")]
def completion_split(db) -> dict:
"""finished / abandoned / never-opened by size — invisible without Tautulli."""
row = db.one("""
SELECT
COALESCE(SUM(CASE WHEN watch_count > 0 THEN size_bytes ELSE 0 END),0) AS finished,
COALESCE(SUM(CASE WHEN watch_count = 0 AND (abandoned_count > 0 OR partial_count > 0)
THEN size_bytes ELSE 0 END),0) AS started,
COALESCE(SUM(CASE WHEN watch_count = 0 AND abandoned_count = 0 AND partial_count = 0
THEN size_bytes ELSE 0 END),0) AS never
FROM media_item WHERE kind IN ('movie','season') AND status='present'
""")
return dict(row)
def size_vs_lastwatched(db, limit=3000) -> list[dict]:
return [dict(r) for r in db.query("""
SELECT id, title, kind, size_bytes, last_watched_at, added_at, kept,
watch_count, pre_history
FROM media_item
WHERE kind IN ('movie','season') AND status='present' AND size_bytes > 0
ORDER BY size_bytes DESC LIMIT ?
""", (limit,))]
def duplicate_groups(db) -> list[dict]:
"""Same content held more than once — usually Movies vs 4K Movies (§6.5)."""
rows = db.query("""
SELECT i.guid, i.id, i.title, i.year, i.size_bytes, i.resolution,
i.watch_count, i.last_watched_at, i.kept, lib.title AS library_title
FROM media_item i JOIN library lib ON lib.id = i.library_id
WHERE i.kind='movie' AND i.status='present' AND i.guid IS NOT NULL
AND i.guid IN (SELECT guid FROM media_item WHERE kind='movie'
AND status='present' AND guid IS NOT NULL
GROUP BY guid HAVING COUNT(*) > 1)
ORDER BY i.guid, i.size_bytes DESC
""")
groups: dict[str, dict] = {}
for r in rows:
g = groups.setdefault(r["guid"], {"guid": r["guid"], "title": r["title"],
"year": r["year"], "copies": []})
g["copies"].append(dict(r))
out = []
for g in groups.values():
sizes = [c["size_bytes"] or 0 for c in g["copies"]]
g["total_bytes"] = sum(sizes)
g["redundant_bytes"] = sum(sizes) - max(sizes) if sizes else 0
out.append(g)
out.sort(key=lambda g: -g["redundant_bytes"])
return out