#!/usr/bin/env python3 """ MediaShelf probe — read-only reconnaissance against Plex and Tautulli. Run this on the LAN (anywhere that can reach both servers). It answers the open questions in docs/design.md against real data, so the design can be validated without exposing either service outside the network. It is READ-ONLY. It issues GET requests only. It never deletes, modifies, or writes anything to either server, and it never prints your credentials. Standard library only — no pip install needed. Python 3.8+. Usage: export PLEX_BASE_URL=http://192.168.1.10:32400 export PLEX_TOKEN=... export TAUTULLI_BASE_URL=http://192.168.1.100:8181 export TAUTULLI_API_KEY=... python3 probe.py # or, to also write the full item inventory out for offline analysis: python3 probe.py --dump inventory.json Everything it prints is safe to share. """ import argparse import json import os import sys import time import urllib.error import urllib.parse import urllib.request from collections import Counter, defaultdict TIMEOUT = 30 PAGE = 500 HIST_PAGE = 1000 # ─────────────────────────────── plumbing ──────────────────────────────── class Fail(Exception): pass def _get(url, headers=None): req = urllib.request.Request(url, headers=headers or {}, method="GET") try: with urllib.request.urlopen(req, timeout=TIMEOUT) as r: return json.loads(r.read().decode("utf-8", "replace")) except urllib.error.HTTPError as e: raise Fail("HTTP %s from %s" % (e.code, _safe(url))) except urllib.error.URLError as e: raise Fail("cannot reach %s (%s)" % (_safe(url), e.reason)) except json.JSONDecodeError: raise Fail("non-JSON response from %s" % _safe(url)) def _safe(url): """Strip credentials out of a URL before it appears in any output.""" parts = urllib.parse.urlsplit(url) q = urllib.parse.parse_qsl(parts.query) q = [(k, "***" if k.lower() in ("apikey", "x-plex-token") else v) for k, v in q] return urllib.parse.urlunsplit( (parts.scheme, parts.netloc, parts.path, urllib.parse.urlencode(q), "") ) def human(n): if n is None: return "?" for unit in ("B", "KB", "MB", "GB", "TB", "PB"): if abs(n) < 1024: return "%.1f %s" % (n, unit) n /= 1024.0 return "%.1f EB" % n def ymd(ts): if not ts: return "—" return time.strftime("%Y-%m-%d", time.localtime(int(ts))) def days_since(ts): if not ts: return None return int((time.time() - int(ts)) / 86400) # ──────────────────────────────── plex ─────────────────────────────────── class Plex: def __init__(self, base, token): self.base = base.rstrip("/") self.headers = { "X-Plex-Token": token, "Accept": "application/json", "X-Plex-Product": "MediaShelf-probe", "X-Plex-Client-Identifier": "mediashelf-probe", } def get(self, path, **params): url = self.base + path if params: url += "?" + urllib.parse.urlencode(params) return _get(url, self.headers).get("MediaContainer", {}) def identity(self): return self.get("/identity") def sections(self): return self.get("/library/sections").get("Directory", []) def section_detail(self, key): return self.get("/library/sections/" + str(key)) def items(self, key, libtype): """Page through a section, yielding raw Video/Directory dicts.""" start = 0 while True: mc = self.get( "/library/sections/%s/all" % key, type=libtype, **{ "X-Plex-Container-Start": start, "X-Plex-Container-Size": PAGE, } ) batch = mc.get("Metadata", []) or [] if not batch: return for row in batch: yield row start += len(batch) total = mc.get("totalSize") or mc.get("size") or 0 if start >= int(total): return def item_size(row): total = 0 parts = 0 for media in row.get("Media", []) or []: for part in media.get("Part", []) or []: total += int(part.get("size") or 0) parts += 1 return total, parts def item_paths(row): out = [] for media in row.get("Media", []) or []: for part in media.get("Part", []) or []: if part.get("file"): out.append(part["file"]) return out # ────────────────────────────── tautulli ───────────────────────────────── class Tautulli: def __init__(self, base, key): self.base = base.rstrip("/") + "/api/v2" self.key = key def cmd(self, command, **params): params["apikey"] = self.key params["cmd"] = command url = self.base + "?" + urllib.parse.urlencode(params) body = _get(url) resp = body.get("response") or {} if resp.get("result") != "success": raise Fail( "Tautulli cmd=%s returned %s: %s" % (command, resp.get("result"), resp.get("message")) ) return resp.get("data") def server_info(self): return self.cmd("get_server_info") or {} def users(self): return self.cmd("get_users") or [] def history_edge(self, direction): """Oldest (asc) or newest (desc) single history row.""" d = self.cmd( "get_history", order_column="date", order_dir=direction, grouping=0, start=0, length=1, ) or {} rows = d.get("data") or [] return (rows[0] if rows else None), d.get("recordsFiltered") or d.get("recordsTotal") or 0 def history_sample(self, length=2000): d = self.cmd( "get_history", order_column="date", order_dir="desc", grouping=0, start=0, length=length, ) or {} return d.get("data") or [] # ───────────────────────────────── report ──────────────────────────────── def section(title): print("\n" + title) print("─" * len(title)) def main(): ap = argparse.ArgumentParser(description="Read-only Plex + Tautulli probe for MediaShelf.") ap.add_argument("--dump", metavar="FILE", help="write the full item inventory to FILE as JSON") ap.add_argument("--history-sample", type=int, default=2000, help="how many recent history rows to sample (default 2000)") args = ap.parse_args() plex_url = os.environ.get("PLEX_BASE_URL") plex_token = os.environ.get("PLEX_TOKEN") taut_url = os.environ.get("TAUTULLI_BASE_URL") taut_key = os.environ.get("TAUTULLI_API_KEY") if not plex_url or not plex_token: sys.exit("PLEX_BASE_URL and PLEX_TOKEN must be set.") print("MediaShelf probe — read-only. No data is modified.") print("Run at %s" % time.strftime("%Y-%m-%d %H:%M:%S")) plex = Plex(plex_url, plex_token) taut = Tautulli(taut_url, taut_key) if (taut_url and taut_key) else None # ── identity and the cross-check ─────────────────────────────────── section("Servers") try: ident = plex.identity() pms_id = ident.get("machineIdentifier") print("Plex : version %s, machineIdentifier %s…" % (ident.get("version"), (pms_id or "")[:8])) except Fail as e: sys.exit("Plex unreachable: %s" % e) taut_ok = False if taut: try: si = taut.server_info() taut_ok = True match = si.get("pms_identifier") == pms_id print("Tautulli : watching %s (%s:%s)" % (si.get("pms_name"), si.get("pms_ip"), si.get("pms_port"))) print(" pms_identifier match: %s" % ("YES" if match else "NO — Tautulli is watching a DIFFERENT Plex server")) if not match: print(" ** MediaShelf would refuse to join this history data. **") except Fail as e: print("Tautulli : UNREACHABLE — %s" % e) else: print("Tautulli : not configured (set TAUTULLI_BASE_URL and TAUTULLI_API_KEY)") # ── question 1: coverage horizon ─────────────────────────────────── coverage_since = None if taut_ok: section("Q1 · Tautulli coverage horizon") try: oldest, total = taut.history_edge("asc") newest, _ = taut.history_edge("desc") if oldest: coverage_since = int(oldest.get("date") or 0) print("Oldest logged play : %s (%s days ago)" % (ymd(coverage_since), days_since(coverage_since))) print("Newest logged play : %s" % ymd(newest.get("date") if newest else None)) print("Total history rows : %s" % f"{int(total):,}") print("\n-> Anything added before %s that was watched before that date" % ymd(coverage_since)) print(" will look never-watched. MediaShelf flags these 'pre_history'.") else: print("No history rows found — Tautulli has logged nothing yet.") except Fail as e: print("Could not read history: %s" % e) # ── question 2: grouping sanity + user spread ────────────────────── if taut_ok: section("Q2 · History shape (sample)") try: rows = taut.history_sample(args.history_sample) print("Sampled %s most recent rows." % f"{len(rows):,}") if rows: users = Counter(r.get("friendly_name") or r.get("user") for r in rows) print("\nDistinct users in sample: %d" % len(users)) for name, n in users.most_common(10): print(" %-24s %6d plays" % (str(name)[:24], n)) pcs = [int(r.get("percent_complete") or 0) for r in rows] finished = sum(1 for p in pcs if p >= 85) abandoned = sum(1 for p in pcs if p < 15) partial = len(pcs) - finished - abandoned print("\nCompletion breakdown of sampled plays:") print(" finished (>=85%%) : %6d (%.1f%%)" % (finished, 100.0 * finished / len(pcs))) print(" partial (15-85%%): %6d (%.1f%%)" % (partial, 100.0 * partial / len(pcs))) print(" abandoned (<15%%) : %6d (%.1f%%)" % (abandoned, 100.0 * abandoned / len(pcs))) print("\n-> The 'abandoned' share is the signal Plex cannot provide.") refs = Counter(r.get("reference_id") for r in rows if r.get("reference_id")) dupes = sum(1 for c in refs.values() if c > 1) print("\nRows sharing a reference_id: %d groups" % dupes) print("-> Non-zero is expected and fine: it means successive plays exist") print(" and MediaShelf's own session merging has real work to do.") except Fail as e: print("Could not sample history: %s" % e) # ── questions 3 & 4: library shape and paths ─────────────────────── section("Q3/Q4 · Libraries, sizes and path roots") try: secs = plex.sections() except Fail as e: sys.exit("Could not list sections: %s" % e) inventory = [] grand_total = 0 grand_count = 0 root_bytes = defaultdict(int) for s in secs: stype = s.get("type") if stype not in ("movie", "show"): print("\n%-24s %-8s (skipped — not movie/show)" % (s.get("title"), stype)) continue key = s.get("key") locs = [d.get("path") for d in (s.get("Location") or [])] libtype = 1 if stype == "movie" else 4 # movie | episode n = 0 size = 0 parts = 0 oldest_add = None multi_version = 0 shows = set() seasons = set() try: for row in plex.items(key, libtype): sz, pc = item_size(row) n += 1 size += sz parts += pc if pc > 1: multi_version += 1 added = int(row.get("addedAt") or 0) if added and (oldest_add is None or added < oldest_add): oldest_add = added if stype == "show": shows.add(row.get("grandparentRatingKey")) seasons.add(row.get("parentRatingKey")) for p in item_paths(row): # bucket by the first two path components, e.g. /mnt/vault2 bits = p.split("/") root = "/".join(bits[:3]) if len(bits) > 3 else p root_bytes[root] += sz if pc == 1 else sz // max(pc, 1) if args.dump is not None: inventory.append({ "section": s.get("title"), "type": stype, "ratingKey": row.get("ratingKey"), "title": row.get("title"), "grandparentTitle": row.get("grandparentTitle"), "parentIndex": row.get("parentIndex"), "year": row.get("year"), "addedAt": added, "viewCount": int(row.get("viewCount") or 0), "lastViewedAt": int(row.get("lastViewedAt") or 0) or None, "size": sz, "parts": pc, "paths": item_paths(row), }) except Fail as e: print("\n%-24s ERROR: %s" % (s.get("title"), e)) continue grand_total += size grand_count += n print("\n%s [%s]" % (s.get("title"), stype)) if stype == "movie": print(" movies : %s" % f"{n:,}") else: print(" episodes : %s" % f"{n:,}") print(" seasons : %s (MediaShelf's unit of analysis)" % f"{len(seasons):,}") print(" shows : %s" % f"{len(shows):,}") print(" total size : %s" % human(size)) print(" files (parts) : %s" % f"{parts:,}") if multi_version: print(" multi-file items : %s <- these are the ones a naive scan undercounts" % f"{multi_version:,}") print(" oldest addedAt : %s" % ymd(oldest_add)) for loc in locs: print(" root : %s" % loc) section("Totals") print("Items scanned : %s" % f"{grand_count:,}") print("Total size : %s" % human(grand_total)) if root_bytes: print("\nApproximate size by path root:") for root, b in sorted(root_bytes.items(), key=lambda kv: -kv[1]): print(" %-40s %s" % (root[:40], human(b))) print("\n-> If these map to separate arrays, MediaShelf can group by vault,") print(" which matters when the goal is freeing one specific array.") # ── a first taste of the actual output ───────────────────────────── if args.dump is not None and inventory: with open(args.dump, "w") as fh: json.dump(inventory, fh) print("\nWrote %s items to %s" % (f"{len(inventory):,}", args.dump)) section("Preview · 20 largest items Plex thinks were never watched") print("(Plex's viewCount is token-scoped, so treat this as indicative only —") print(" the real report uses Tautulli. This is a sanity check on sizes.)\n") never = [i for i in inventory if not i["viewCount"]] never.sort(key=lambda i: -i["size"]) for i in never[:20]: label = i["title"] if i.get("grandparentTitle"): label = "%s — S%s %s" % (i["grandparentTitle"], i.get("parentIndex"), i["title"]) print(" %10s added %s %s" % (human(i["size"]), ymd(i["addedAt"]), label[:60])) reclaimable = sum(i["size"] for i in never) print("\n %s across %s items Plex records no views for." % (human(reclaimable), f"{len(never):,}")) section("Done") print("Nothing was modified. Paste this output back to review the design against it.") if coverage_since: print("Key number: Tautulli coverage begins %s." % ymd(coverage_since)) if __name__ == "__main__": try: main() except KeyboardInterrupt: sys.exit("\nInterrupted.") except Fail as e: sys.exit("Failed: %s" % e)