"""Token usage and API-price cost from Claude Code transcripts (jsonl), per model. One API request is streamed as several transcript entries (one per content block) that repeat its usage: count each (message.id, requestId) once, taking its last entry (final output_tokens). Subagent transcripts never get a final entry (stop_reason null, output_tokens = the message_start placeholder): there output is estimated from the content (out_estimate), counted in Usage.est. """ from __future__ import annotations import json import re from dataclasses import dataclass, fields from statistics import median # $ per million tokens: input, output, cache read. Cache writes: 1.25x input (5 min), 2x input (1 h). # Source: Anthropic pricing as of 2026-09-25. Longest matching prefix of the model id wins. PRICES = { "claude-fable-5-1": (10.0, 50.0, 0.25), "claude-fable-5": (10.0, 50.0, 1.0), "claude-opus-5-5": (4.0, 20.0, 0.20), "claude-opus-5": (5.0, 25.0, 0.50), "claude-opus-4-8": (5.0, 25.0, 0.50), "claude-sonnet-5-5": (2.0, 10.0, 0.20), "claude-sonnet-5": (2.0, 10.0, 0.20), "claude-sonnet-4-6": (3.0, 15.0, 0.30), "claude-haiku-4-5": (1.0, 5.0, 0.10), } @dataclass class Usage: turns: int = 0 inp: int = 0 cw5: int = 0 cw1h: int = 0 cr: int = 0 out: int = 0 est: int = 0 # requests whose out is estimated from content @property def cw(self) -> int: return self.cw5 + self.cw1h def add(self, other: "Usage") -> None: for f in fields(self): setattr(self, f.name, getattr(self, f.name) + getattr(other, f.name)) def price(model: str) -> tuple[float, float, float] | None: keys = [k for k in PRICES if model == k or model.startswith(k + "-")] return PRICES[max(keys, key=len)] if keys else None def cost(model: str, u: Usage) -> float | None: """API-price $ (subscription sessions: what it would cost on the API); None = unknown model.""" p = price(model) if p is None: return None inp, out, read = p return (u.inp * inp + u.cw5 * inp * 1.25 + u.cw1h * inp * 2 + u.cr * read + u.out * out) / 1e6 def _usage(raw: dict) -> Usage: cw = raw.get("cache_creation_input_tokens") or 0 cw1h = (raw.get("cache_creation") or {}).get("ephemeral_1h_input_tokens") or 0 return Usage(1, raw.get("input_tokens") or 0, cw - cw1h, cw1h, raw.get("cache_read_input_tokens") or 0, raw.get("output_tokens") or 0) # Output-token estimate per content block, fitted on ~100k main-session requests with final usage # (2026-10): per-session totals within ±5%. Thinking text is usually empty; its signature grows ~3.3 # chars per thinking token over a ~800-char base. EST_TEXT, EST_TOOL_CHAR, EST_TOOL, EST_SIG, EST_SIG_BASE = 0.3, 0.44, 30, 0.3, 800 def out_estimate(blocks: list) -> int: t = 0.0 for b in blocks: if not isinstance(b, dict): continue kind = b.get("type") if kind == "text": t += len(b.get("text") or "") * EST_TEXT elif kind == "tool_use": t += EST_TOOL + len(json.dumps(b.get("input") or {}, ensure_ascii=False)) * EST_TOOL_CHAR elif kind == "thinking": t += (len(b.get("thinking") or "") * EST_TEXT + max(0, len(b.get("signature") or "") - EST_SIG_BASE) * EST_SIG) return round(t) def parse(text: str, since: str | None = None, until: str | None = None) -> dict[str, Usage]: """Model id → summed Usage. since/until: ISO prefixes compared with the UTC timestamps.""" last: dict[tuple, tuple[str, dict]] = {} first_ts: dict[tuple, str] = {} final: set[tuple] = set() blocks: dict[tuple, list] = {} seen: set[str] = set() for line in text.split("\n"): try: d = json.loads(line) except ValueError: continue if not isinstance(d, dict) or d.get("type") != "assistant": continue m = d.get("message") or {} model, raw = m.get("model") or "?", m.get("usage") if not raw or model == "": continue key = (m.get("id"), d.get("requestId")) first_ts.setdefault(key, d.get("timestamp") or "") last[key] = (model, raw) if m.get("stop_reason"): final.add(key) uid = d.get("uuid") if uid not in seen and isinstance(m.get("content"), list): blocks.setdefault(key, []).extend(m["content"]) if uid: seen.add(uid) out: dict[str, Usage] = {} for key, (model, raw) in last.items(): ts = first_ts[key] if (since and ts < since) or (until and ts >= until): continue u = _usage(raw) if key not in final: guess = out_estimate(blocks.get(key, [])) if guess > u.out: u.out, u.est = guess, 1 out.setdefault(model, Usage()).add(u) return out def context_tokens(text: str) -> int | None: """Prompt size of the last main-thread request (input + cache write + cache read); None = no request.""" size = None for line in text.split("\n"): try: d = json.loads(line) except ValueError: continue if not isinstance(d, dict) or d.get("type") != "assistant" or d.get("isSidechain"): continue m = d.get("message") or {} raw = m.get("usage") if raw and m.get("model") != "": u = _usage(raw) size = u.inp + u.cw + u.cr return size def ctx_hint(tokens: int | None, limit: int) -> str | None: if not limit or tokens is None or tokens <= limit: return None return (f"context ~{tokens // 1000}k tokens (> {limit // 1000}k): ask the owner to /clear, then continue " "(subagent: ignore, this is the main session)") def short(model: str) -> str: return re.sub(r"-\d{8}$", "", model.removeprefix("claude-")) def fmt(n: int) -> str: if n < 1000: return str(n) return f"{n / 1e3:.1f}k" if n < 1e6 else f"{n / 1e6:.2f}M" # ------------------------------------------------------------------ cost log (out/wf-cost.log) def log_line(time: str, project: str, task: str, effort: str, outcome: str, agent: str, by_model: dict[str, Usage], lane: str | None = None, dur: int | None = None) -> str: """One key=value line per worker; model = the model with most turns, lane = given or model; usd = priced models only.""" total = Usage() for u in by_model.values(): total.add(u) model = short(max(by_model, key=lambda m: by_model[m].turns)).split("-")[0] if by_model else "?" usd = sum(cost(m, u) or 0 for m, u in by_model.items()) return (f"{time} project={project} task={task} lane={lane or model} model={model} effort={effort} outcome={outcome} " f"turns={total.turns} in={total.inp} cw={total.cw} cr={total.cr} out={total.out} " + (f"est={total.est} " if total.est else "") + f"usd={usd:.4f} " + (f"dur={dur} " if dur is not None else "") + f"agent={agent}") def parse_log(text: str, since: str | None = None) -> list[dict]: out = [] for line in text.split("\n"): time, _, rest = line.partition(" ") fields_ = dict(f.split("=", 1) for f in rest.split() if "=" in f) if not re.match(r"\d{4}-\d\d-\d\dT", time) or "lane" not in fields_ or (since and time < since): continue try: out.append({**fields_, "time": time, "usd": float(fields_.get("usd", 0)), "turns": int(fields_.get("turns", 0)), **({"dur": int(fields_["dur"])} if "dur" in fields_ else {})}) except ValueError: continue return out def _num(x: float) -> float | int: x = round(x, 4) return int(x) if x == int(x) and not isinstance(x, bool) else x def report(entries: list[dict], efforts: tuple[str, ...]) -> list[tuple]: """Per lane, then per lane+effort: (lane, effort, n, done, other, median $, total $, $/done, median turns, median dur s or None).""" def row(lane, effort, es): done = sum(e.get("outcome") in ("done", "done+gate-red") for e in es) total = sum(e["usd"] for e in es) return (lane, effort, len(es), done, len(es) - done, round(median(e["usd"] for e in es), 4), round(total, 4), round(total / done, 4) if done else None, _num(median(e["turns"] for e in es)), _num(median(durs)) if (durs := [e["dur"] for e in es if "dur" in e]) else None) order = {e: i for i, e in enumerate(efforts)} rows = [] key = lambda e: f"{e['lane']}/{e['model']}" if "model" in e and e["model"] != e["lane"] else e["lane"] for lane in sorted({key(e) for e in entries}): es = [e for e in entries if key(e) == lane] rows.append(row(lane, "all", es)) for effort in sorted({e.get("effort", "-") for e in es}, key=lambda x: (order.get(x, len(order)), x)): rows.append(row(lane, effort, [e for e in es if e.get("effort", "-") == effort])) return rows # --- explore: where an agent's billed input goes (exploring vs editing/running/bookkeeping) --------------------- _MUT = re.compile(r"sed -i|python3 - <<|cat > |tee |expect-init|>> [a-zA-Z]|apply_patch|patch ") _RUN = re.compile(r"dotnet (test|build|run)|own-fx\.py (check|measure|dump)|npx |npm |gate-bg|check\.sh|wf res run|pytest|unittest") _BOOK = re.compile(r"wf\.py (done|finish|add|status|note|set|merge|check|tick|body)" r"|git (commit|add|switch|rebase|merge|push)|git worktree (add|remove)|git branch -[dDmM]") _FILE = re.compile(r"(? str: """mut (edit/write) | book (wf, git writes) | run (build/test) | exp (everything else: exploring).""" if tool.get("name") in ("Edit", "Write", "NotebookEdit"): return "mut" if tool.get("name") != "Bash": return "exp" c = (tool.get("input") or {}).get("command") or "" return "book" if _BOOK.search(c) else "run" if _RUN.search(c) else "mut" if _MUT.search(c) else "exp" def result_kind(tool: dict) -> str: if tool.get("name") == "Bash": c = (tool.get("input") or {}).get("command") or "" return next((k for k, rx in _RESULT_KINDS if rx.search(c)), "other") return tool.get("name") or "other" def tool_files(tool: dict, root: str = "") -> list[str]: """Files a Read / Bash call touches, relative to the project (.worktrees// and root stripped).""" inp = tool.get("input") or {} if tool.get("name") == "Read": paths = [inp.get("file_path") or ""] elif tool.get("name") == "Bash": paths = [p for p in _FILE.findall(inp.get("command") or "") if not p.startswith("/") or p.startswith(root + "/")] else: return [] out = [] for p in paths: p = re.sub(r"^.*?/\.worktrees/[^/]+/", "", p) if root and p.startswith(root.rstrip("/") + "/"): p = p[len(root.rstrip("/")) + 1:] if p and p not in out: out.append(p) return out def _ctx(u: dict) -> int: return (u.get("input_tokens") or 0) + (u.get("cache_read_input_tokens") or 0) + (u.get("cache_creation_input_tokens") or 0) @dataclass class Explore: calls: int = 0 first_edit: int | None = None # calls before the first edit; None = never edited cost: dict = None # kind → billed input tokens pre: int = 0 # billed input before the first edit ctx_growth: int = 0 # context growth up to the first edit results: dict = None # result kind → tokens files: dict = None # file → result tokens (read by this agent) @property def total(self) -> int: return sum(self.cost.values()) @property def edited(self) -> bool: return self.first_edit is not None def explore(text: str, since: str | None = None, root: str = "") -> Explore: """One transcript → billed-input split by call kind, result tokens per tool kind, files read.""" calls: dict[str, dict] = {} for line in text.split("\n"): try: d = json.loads(line) except ValueError: continue if not isinstance(d, dict) or d.get("type") != "assistant": continue m = d.get("message") or {} key = m.get("id") or d.get("uuid") c = calls.setdefault(key, {"u": {}, "t": [], "ts": d.get("timestamp") or ""}) c["u"] = m.get("usage") or c["u"] c["t"] += [t for t in m.get("content") or [] if isinstance(t, dict) and t.get("type") == "tool_use"] kept = [c for c in calls.values() if not since or c["ts"] >= since] tools = {t["id"]: t for c in kept for t in c["t"] if "id" in t} ex = Explore(calls=len(kept), cost={}, results={}, files={}) for i, c in enumerate(kept): ks = [call_kind(t) for t in c["t"]] or ["exp"] k = "mut" if "mut" in ks else ks[0] if k == "mut" and ex.first_edit is None: ex.first_edit = i ex.pre = sum(_ctx(x["u"]) for x in kept[:i]) ex.ctx_growth = _ctx(c["u"]) - _ctx(kept[0]["u"]) ex.cost[k] = ex.cost.get(k, 0) + _ctx(c["u"]) for line in text.split("\n"): try: d = json.loads(line) except ValueError: continue content = (d.get("message") or {}).get("content") if isinstance(d, dict) and d.get("type") == "user" else None for c in content if isinstance(content, list) else []: if not isinstance(c, dict) or c.get("type") != "tool_result" or c.get("tool_use_id") not in tools: continue body = c.get("content") size = len(body if isinstance(body, str) else json.dumps(body)) // 4 t = tools[c["tool_use_id"]] k = result_kind(t) ex.results[k] = ex.results.get(k, 0) + size for f in tool_files(t, root): ex.files[f] = ex.files.get(f, 0) + size return ex def explore_report(agents: list[tuple[str, Explore]], min_calls: int = 4, min_pre: int = 10, top: int = 20) -> dict: """agents: (label, Explore). Agents under min_calls are skipped; the pre-edit median uses agents with an edit and >= min_pre calls. Returns rows, totals, results, files (≥2 agents: (file, agents, tokens)).""" use = [(l, e) for l, e in agents if e.calls >= min_calls] rows = [] for label, e in use: t = e.total or 1 rows.append((label, e.calls, e.first_edit if e.edited else e.calls, e.cost.get("exp", 0) / t, e.pre / t)) cost: dict[str, int] = {} results: dict[str, int] = {} for _, e in use: for k, v in e.cost.items(): cost[k] = cost.get(k, 0) + v for k, v in e.results.items(): results[k] = results.get(k, 0) + v pre = [e for _, e in use if e.edited and e.calls >= min_pre] seen: dict[str, list[int]] = {} for _, e in use: for f, n in e.files.items(): s = seen.setdefault(f, [0, 0]) s[0] += 1 s[1] += n files = sorted(((f, a, n) for f, (a, n) in seen.items() if a >= 2), key=lambda x: (-x[1], -x[2], x[0]))[:top] total = sum(cost.values()) return { "rows": rows, "agents": len(use), "total": total, "cost": cost, "results": results, "files": files, "pre_n": len(pre), "pre_share": median(e.pre / (e.total or 1) for e in pre) if pre else None, "pre_calls": median(e.first_edit for e in pre) if pre else None, "pre_ctx": median(e.ctx_growth for e in pre) if pre else None, }