aboutsummaryrefslogtreecommitdiffziptar.gz
path: root/wflib/usage.py
diff options
context:
space:
mode:
Diffstat (limited to 'wflib/usage.py')
-rw-r--r--wflib/usage.py375
1 files changed, 375 insertions, 0 deletions
diff --git a/wflib/usage.py b/wflib/usage.py
new file mode 100644
index 0000000..5b7c1ee
--- /dev/null
+++ b/wflib/usage.py
@@ -0,0 +1,375 @@
+"""Token usage and API-price cost from Claude Code transcripts (jsonl), per model.
+
+One API request is streamed as several transcript entries (one per content block) that repeat its
+usage: count each (message.id, requestId) once, taking its last entry (final output_tokens).
+Subagent transcripts never get a final entry (stop_reason null, output_tokens = the message_start
+placeholder): there output is estimated from the content (out_estimate), counted in Usage.est.
+"""
+from __future__ import annotations
+
+import json
+import re
+from dataclasses import dataclass, fields
+from statistics import median
+
+# $ per million tokens: input, output, cache read. Cache writes: 1.25x input (5 min), 2x input (1 h).
+# Source: Anthropic pricing as of 2026-09-25. Longest matching prefix of the model id wins.
+PRICES = {
+ "claude-fable-5-1": (10.0, 50.0, 0.25),
+ "claude-fable-5": (10.0, 50.0, 1.0),
+ "claude-opus-5-5": (4.0, 20.0, 0.20),
+ "claude-opus-5": (5.0, 25.0, 0.50),
+ "claude-opus-4-8": (5.0, 25.0, 0.50),
+ "claude-sonnet-5-5": (2.0, 10.0, 0.20),
+ "claude-sonnet-5": (2.0, 10.0, 0.20),
+ "claude-sonnet-4-6": (3.0, 15.0, 0.30),
+ "claude-haiku-4-5": (1.0, 5.0, 0.10),
+}
+
+
+@dataclass
+class Usage:
+ turns: int = 0
+ inp: int = 0
+ cw5: int = 0
+ cw1h: int = 0
+ cr: int = 0
+ out: int = 0
+ est: int = 0 # requests whose out is estimated from content
+
+ @property
+ def cw(self) -> int:
+ return self.cw5 + self.cw1h
+
+ def add(self, other: "Usage") -> None:
+ for f in fields(self):
+ setattr(self, f.name, getattr(self, f.name) + getattr(other, f.name))
+
+
+def price(model: str) -> tuple[float, float, float] | None:
+ keys = [k for k in PRICES if model == k or model.startswith(k + "-")]
+ return PRICES[max(keys, key=len)] if keys else None
+
+
+def cost(model: str, u: Usage) -> float | None:
+ """API-price $ (subscription sessions: what it would cost on the API); None = unknown model."""
+ p = price(model)
+ if p is None:
+ return None
+ inp, out, read = p
+ return (u.inp * inp + u.cw5 * inp * 1.25 + u.cw1h * inp * 2 + u.cr * read + u.out * out) / 1e6
+
+
+def _usage(raw: dict) -> Usage:
+ cw = raw.get("cache_creation_input_tokens") or 0
+ cw1h = (raw.get("cache_creation") or {}).get("ephemeral_1h_input_tokens") or 0
+ return Usage(1, raw.get("input_tokens") or 0, cw - cw1h, cw1h,
+ raw.get("cache_read_input_tokens") or 0, raw.get("output_tokens") or 0)
+
+
+# Output-token estimate per content block, fitted on ~100k main-session requests with final usage
+# (2026-10): per-session totals within ±5%. Thinking text is usually empty; its signature grows ~3.3
+# chars per thinking token over a ~800-char base.
+EST_TEXT, EST_TOOL_CHAR, EST_TOOL, EST_SIG, EST_SIG_BASE = 0.3, 0.44, 30, 0.3, 800
+
+
+def out_estimate(blocks: list) -> int:
+ t = 0.0
+ for b in blocks:
+ if not isinstance(b, dict):
+ continue
+ kind = b.get("type")
+ if kind == "text":
+ t += len(b.get("text") or "") * EST_TEXT
+ elif kind == "tool_use":
+ t += EST_TOOL + len(json.dumps(b.get("input") or {}, ensure_ascii=False)) * EST_TOOL_CHAR
+ elif kind == "thinking":
+ t += (len(b.get("thinking") or "") * EST_TEXT
+ + max(0, len(b.get("signature") or "") - EST_SIG_BASE) * EST_SIG)
+ return round(t)
+
+
+def parse(text: str, since: str | None = None, until: str | None = None) -> dict[str, Usage]:
+ """Model id → summed Usage. since/until: ISO prefixes compared with the UTC timestamps."""
+ last: dict[tuple, tuple[str, dict]] = {}
+ first_ts: dict[tuple, str] = {}
+ final: set[tuple] = set()
+ blocks: dict[tuple, list] = {}
+ seen: set[str] = set()
+ for line in text.split("\n"):
+ try:
+ d = json.loads(line)
+ except ValueError:
+ continue
+ if not isinstance(d, dict) or d.get("type") != "assistant":
+ continue
+ m = d.get("message") or {}
+ model, raw = m.get("model") or "?", m.get("usage")
+ if not raw or model == "<synthetic>":
+ continue
+ key = (m.get("id"), d.get("requestId"))
+ first_ts.setdefault(key, d.get("timestamp") or "")
+ last[key] = (model, raw)
+ if m.get("stop_reason"):
+ final.add(key)
+ uid = d.get("uuid")
+ if uid not in seen and isinstance(m.get("content"), list):
+ blocks.setdefault(key, []).extend(m["content"])
+ if uid:
+ seen.add(uid)
+ out: dict[str, Usage] = {}
+ for key, (model, raw) in last.items():
+ ts = first_ts[key]
+ if (since and ts < since) or (until and ts >= until):
+ continue
+ u = _usage(raw)
+ if key not in final:
+ guess = out_estimate(blocks.get(key, []))
+ if guess > u.out:
+ u.out, u.est = guess, 1
+ out.setdefault(model, Usage()).add(u)
+ return out
+
+
+def context_tokens(text: str) -> int | None:
+ """Prompt size of the last main-thread request (input + cache write + cache read); None = no request."""
+ size = None
+ for line in text.split("\n"):
+ try:
+ d = json.loads(line)
+ except ValueError:
+ continue
+ if not isinstance(d, dict) or d.get("type") != "assistant" or d.get("isSidechain"):
+ continue
+ m = d.get("message") or {}
+ raw = m.get("usage")
+ if raw and m.get("model") != "<synthetic>":
+ u = _usage(raw)
+ size = u.inp + u.cw + u.cr
+ return size
+
+
+def ctx_hint(tokens: int | None, limit: int) -> str | None:
+ if not limit or tokens is None or tokens <= limit:
+ return None
+ return (f"context ~{tokens // 1000}k tokens (> {limit // 1000}k): ask the owner to /clear, then continue "
+ "(subagent: ignore, this is the main session)")
+
+
+def short(model: str) -> str:
+ return re.sub(r"-\d{8}$", "", model.removeprefix("claude-"))
+
+
+def fmt(n: int) -> str:
+ if n < 1000:
+ return str(n)
+ return f"{n / 1e3:.1f}k" if n < 1e6 else f"{n / 1e6:.2f}M"
+
+
+# ------------------------------------------------------------------ cost log (out/wf-cost.log)
+
+def log_line(time: str, project: str, task: str, effort: str, outcome: str, agent: str,
+ by_model: dict[str, Usage], lane: str | None = None, dur: int | None = None) -> str:
+ """One key=value line per worker; model = the model with most turns, lane = given or model; usd = priced models only."""
+ total = Usage()
+ for u in by_model.values():
+ total.add(u)
+ model = short(max(by_model, key=lambda m: by_model[m].turns)).split("-")[0] if by_model else "?"
+ usd = sum(cost(m, u) or 0 for m, u in by_model.items())
+ return (f"{time} project={project} task={task} lane={lane or model} model={model} effort={effort} outcome={outcome} "
+ f"turns={total.turns} in={total.inp} cw={total.cw} cr={total.cr} out={total.out} "
+ + (f"est={total.est} " if total.est else "") + f"usd={usd:.4f} " + (f"dur={dur} " if dur is not None else "") + f"agent={agent}")
+
+
+def parse_log(text: str, since: str | None = None) -> list[dict]:
+ out = []
+ for line in text.split("\n"):
+ time, _, rest = line.partition(" ")
+ fields_ = dict(f.split("=", 1) for f in rest.split() if "=" in f)
+ if not re.match(r"\d{4}-\d\d-\d\dT", time) or "lane" not in fields_ or (since and time < since):
+ continue
+ try:
+ out.append({**fields_, "time": time, "usd": float(fields_.get("usd", 0)),
+ "turns": int(fields_.get("turns", 0)),
+ **({"dur": int(fields_["dur"])} if "dur" in fields_ else {})})
+ except ValueError:
+ continue
+ return out
+
+
+def _num(x: float) -> float | int:
+ x = round(x, 4)
+ return int(x) if x == int(x) and not isinstance(x, bool) else x
+
+
+def report(entries: list[dict], efforts: tuple[str, ...]) -> list[tuple]:
+ """Per lane, then per lane+effort: (lane, effort, n, done, other, median $, total $, $/done, median turns, median dur s or None)."""
+ def row(lane, effort, es):
+ done = sum(e.get("outcome") in ("done", "done+gate-red") for e in es)
+ total = sum(e["usd"] for e in es)
+ return (lane, effort, len(es), done, len(es) - done, round(median(e["usd"] for e in es), 4),
+ round(total, 4), round(total / done, 4) if done else None, _num(median(e["turns"] for e in es)),
+ _num(median(durs)) if (durs := [e["dur"] for e in es if "dur" in e]) else None)
+
+ order = {e: i for i, e in enumerate(efforts)}
+ rows = []
+ key = lambda e: f"{e['lane']}/{e['model']}" if "model" in e and e["model"] != e["lane"] else e["lane"]
+ for lane in sorted({key(e) for e in entries}):
+ es = [e for e in entries if key(e) == lane]
+ rows.append(row(lane, "all", es))
+ for effort in sorted({e.get("effort", "-") for e in es}, key=lambda x: (order.get(x, len(order)), x)):
+ rows.append(row(lane, effort, [e for e in es if e.get("effort", "-") == effort]))
+ return rows
+
+
+# --- explore: where an agent's billed input goes (exploring vs editing/running/bookkeeping) ---------------------
+_MUT = re.compile(r"sed -i|python3 - <<|cat > |tee |expect-init|>> [a-zA-Z]|apply_patch|patch ")
+_RUN = re.compile(r"dotnet (test|build|run)|own-fx\.py (check|measure|dump)|npx |npm |gate-bg|check\.sh|wf res run|pytest|unittest")
+_BOOK = re.compile(r"wf\.py (done|finish|add|status|note|set|merge|check|tick|body)"
+ r"|git (commit|add|switch|rebase|merge|push)|git worktree (add|remove)|git branch -[dDmM]")
+_FILE = re.compile(r"(?<![\w.-])((?:\./|/)?[\w.-]+(?:/[\w.-]+)+\.\w+)(?![\w/-])")
+_RESULT_KINDS = ( # first match wins; Bash result kinds
+ ("git-hist", re.compile(r"git (show|log|diff)")),
+ ("wf", re.compile(r"wf\.py")),
+ ("build/test", re.compile(r"dotnet|npx|npm")),
+ ("grep", re.compile(r"\bgrep|\brg\b|find ")),
+ ("sed-cat", re.compile(r"sed -n|cat |head|tail")),
+)
+
+
+def call_kind(tool: dict) -> str:
+ """mut (edit/write) | book (wf, git writes) | run (build/test) | exp (everything else: exploring)."""
+ if tool.get("name") in ("Edit", "Write", "NotebookEdit"):
+ return "mut"
+ if tool.get("name") != "Bash":
+ return "exp"
+ c = (tool.get("input") or {}).get("command") or ""
+ return "book" if _BOOK.search(c) else "run" if _RUN.search(c) else "mut" if _MUT.search(c) else "exp"
+
+
+def result_kind(tool: dict) -> str:
+ if tool.get("name") == "Bash":
+ c = (tool.get("input") or {}).get("command") or ""
+ return next((k for k, rx in _RESULT_KINDS if rx.search(c)), "other")
+ return tool.get("name") or "other"
+
+
+def tool_files(tool: dict, root: str = "") -> list[str]:
+ """Files a Read / Bash call touches, relative to the project (.worktrees/<x>/ and root stripped)."""
+ inp = tool.get("input") or {}
+ if tool.get("name") == "Read":
+ paths = [inp.get("file_path") or ""]
+ elif tool.get("name") == "Bash":
+ paths = [p for p in _FILE.findall(inp.get("command") or "") if not p.startswith("/") or p.startswith(root + "/")]
+ else:
+ return []
+ out = []
+ for p in paths:
+ p = re.sub(r"^.*?/\.worktrees/[^/]+/", "", p)
+ if root and p.startswith(root.rstrip("/") + "/"):
+ p = p[len(root.rstrip("/")) + 1:]
+ if p and p not in out:
+ out.append(p)
+ return out
+
+
+def _ctx(u: dict) -> int:
+ return (u.get("input_tokens") or 0) + (u.get("cache_read_input_tokens") or 0) + (u.get("cache_creation_input_tokens") or 0)
+
+
+@dataclass
+class Explore:
+ calls: int = 0
+ first_edit: int | None = None # calls before the first edit; None = never edited
+ cost: dict = None # kind → billed input tokens
+ pre: int = 0 # billed input before the first edit
+ ctx_growth: int = 0 # context growth up to the first edit
+ results: dict = None # result kind → tokens
+ files: dict = None # file → result tokens (read by this agent)
+
+ @property
+ def total(self) -> int:
+ return sum(self.cost.values())
+
+ @property
+ def edited(self) -> bool:
+ return self.first_edit is not None
+
+
+def explore(text: str, since: str | None = None, root: str = "") -> Explore:
+ """One transcript → billed-input split by call kind, result tokens per tool kind, files read."""
+ calls: dict[str, dict] = {}
+ for line in text.split("\n"):
+ try:
+ d = json.loads(line)
+ except ValueError:
+ continue
+ if not isinstance(d, dict) or d.get("type") != "assistant":
+ continue
+ m = d.get("message") or {}
+ key = m.get("id") or d.get("uuid")
+ c = calls.setdefault(key, {"u": {}, "t": [], "ts": d.get("timestamp") or ""})
+ c["u"] = m.get("usage") or c["u"]
+ c["t"] += [t for t in m.get("content") or [] if isinstance(t, dict) and t.get("type") == "tool_use"]
+ kept = [c for c in calls.values() if not since or c["ts"] >= since]
+ tools = {t["id"]: t for c in kept for t in c["t"] if "id" in t}
+ ex = Explore(calls=len(kept), cost={}, results={}, files={})
+ for i, c in enumerate(kept):
+ ks = [call_kind(t) for t in c["t"]] or ["exp"]
+ k = "mut" if "mut" in ks else ks[0]
+ if k == "mut" and ex.first_edit is None:
+ ex.first_edit = i
+ ex.pre = sum(_ctx(x["u"]) for x in kept[:i])
+ ex.ctx_growth = _ctx(c["u"]) - _ctx(kept[0]["u"])
+ ex.cost[k] = ex.cost.get(k, 0) + _ctx(c["u"])
+ for line in text.split("\n"):
+ try:
+ d = json.loads(line)
+ except ValueError:
+ continue
+ content = (d.get("message") or {}).get("content") if isinstance(d, dict) and d.get("type") == "user" else None
+ for c in content if isinstance(content, list) else []:
+ if not isinstance(c, dict) or c.get("type") != "tool_result" or c.get("tool_use_id") not in tools:
+ continue
+ body = c.get("content")
+ size = len(body if isinstance(body, str) else json.dumps(body)) // 4
+ t = tools[c["tool_use_id"]]
+ k = result_kind(t)
+ ex.results[k] = ex.results.get(k, 0) + size
+ for f in tool_files(t, root):
+ ex.files[f] = ex.files.get(f, 0) + size
+ return ex
+
+
+def explore_report(agents: list[tuple[str, Explore]], min_calls: int = 4, min_pre: int = 10, top: int = 20) -> dict:
+ """agents: (label, Explore). Agents under min_calls are skipped; the pre-edit median uses agents with an
+ edit and >= min_pre calls. Returns rows, totals, results, files (≥2 agents: (file, agents, tokens))."""
+ use = [(l, e) for l, e in agents if e.calls >= min_calls]
+ rows = []
+ for label, e in use:
+ t = e.total or 1
+ rows.append((label, e.calls, e.first_edit if e.edited else e.calls, e.cost.get("exp", 0) / t, e.pre / t))
+ cost: dict[str, int] = {}
+ results: dict[str, int] = {}
+ for _, e in use:
+ for k, v in e.cost.items():
+ cost[k] = cost.get(k, 0) + v
+ for k, v in e.results.items():
+ results[k] = results.get(k, 0) + v
+ pre = [e for _, e in use if e.edited and e.calls >= min_pre]
+ seen: dict[str, list[int]] = {}
+ for _, e in use:
+ for f, n in e.files.items():
+ s = seen.setdefault(f, [0, 0])
+ s[0] += 1
+ s[1] += n
+ files = sorted(((f, a, n) for f, (a, n) in seen.items() if a >= 2), key=lambda x: (-x[1], -x[2], x[0]))[:top]
+ total = sum(cost.values())
+ return {
+ "rows": rows, "agents": len(use), "total": total,
+ "cost": cost, "results": results, "files": files,
+ "pre_n": len(pre),
+ "pre_share": median(e.pre / (e.total or 1) for e in pre) if pre else None,
+ "pre_calls": median(e.first_edit for e in pre) if pre else None,
+ "pre_ctx": median(e.ctx_growth for e in pre) if pre else None,
+ }