1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
|
"""Token usage and API-price cost from Claude Code transcripts (jsonl), per model.
One API request is streamed as several transcript entries (one per content block) that repeat its
usage: count each (message.id, requestId) once, taking its last entry (final output_tokens).
Subagent transcripts never get a final entry (stop_reason null, output_tokens = the message_start
placeholder): there output is estimated from the content (out_estimate), counted in Usage.est.
"""
from __future__ import annotations
import json
import re
from dataclasses import dataclass, fields
from statistics import median
# $ per million tokens: input, output, cache read. Cache writes: 1.25x input (5 min), 2x input (1 h).
# Source: Anthropic pricing as of 2026-09-25. Longest matching prefix of the model id wins.
PRICES = {
"claude-fable-5-1": (10.0, 50.0, 0.25),
"claude-fable-5": (10.0, 50.0, 1.0),
"claude-opus-5-5": (4.0, 20.0, 0.20),
"claude-opus-5": (5.0, 25.0, 0.50),
"claude-opus-4-8": (5.0, 25.0, 0.50),
"claude-sonnet-5-5": (2.0, 10.0, 0.20),
"claude-sonnet-5": (2.0, 10.0, 0.20),
"claude-sonnet-4-6": (3.0, 15.0, 0.30),
"claude-haiku-4-5": (1.0, 5.0, 0.10),
}
@dataclass
class Usage:
turns: int = 0
inp: int = 0
cw5: int = 0
cw1h: int = 0
cr: int = 0
out: int = 0
est: int = 0 # requests whose out is estimated from content
@property
def cw(self) -> int:
return self.cw5 + self.cw1h
def add(self, other: "Usage") -> None:
for f in fields(self):
setattr(self, f.name, getattr(self, f.name) + getattr(other, f.name))
def price(model: str) -> tuple[float, float, float] | None:
keys = [k for k in PRICES if model == k or model.startswith(k + "-")]
return PRICES[max(keys, key=len)] if keys else None
def cost(model: str, u: Usage) -> float | None:
"""API-price $ (subscription sessions: what it would cost on the API); None = unknown model."""
p = price(model)
if p is None:
return None
inp, out, read = p
return (u.inp * inp + u.cw5 * inp * 1.25 + u.cw1h * inp * 2 + u.cr * read + u.out * out) / 1e6
def _usage(raw: dict) -> Usage:
cw = raw.get("cache_creation_input_tokens") or 0
cw1h = (raw.get("cache_creation") or {}).get("ephemeral_1h_input_tokens") or 0
return Usage(1, raw.get("input_tokens") or 0, cw - cw1h, cw1h,
raw.get("cache_read_input_tokens") or 0, raw.get("output_tokens") or 0)
# Output-token estimate per content block, fitted on ~100k main-session requests with final usage
# (2026-10): per-session totals within ±5%. Thinking text is usually empty; its signature grows ~3.3
# chars per thinking token over a ~800-char base.
EST_TEXT, EST_TOOL_CHAR, EST_TOOL, EST_SIG, EST_SIG_BASE = 0.3, 0.44, 30, 0.3, 800
def out_estimate(blocks: list) -> int:
t = 0.0
for b in blocks:
if not isinstance(b, dict):
continue
kind = b.get("type")
if kind == "text":
t += len(b.get("text") or "") * EST_TEXT
elif kind == "tool_use":
t += EST_TOOL + len(json.dumps(b.get("input") or {}, ensure_ascii=False)) * EST_TOOL_CHAR
elif kind == "thinking":
t += (len(b.get("thinking") or "") * EST_TEXT
+ max(0, len(b.get("signature") or "") - EST_SIG_BASE) * EST_SIG)
return round(t)
def parse(text: str, since: str | None = None, until: str | None = None) -> dict[str, Usage]:
"""Model id → summed Usage. since/until: ISO prefixes compared with the UTC timestamps."""
last: dict[tuple, tuple[str, dict]] = {}
first_ts: dict[tuple, str] = {}
final: set[tuple] = set()
blocks: dict[tuple, list] = {}
seen: set[str] = set()
for line in text.split("\n"):
try:
d = json.loads(line)
except ValueError:
continue
if not isinstance(d, dict) or d.get("type") != "assistant":
continue
m = d.get("message") or {}
model, raw = m.get("model") or "?", m.get("usage")
if not raw or model == "<synthetic>":
continue
key = (m.get("id"), d.get("requestId"))
first_ts.setdefault(key, d.get("timestamp") or "")
last[key] = (model, raw)
if m.get("stop_reason"):
final.add(key)
uid = d.get("uuid")
if uid not in seen and isinstance(m.get("content"), list):
blocks.setdefault(key, []).extend(m["content"])
if uid:
seen.add(uid)
out: dict[str, Usage] = {}
for key, (model, raw) in last.items():
ts = first_ts[key]
if (since and ts < since) or (until and ts >= until):
continue
u = _usage(raw)
if key not in final:
guess = out_estimate(blocks.get(key, []))
if guess > u.out:
u.out, u.est = guess, 1
out.setdefault(model, Usage()).add(u)
return out
def context_tokens(text: str) -> int | None:
"""Prompt size of the last main-thread request (input + cache write + cache read); None = no request."""
size = None
for line in text.split("\n"):
try:
d = json.loads(line)
except ValueError:
continue
if not isinstance(d, dict) or d.get("type") != "assistant" or d.get("isSidechain"):
continue
m = d.get("message") or {}
raw = m.get("usage")
if raw and m.get("model") != "<synthetic>":
u = _usage(raw)
size = u.inp + u.cw + u.cr
return size
def ctx_hint(tokens: int | None, limit: int) -> str | None:
if not limit or tokens is None or tokens <= limit:
return None
return (f"context ~{tokens // 1000}k tokens (> {limit // 1000}k): ask the owner to /clear, then continue "
"(subagent: ignore, this is the main session)")
def short(model: str) -> str:
return re.sub(r"-\d{8}$", "", model.removeprefix("claude-"))
def fmt(n: int) -> str:
if n < 1000:
return str(n)
return f"{n / 1e3:.1f}k" if n < 1e6 else f"{n / 1e6:.2f}M"
# ------------------------------------------------------------------ cost log (out/wf-cost.log)
def log_line(time: str, project: str, task: str, effort: str, outcome: str, agent: str,
by_model: dict[str, Usage], lane: str | None = None, dur: int | None = None) -> str:
"""One key=value line per worker; model = the model with most turns, lane = given or model; usd = priced models only."""
total = Usage()
for u in by_model.values():
total.add(u)
model = short(max(by_model, key=lambda m: by_model[m].turns)).split("-")[0] if by_model else "?"
usd = sum(cost(m, u) or 0 for m, u in by_model.items())
return (f"{time} project={project} task={task} lane={lane or model} model={model} effort={effort} outcome={outcome} "
f"turns={total.turns} in={total.inp} cw={total.cw} cr={total.cr} out={total.out} "
+ (f"est={total.est} " if total.est else "") + f"usd={usd:.4f} " + (f"dur={dur} " if dur is not None else "") + f"agent={agent}")
def parse_log(text: str, since: str | None = None) -> list[dict]:
out = []
for line in text.split("\n"):
time, _, rest = line.partition(" ")
fields_ = dict(f.split("=", 1) for f in rest.split() if "=" in f)
if not re.match(r"\d{4}-\d\d-\d\dT", time) or "lane" not in fields_ or (since and time < since):
continue
try:
out.append({**fields_, "time": time, "usd": float(fields_.get("usd", 0)),
"turns": int(fields_.get("turns", 0)),
**({"dur": int(fields_["dur"])} if "dur" in fields_ else {})})
except ValueError:
continue
return out
def _num(x: float) -> float | int:
x = round(x, 4)
return int(x) if x == int(x) and not isinstance(x, bool) else x
def report(entries: list[dict], efforts: tuple[str, ...]) -> list[tuple]:
"""Per lane, then per lane+effort: (lane, effort, n, done, other, median $, total $, $/done, median turns, median dur s or None)."""
def row(lane, effort, es):
done = sum(e.get("outcome") in ("done", "done+gate-red") for e in es)
total = sum(e["usd"] for e in es)
return (lane, effort, len(es), done, len(es) - done, round(median(e["usd"] for e in es), 4),
round(total, 4), round(total / done, 4) if done else None, _num(median(e["turns"] for e in es)),
_num(median(durs)) if (durs := [e["dur"] for e in es if "dur" in e]) else None)
order = {e: i for i, e in enumerate(efforts)}
rows = []
key = lambda e: f"{e['lane']}/{e['model']}" if "model" in e and e["model"] != e["lane"] else e["lane"]
for lane in sorted({key(e) for e in entries}):
es = [e for e in entries if key(e) == lane]
rows.append(row(lane, "all", es))
for effort in sorted({e.get("effort", "-") for e in es}, key=lambda x: (order.get(x, len(order)), x)):
rows.append(row(lane, effort, [e for e in es if e.get("effort", "-") == effort]))
return rows
# --- explore: where an agent's billed input goes (exploring vs editing/running/bookkeeping) ---------------------
_MUT = re.compile(r"sed -i|python3 - <<|cat > |tee |expect-init|>> [a-zA-Z]|apply_patch|patch ")
_RUN = re.compile(r"dotnet (test|build|run)|own-fx\.py (check|measure|dump)|npx |npm |gate-bg|check\.sh|wf res run|pytest|unittest")
_BOOK = re.compile(r"wf\.py (done|finish|add|status|note|set|merge|check|tick|body)"
r"|git (commit|add|switch|rebase|merge|push)|git worktree (add|remove)|git branch -[dDmM]")
_FILE = re.compile(r"(?<![\w.-])((?:\./|/)?[\w.-]+(?:/[\w.-]+)+\.\w+)(?![\w/-])")
_RESULT_KINDS = ( # first match wins; Bash result kinds
("git-hist", re.compile(r"git (show|log|diff)")),
("wf", re.compile(r"wf\.py")),
("build/test", re.compile(r"dotnet|npx|npm")),
("grep", re.compile(r"\bgrep|\brg\b|find ")),
("sed-cat", re.compile(r"sed -n|cat |head|tail")),
)
def call_kind(tool: dict) -> str:
"""mut (edit/write) | book (wf, git writes) | run (build/test) | exp (everything else: exploring)."""
if tool.get("name") in ("Edit", "Write", "NotebookEdit"):
return "mut"
if tool.get("name") != "Bash":
return "exp"
c = (tool.get("input") or {}).get("command") or ""
return "book" if _BOOK.search(c) else "run" if _RUN.search(c) else "mut" if _MUT.search(c) else "exp"
def result_kind(tool: dict) -> str:
if tool.get("name") == "Bash":
c = (tool.get("input") or {}).get("command") or ""
return next((k for k, rx in _RESULT_KINDS if rx.search(c)), "other")
return tool.get("name") or "other"
def tool_files(tool: dict, root: str = "") -> list[str]:
"""Files a Read / Bash call touches, relative to the project (.worktrees/<x>/ and root stripped)."""
inp = tool.get("input") or {}
if tool.get("name") == "Read":
paths = [inp.get("file_path") or ""]
elif tool.get("name") == "Bash":
paths = [p for p in _FILE.findall(inp.get("command") or "") if not p.startswith("/") or p.startswith(root + "/")]
else:
return []
out = []
for p in paths:
p = re.sub(r"^.*?/\.worktrees/[^/]+/", "", p)
if root and p.startswith(root.rstrip("/") + "/"):
p = p[len(root.rstrip("/")) + 1:]
if p and p not in out:
out.append(p)
return out
def _ctx(u: dict) -> int:
return (u.get("input_tokens") or 0) + (u.get("cache_read_input_tokens") or 0) + (u.get("cache_creation_input_tokens") or 0)
@dataclass
class Explore:
calls: int = 0
first_edit: int | None = None # calls before the first edit; None = never edited
cost: dict = None # kind → billed input tokens
pre: int = 0 # billed input before the first edit
ctx_growth: int = 0 # context growth up to the first edit
results: dict = None # result kind → tokens
files: dict = None # file → result tokens (read by this agent)
@property
def total(self) -> int:
return sum(self.cost.values())
@property
def edited(self) -> bool:
return self.first_edit is not None
def explore(text: str, since: str | None = None, root: str = "") -> Explore:
"""One transcript → billed-input split by call kind, result tokens per tool kind, files read."""
calls: dict[str, dict] = {}
for line in text.split("\n"):
try:
d = json.loads(line)
except ValueError:
continue
if not isinstance(d, dict) or d.get("type") != "assistant":
continue
m = d.get("message") or {}
key = m.get("id") or d.get("uuid")
c = calls.setdefault(key, {"u": {}, "t": [], "ts": d.get("timestamp") or ""})
c["u"] = m.get("usage") or c["u"]
c["t"] += [t for t in m.get("content") or [] if isinstance(t, dict) and t.get("type") == "tool_use"]
kept = [c for c in calls.values() if not since or c["ts"] >= since]
tools = {t["id"]: t for c in kept for t in c["t"] if "id" in t}
ex = Explore(calls=len(kept), cost={}, results={}, files={})
for i, c in enumerate(kept):
ks = [call_kind(t) for t in c["t"]] or ["exp"]
k = "mut" if "mut" in ks else ks[0]
if k == "mut" and ex.first_edit is None:
ex.first_edit = i
ex.pre = sum(_ctx(x["u"]) for x in kept[:i])
ex.ctx_growth = _ctx(c["u"]) - _ctx(kept[0]["u"])
ex.cost[k] = ex.cost.get(k, 0) + _ctx(c["u"])
for line in text.split("\n"):
try:
d = json.loads(line)
except ValueError:
continue
content = (d.get("message") or {}).get("content") if isinstance(d, dict) and d.get("type") == "user" else None
for c in content if isinstance(content, list) else []:
if not isinstance(c, dict) or c.get("type") != "tool_result" or c.get("tool_use_id") not in tools:
continue
body = c.get("content")
size = len(body if isinstance(body, str) else json.dumps(body)) // 4
t = tools[c["tool_use_id"]]
k = result_kind(t)
ex.results[k] = ex.results.get(k, 0) + size
for f in tool_files(t, root):
ex.files[f] = ex.files.get(f, 0) + size
return ex
def explore_report(agents: list[tuple[str, Explore]], min_calls: int = 4, min_pre: int = 10, top: int = 20) -> dict:
"""agents: (label, Explore). Agents under min_calls are skipped; the pre-edit median uses agents with an
edit and >= min_pre calls. Returns rows, totals, results, files (≥2 agents: (file, agents, tokens))."""
use = [(l, e) for l, e in agents if e.calls >= min_calls]
rows = []
for label, e in use:
t = e.total or 1
rows.append((label, e.calls, e.first_edit if e.edited else e.calls, e.cost.get("exp", 0) / t, e.pre / t))
cost: dict[str, int] = {}
results: dict[str, int] = {}
for _, e in use:
for k, v in e.cost.items():
cost[k] = cost.get(k, 0) + v
for k, v in e.results.items():
results[k] = results.get(k, 0) + v
pre = [e for _, e in use if e.edited and e.calls >= min_pre]
seen: dict[str, list[int]] = {}
for _, e in use:
for f, n in e.files.items():
s = seen.setdefault(f, [0, 0])
s[0] += 1
s[1] += n
files = sorted(((f, a, n) for f, (a, n) in seen.items() if a >= 2), key=lambda x: (-x[1], -x[2], x[0]))[:top]
total = sum(cost.values())
return {
"rows": rows, "agents": len(use), "total": total,
"cost": cost, "results": results, "files": files,
"pre_n": len(pre),
"pre_share": median(e.pre / (e.total or 1) for e in pre) if pre else None,
"pre_calls": median(e.first_edit for e in pre) if pre else None,
"pre_ctx": median(e.ctx_growth for e in pre) if pre else None,
}
|