From 81d4e80fd5aabe4e80f58e960affa795cf7d34ec Mon Sep 17 00:00:00 2001 From: godosa Date: Wed, 7 Oct 2026 07:27:17 +0200 Subject: workflow: initial public history --- wflib/usage.py | 375 +++++++++++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 375 insertions(+) create mode 100644 wflib/usage.py (limited to 'wflib/usage.py') diff --git a/wflib/usage.py b/wflib/usage.py new file mode 100644 index 0000000..5b7c1ee --- /dev/null +++ b/wflib/usage.py @@ -0,0 +1,375 @@ +"""Token usage and API-price cost from Claude Code transcripts (jsonl), per model. + +One API request is streamed as several transcript entries (one per content block) that repeat its +usage: count each (message.id, requestId) once, taking its last entry (final output_tokens). +Subagent transcripts never get a final entry (stop_reason null, output_tokens = the message_start +placeholder): there output is estimated from the content (out_estimate), counted in Usage.est. +""" +from __future__ import annotations + +import json +import re +from dataclasses import dataclass, fields +from statistics import median + +# $ per million tokens: input, output, cache read. Cache writes: 1.25x input (5 min), 2x input (1 h). +# Source: Anthropic pricing as of 2026-09-25. Longest matching prefix of the model id wins. +PRICES = { + "claude-fable-5-1": (10.0, 50.0, 0.25), + "claude-fable-5": (10.0, 50.0, 1.0), + "claude-opus-5-5": (4.0, 20.0, 0.20), + "claude-opus-5": (5.0, 25.0, 0.50), + "claude-opus-4-8": (5.0, 25.0, 0.50), + "claude-sonnet-5-5": (2.0, 10.0, 0.20), + "claude-sonnet-5": (2.0, 10.0, 0.20), + "claude-sonnet-4-6": (3.0, 15.0, 0.30), + "claude-haiku-4-5": (1.0, 5.0, 0.10), +} + + +@dataclass +class Usage: + turns: int = 0 + inp: int = 0 + cw5: int = 0 + cw1h: int = 0 + cr: int = 0 + out: int = 0 + est: int = 0 # requests whose out is estimated from content + + @property + def cw(self) -> int: + return self.cw5 + self.cw1h + + def add(self, other: "Usage") -> None: + for f in fields(self): + setattr(self, f.name, getattr(self, f.name) + getattr(other, f.name)) + + +def price(model: str) -> tuple[float, float, float] | None: + keys = [k for k in PRICES if model == k or model.startswith(k + "-")] + return PRICES[max(keys, key=len)] if keys else None + + +def cost(model: str, u: Usage) -> float | None: + """API-price $ (subscription sessions: what it would cost on the API); None = unknown model.""" + p = price(model) + if p is None: + return None + inp, out, read = p + return (u.inp * inp + u.cw5 * inp * 1.25 + u.cw1h * inp * 2 + u.cr * read + u.out * out) / 1e6 + + +def _usage(raw: dict) -> Usage: + cw = raw.get("cache_creation_input_tokens") or 0 + cw1h = (raw.get("cache_creation") or {}).get("ephemeral_1h_input_tokens") or 0 + return Usage(1, raw.get("input_tokens") or 0, cw - cw1h, cw1h, + raw.get("cache_read_input_tokens") or 0, raw.get("output_tokens") or 0) + + +# Output-token estimate per content block, fitted on ~100k main-session requests with final usage +# (2026-10): per-session totals within ±5%. Thinking text is usually empty; its signature grows ~3.3 +# chars per thinking token over a ~800-char base. +EST_TEXT, EST_TOOL_CHAR, EST_TOOL, EST_SIG, EST_SIG_BASE = 0.3, 0.44, 30, 0.3, 800 + + +def out_estimate(blocks: list) -> int: + t = 0.0 + for b in blocks: + if not isinstance(b, dict): + continue + kind = b.get("type") + if kind == "text": + t += len(b.get("text") or "") * EST_TEXT + elif kind == "tool_use": + t += EST_TOOL + len(json.dumps(b.get("input") or {}, ensure_ascii=False)) * EST_TOOL_CHAR + elif kind == "thinking": + t += (len(b.get("thinking") or "") * EST_TEXT + + max(0, len(b.get("signature") or "") - EST_SIG_BASE) * EST_SIG) + return round(t) + + +def parse(text: str, since: str | None = None, until: str | None = None) -> dict[str, Usage]: + """Model id → summed Usage. since/until: ISO prefixes compared with the UTC timestamps.""" + last: dict[tuple, tuple[str, dict]] = {} + first_ts: dict[tuple, str] = {} + final: set[tuple] = set() + blocks: dict[tuple, list] = {} + seen: set[str] = set() + for line in text.split("\n"): + try: + d = json.loads(line) + except ValueError: + continue + if not isinstance(d, dict) or d.get("type") != "assistant": + continue + m = d.get("message") or {} + model, raw = m.get("model") or "?", m.get("usage") + if not raw or model == "": + continue + key = (m.get("id"), d.get("requestId")) + first_ts.setdefault(key, d.get("timestamp") or "") + last[key] = (model, raw) + if m.get("stop_reason"): + final.add(key) + uid = d.get("uuid") + if uid not in seen and isinstance(m.get("content"), list): + blocks.setdefault(key, []).extend(m["content"]) + if uid: + seen.add(uid) + out: dict[str, Usage] = {} + for key, (model, raw) in last.items(): + ts = first_ts[key] + if (since and ts < since) or (until and ts >= until): + continue + u = _usage(raw) + if key not in final: + guess = out_estimate(blocks.get(key, [])) + if guess > u.out: + u.out, u.est = guess, 1 + out.setdefault(model, Usage()).add(u) + return out + + +def context_tokens(text: str) -> int | None: + """Prompt size of the last main-thread request (input + cache write + cache read); None = no request.""" + size = None + for line in text.split("\n"): + try: + d = json.loads(line) + except ValueError: + continue + if not isinstance(d, dict) or d.get("type") != "assistant" or d.get("isSidechain"): + continue + m = d.get("message") or {} + raw = m.get("usage") + if raw and m.get("model") != "": + u = _usage(raw) + size = u.inp + u.cw + u.cr + return size + + +def ctx_hint(tokens: int | None, limit: int) -> str | None: + if not limit or tokens is None or tokens <= limit: + return None + return (f"context ~{tokens // 1000}k tokens (> {limit // 1000}k): ask the owner to /clear, then continue " + "(subagent: ignore, this is the main session)") + + +def short(model: str) -> str: + return re.sub(r"-\d{8}$", "", model.removeprefix("claude-")) + + +def fmt(n: int) -> str: + if n < 1000: + return str(n) + return f"{n / 1e3:.1f}k" if n < 1e6 else f"{n / 1e6:.2f}M" + + +# ------------------------------------------------------------------ cost log (out/wf-cost.log) + +def log_line(time: str, project: str, task: str, effort: str, outcome: str, agent: str, + by_model: dict[str, Usage], lane: str | None = None, dur: int | None = None) -> str: + """One key=value line per worker; model = the model with most turns, lane = given or model; usd = priced models only.""" + total = Usage() + for u in by_model.values(): + total.add(u) + model = short(max(by_model, key=lambda m: by_model[m].turns)).split("-")[0] if by_model else "?" + usd = sum(cost(m, u) or 0 for m, u in by_model.items()) + return (f"{time} project={project} task={task} lane={lane or model} model={model} effort={effort} outcome={outcome} " + f"turns={total.turns} in={total.inp} cw={total.cw} cr={total.cr} out={total.out} " + + (f"est={total.est} " if total.est else "") + f"usd={usd:.4f} " + (f"dur={dur} " if dur is not None else "") + f"agent={agent}") + + +def parse_log(text: str, since: str | None = None) -> list[dict]: + out = [] + for line in text.split("\n"): + time, _, rest = line.partition(" ") + fields_ = dict(f.split("=", 1) for f in rest.split() if "=" in f) + if not re.match(r"\d{4}-\d\d-\d\dT", time) or "lane" not in fields_ or (since and time < since): + continue + try: + out.append({**fields_, "time": time, "usd": float(fields_.get("usd", 0)), + "turns": int(fields_.get("turns", 0)), + **({"dur": int(fields_["dur"])} if "dur" in fields_ else {})}) + except ValueError: + continue + return out + + +def _num(x: float) -> float | int: + x = round(x, 4) + return int(x) if x == int(x) and not isinstance(x, bool) else x + + +def report(entries: list[dict], efforts: tuple[str, ...]) -> list[tuple]: + """Per lane, then per lane+effort: (lane, effort, n, done, other, median $, total $, $/done, median turns, median dur s or None).""" + def row(lane, effort, es): + done = sum(e.get("outcome") in ("done", "done+gate-red") for e in es) + total = sum(e["usd"] for e in es) + return (lane, effort, len(es), done, len(es) - done, round(median(e["usd"] for e in es), 4), + round(total, 4), round(total / done, 4) if done else None, _num(median(e["turns"] for e in es)), + _num(median(durs)) if (durs := [e["dur"] for e in es if "dur" in e]) else None) + + order = {e: i for i, e in enumerate(efforts)} + rows = [] + key = lambda e: f"{e['lane']}/{e['model']}" if "model" in e and e["model"] != e["lane"] else e["lane"] + for lane in sorted({key(e) for e in entries}): + es = [e for e in entries if key(e) == lane] + rows.append(row(lane, "all", es)) + for effort in sorted({e.get("effort", "-") for e in es}, key=lambda x: (order.get(x, len(order)), x)): + rows.append(row(lane, effort, [e for e in es if e.get("effort", "-") == effort])) + return rows + + +# --- explore: where an agent's billed input goes (exploring vs editing/running/bookkeeping) --------------------- +_MUT = re.compile(r"sed -i|python3 - <<|cat > |tee |expect-init|>> [a-zA-Z]|apply_patch|patch ") +_RUN = re.compile(r"dotnet (test|build|run)|own-fx\.py (check|measure|dump)|npx |npm |gate-bg|check\.sh|wf res run|pytest|unittest") +_BOOK = re.compile(r"wf\.py (done|finish|add|status|note|set|merge|check|tick|body)" + r"|git (commit|add|switch|rebase|merge|push)|git worktree (add|remove)|git branch -[dDmM]") +_FILE = re.compile(r"(? str: + """mut (edit/write) | book (wf, git writes) | run (build/test) | exp (everything else: exploring).""" + if tool.get("name") in ("Edit", "Write", "NotebookEdit"): + return "mut" + if tool.get("name") != "Bash": + return "exp" + c = (tool.get("input") or {}).get("command") or "" + return "book" if _BOOK.search(c) else "run" if _RUN.search(c) else "mut" if _MUT.search(c) else "exp" + + +def result_kind(tool: dict) -> str: + if tool.get("name") == "Bash": + c = (tool.get("input") or {}).get("command") or "" + return next((k for k, rx in _RESULT_KINDS if rx.search(c)), "other") + return tool.get("name") or "other" + + +def tool_files(tool: dict, root: str = "") -> list[str]: + """Files a Read / Bash call touches, relative to the project (.worktrees// and root stripped).""" + inp = tool.get("input") or {} + if tool.get("name") == "Read": + paths = [inp.get("file_path") or ""] + elif tool.get("name") == "Bash": + paths = [p for p in _FILE.findall(inp.get("command") or "") if not p.startswith("/") or p.startswith(root + "/")] + else: + return [] + out = [] + for p in paths: + p = re.sub(r"^.*?/\.worktrees/[^/]+/", "", p) + if root and p.startswith(root.rstrip("/") + "/"): + p = p[len(root.rstrip("/")) + 1:] + if p and p not in out: + out.append(p) + return out + + +def _ctx(u: dict) -> int: + return (u.get("input_tokens") or 0) + (u.get("cache_read_input_tokens") or 0) + (u.get("cache_creation_input_tokens") or 0) + + +@dataclass +class Explore: + calls: int = 0 + first_edit: int | None = None # calls before the first edit; None = never edited + cost: dict = None # kind → billed input tokens + pre: int = 0 # billed input before the first edit + ctx_growth: int = 0 # context growth up to the first edit + results: dict = None # result kind → tokens + files: dict = None # file → result tokens (read by this agent) + + @property + def total(self) -> int: + return sum(self.cost.values()) + + @property + def edited(self) -> bool: + return self.first_edit is not None + + +def explore(text: str, since: str | None = None, root: str = "") -> Explore: + """One transcript → billed-input split by call kind, result tokens per tool kind, files read.""" + calls: dict[str, dict] = {} + for line in text.split("\n"): + try: + d = json.loads(line) + except ValueError: + continue + if not isinstance(d, dict) or d.get("type") != "assistant": + continue + m = d.get("message") or {} + key = m.get("id") or d.get("uuid") + c = calls.setdefault(key, {"u": {}, "t": [], "ts": d.get("timestamp") or ""}) + c["u"] = m.get("usage") or c["u"] + c["t"] += [t for t in m.get("content") or [] if isinstance(t, dict) and t.get("type") == "tool_use"] + kept = [c for c in calls.values() if not since or c["ts"] >= since] + tools = {t["id"]: t for c in kept for t in c["t"] if "id" in t} + ex = Explore(calls=len(kept), cost={}, results={}, files={}) + for i, c in enumerate(kept): + ks = [call_kind(t) for t in c["t"]] or ["exp"] + k = "mut" if "mut" in ks else ks[0] + if k == "mut" and ex.first_edit is None: + ex.first_edit = i + ex.pre = sum(_ctx(x["u"]) for x in kept[:i]) + ex.ctx_growth = _ctx(c["u"]) - _ctx(kept[0]["u"]) + ex.cost[k] = ex.cost.get(k, 0) + _ctx(c["u"]) + for line in text.split("\n"): + try: + d = json.loads(line) + except ValueError: + continue + content = (d.get("message") or {}).get("content") if isinstance(d, dict) and d.get("type") == "user" else None + for c in content if isinstance(content, list) else []: + if not isinstance(c, dict) or c.get("type") != "tool_result" or c.get("tool_use_id") not in tools: + continue + body = c.get("content") + size = len(body if isinstance(body, str) else json.dumps(body)) // 4 + t = tools[c["tool_use_id"]] + k = result_kind(t) + ex.results[k] = ex.results.get(k, 0) + size + for f in tool_files(t, root): + ex.files[f] = ex.files.get(f, 0) + size + return ex + + +def explore_report(agents: list[tuple[str, Explore]], min_calls: int = 4, min_pre: int = 10, top: int = 20) -> dict: + """agents: (label, Explore). Agents under min_calls are skipped; the pre-edit median uses agents with an + edit and >= min_pre calls. Returns rows, totals, results, files (≥2 agents: (file, agents, tokens)).""" + use = [(l, e) for l, e in agents if e.calls >= min_calls] + rows = [] + for label, e in use: + t = e.total or 1 + rows.append((label, e.calls, e.first_edit if e.edited else e.calls, e.cost.get("exp", 0) / t, e.pre / t)) + cost: dict[str, int] = {} + results: dict[str, int] = {} + for _, e in use: + for k, v in e.cost.items(): + cost[k] = cost.get(k, 0) + v + for k, v in e.results.items(): + results[k] = results.get(k, 0) + v + pre = [e for _, e in use if e.edited and e.calls >= min_pre] + seen: dict[str, list[int]] = {} + for _, e in use: + for f, n in e.files.items(): + s = seen.setdefault(f, [0, 0]) + s[0] += 1 + s[1] += n + files = sorted(((f, a, n) for f, (a, n) in seen.items() if a >= 2), key=lambda x: (-x[1], -x[2], x[0]))[:top] + total = sum(cost.values()) + return { + "rows": rows, "agents": len(use), "total": total, + "cost": cost, "results": results, "files": files, + "pre_n": len(pre), + "pre_share": median(e.pre / (e.total or 1) for e in pre) if pre else None, + "pre_calls": median(e.first_edit for e in pre) if pre else None, + "pre_ctx": median(e.ctx_growth for e in pre) if pre else None, + } -- cgit