aboutsummaryrefslogtreecommitdiffziptar.gz
path: root/tests
diff options
context:
space:
mode:
Diffstat (limited to 'tests')
-rw-r--r--tests/test_areas.py319
-rw-r--r--tests/test_batch.py534
-rw-r--r--tests/test_check.py304
-rw-r--r--tests/test_claims.py182
-rw-r--r--tests/test_cli.py752
-rw-r--r--tests/test_cloud.py1174
-rw-r--r--tests/test_config.py134
-rw-r--r--tests/test_ctx_hint.py98
-rw-r--r--tests/test_finish.py210
-rw-r--r--tests/test_gate.py74
-rw-r--r--tests/test_human_done.py40
-rw-r--r--tests/test_lanes.py408
-rw-r--r--tests/test_ledgers.py26
-rw-r--r--tests/test_lock.py30
-rw-r--r--tests/test_merge.py159
-rw-r--r--tests/test_migrate.py160
-rw-r--r--tests/test_model.py247
-rw-r--r--tests/test_orch.py227
-rw-r--r--tests/test_publish_snapshot.py122
-rw-r--r--tests/test_refs.py109
-rw-r--r--tests/test_res.py751
-rw-r--r--tests/test_res_io.py815
-rw-r--r--tests/test_runner.py81
-rw-r--r--tests/test_search.py78
-rw-r--r--tests/test_sessions_field.py197
-rw-r--r--tests/test_setup.py164
-rw-r--r--tests/test_start.py109
-rw-r--r--tests/test_tasks.py523
-rw-r--r--tests/test_usage.py337
29 files changed, 8364 insertions, 0 deletions
diff --git a/tests/test_areas.py b/tests/test_areas.py
new file mode 100644
index 0000000..558230d
--- /dev/null
+++ b/tests/test_areas.py
@@ -0,0 +1,319 @@
+import subprocess
+import tempfile
+import unittest
+from pathlib import Path
+
+from test_cli import Cli
+from test_claims import git
+from test_merge import IDENT
+from wflib import areas # noqa: E402 (tests/ run with the repo root on sys.path)
+
+NOTES = """\
+# CLAUDE.md
+
+## Areas
+### Parser
+- Code map: `parse_item`, `NOPE_GONE`
+- Test recipe: python3 -m unittest tests.test_p
+- Paths: src/
+- Checked: {sha}
+
+### Docs
+- Code map: `intro`
+
+## Gotchas
+- x
+"""
+
+
+def build(root: Path) -> tuple[str, str]:
+ """Repo with src/p.py + docs.md, then 3 commits touching src/; returns (sha1, notes)."""
+ git(root, "init", "-q", "-b", "master")
+ (root / "src").mkdir(exist_ok=True)
+ (root / "src" / "p.py").write_text("def parse_item(): pass\n")
+ (root / "docs.md").write_text("intro\n")
+ (root / ".gitignore").write_text(".wf/\n")
+ git(root, "add", "-A")
+ git(root, "-c", "user.name=t", "-c", "user.email=t@t", "commit", "-qm", "c1")
+ sha1 = subprocess.run(["git", "rev-parse", "--short", "HEAD"], cwd=root, capture_output=True, text=True).stdout.strip()
+ for n in "abc":
+ (root / "src" / f"{n}.txt").write_text(n)
+ git(root, "add", "-A")
+ git(root, "-c", "user.name=t", "-c", "user.email=t@t", "commit", "-qm", n)
+ return sha1, NOTES.format(sha=sha1)
+
+
+class AreasTest(unittest.TestCase):
+ def setUp(self):
+ self.tmp = tempfile.TemporaryDirectory()
+ self.root = Path(self.tmp.name).resolve()
+ self.sha1, self.notes = build(self.root)
+
+ def tearDown(self):
+ self.tmp.cleanup()
+
+ def test_parse(self):
+ a = areas.parse(NOTES.format(sha="abc1234"))
+ self.assertEqual([(x.name, x.slug, x.anchors, x.paths, x.checked) for x in a], [
+ ("Parser", "parser", ["parse_item", "NOPE_GONE"], ["src/"], "abc1234"),
+ ("Docs", "docs", ["intro"], [], None)])
+
+ def test_missing_and_commits(self):
+ parser, docs = areas.parse(self.notes)
+ self.assertEqual(areas.missing(self.root, parser), ["NOPE_GONE"])
+ self.assertEqual(areas.commits_since(self.root, parser), 3)
+ self.assertEqual(areas.missing(self.root, docs), [])
+ self.assertIsNone(areas.commits_since(self.root, docs))
+
+ def test_stale(self):
+ parser, docs = areas.parse(self.notes)
+ self.assertEqual(areas.status(self.root, parser, 20), (True, ["NOPE_GONE"], 3))
+ self.assertEqual(areas.status(self.root, docs, 20), (False, [], None))
+ fixed = areas.parse(self.notes.replace(", `NOPE_GONE`", ""))[0]
+ self.assertEqual(areas.status(self.root, fixed, 3), (True, [], 3))
+ self.assertEqual(areas.status(self.root, fixed, 4), (False, [], 3))
+
+ def test_unknown_checked_sha(self):
+ a = areas.parse(NOTES.format(sha="deadbee"))[0]
+ self.assertIsNone(areas.commits_since(self.root, a))
+
+ def test_mark(self):
+ out = areas.mark(NOTES.format(sha="abc1234"), "Docs", "f00ba44")
+ self.assertIn("### Docs\n- Code map: `intro`\n- Checked: f00ba44\n\n## Gotchas", out)
+ out = areas.mark(out, "Parser", "1111111")
+ self.assertIn("- Paths: src/\n- Checked: 1111111\n", out)
+
+
+class MatchTest(unittest.TestCase):
+ def test_matches(self):
+ parser, docs = areas.parse(NOTES.format(sha="abc1234"))
+ self.assertTrue(areas.matches(parser, "Fix src/p.py crash"))
+ self.assertTrue(areas.matches(parser, "see x.py/src/p.py"))
+ self.assertTrue(areas.matches(parser, "parse_item() loops"))
+ self.assertTrue(areas.matches(parser, "the parser area"))
+ self.assertFalse(areas.matches(parser, "parse_items and srcs"))
+ self.assertFalse(areas.matches(docs, "introduction"))
+ self.assertTrue(areas.matches(docs, "docs intro"))
+ self.assertFalse(areas.matches(parser, "Fix src/p.py crash", {"src"}))
+
+ def test_shared_paths(self):
+ a = areas.parse("## Areas\n### A\n- Paths: x.py lib/\n### B\n- Paths: ./x.py y.py\n### C\n- Paths: lib\n")
+ self.assertEqual(areas.shared_paths(a), {"x.py", "lib"})
+
+ def test_block(self):
+ text = NOTES.format(sha="abc1234")
+ self.assertEqual(areas.block(text, "Docs"), "### Docs\n- Code map: `intro`")
+ self.assertEqual(areas.block(text, "Parser").splitlines()[-1], "- Checked: abc1234")
+ self.assertEqual(areas.block(text, "X"), "")
+
+
+class CodeRootTest(Cli):
+ """code_root: area anchors / Checked / staleness live in another repo than the project."""
+
+ def setUp(self):
+ super().setUp()
+ self.code = self.root.parent / "code"
+ self.code.mkdir()
+ self.sha1, notes = build(self.code)
+ (self.root / "CLAUDE.md").write_text(notes)
+ with open(self.root / "workflow.toml", "a") as f:
+ f.write('code_root = "../code"\n')
+ git(self.root, "init", "-q", "-b", "master")
+ (self.root / "engine").mkdir()
+ (self.root / "engine" / "x.py").write_text("x")
+
+ def test_areas_use_code_root(self):
+ self.assertEqual(self.ok("areas"), f"Parser: stale · missing: NOPE_GONE · 3 commits since {self.sha1}\n"
+ "Docs: ok · no Checked\n")
+
+ def test_mark_uses_code_head(self):
+ self.ok("areas", "--mark", "Docs")
+ head = subprocess.run(["git", "rev-parse", "--short", "HEAD"], cwd=self.code, capture_output=True,
+ text=True).stdout.strip()
+ self.assertIn(f"- Checked: {head}", (self.root / "CLAUDE.md").read_text())
+
+ def test_check_uses_code_root(self):
+ out = self.ok("check")
+ self.assertIn("area Parser: anchor NOPE_GONE not found", out)
+ self.assertNotIn("Checked", out)
+ self.assertNotIn("anchor intro", out)
+
+ def test_no_uncovered_nudge(self):
+ self.assertNotIn("uncovered", self.ok("areas"))
+
+
+class CtxAreaTest(Cli):
+ tasks_text = ("# T\n\n## Pending\n\n- **t-a** [P1] (<1h): A. Fix src/p.py.\n\n"
+ "- **t-b** [P1] (<1h): B. Unrelated.\n")
+
+ def setUp(self):
+ super().setUp()
+ (self.root / "CLAUDE.md").write_text(NOTES.format(sha="abc1234"))
+
+ def test_ctx_prints_matching_area(self):
+ out = self.ok("ctx", "t-a")
+ self.assertIn("\nArea Parser (CLAUDE.md):\n### Parser\n- Code map: `parse_item`, `NOPE_GONE`\n"
+ "- Test recipe: python3 -m unittest tests.test_p\n", out)
+ self.assertNotIn("### Docs", out)
+
+ def test_ctx_no_match(self):
+ self.assertNotIn("Area ", self.ok("ctx", "t-b"))
+
+ def test_ctx_shared_path_no_match(self):
+ (self.root / "CLAUDE.md").write_text(NOTES.format(sha="abc1234").replace("- Code map: `intro`",
+ "- Code map: `intro`\n- Paths: src/"))
+ self.assertNotIn("Area ", self.ok("ctx", "t-a"))
+
+
+class AreasCliTest(Cli):
+ def setUp(self):
+ super().setUp()
+ sha1, notes = build(self.root)
+ self.sha1 = sha1
+ (self.root / "CLAUDE.md").write_text(notes)
+
+ def test_areas_output(self):
+ out = self.ok("areas")
+ self.assertEqual(out, f"Parser: stale · missing: NOPE_GONE · 3 commits since {self.sha1}\n"
+ "Docs: ok · no Checked\n")
+
+ def test_mark_then_show(self):
+ self.ok("areas", "--mark", "Docs")
+ head = subprocess.run(["git", "rev-parse", "--short", "HEAD"], cwd=self.root, capture_output=True,
+ text=True).stdout.strip()
+ self.assertEqual(self.ok("areas", "Docs"), f"Docs: ok · 0 commits since {head}\n")
+
+ def test_mark_unique_prefix(self):
+ self.ok("areas", "--mark", "Do")
+ head = subprocess.run(["git", "rev-parse", "--short", "HEAD"], cwd=self.root, capture_output=True,
+ text=True).stdout.strip()
+ self.assertEqual(self.ok("areas", "Doc"), f"Docs: ok · 0 commits since {head}\n")
+
+ def test_ambiguous_prefix(self):
+ txt = (self.root / "CLAUDE.md").read_text().replace("### Docs", "### Parser2 (x)")
+ (self.root / "CLAUDE.md").write_text(txt)
+ err = self.fails("areas", "--mark", "Par", code=2)
+ self.assertIn("ambiguous area 'Par'", err)
+ self.assertIn("Parser, Parser2 (x)", err)
+ self.ok("areas", "--mark", "Parser")
+
+ def test_unknown_area(self):
+ err = self.fails("areas", "X", code=2)
+ self.assertIn("wf: no area 'X' in CLAUDE.md (Parser, Docs)", err)
+
+ def test_check_warnings(self):
+ out = self.ok("check")
+ self.assertIn("area Parser: anchor NOPE_GONE not found", out)
+ (self.root / "CLAUDE.md").write_text(NOTES.format(sha="deadbee").replace("### Parser", "### X"))
+ self.assertIn("area X: Checked deadbee unknown", self.ok("check"))
+
+
+class AutoTaskTest(Cli):
+ tasks_text = "# T\n\n## Pending\n\n- **t-a** [P1] (<1h): A. Do.\n\n- **t-b** [P1] (<1h): B. Do.\n"
+
+ def setUp(self):
+ super().setUp()
+ sha1, notes = build(self.root)
+ (self.root / "CLAUDE.md").write_text(notes)
+
+ def test_auto_task(self):
+ out = self.ok("done", "t-a", "-m", "x")
+ self.assertIn("added t-map-parser (area map stale)", out)
+ shown = self.ok("show", "t-map-parser")
+ self.assertIn("- **t-map-parser** [P1] (<1h): Refresh Parser area map.", shown)
+ self.assertIn("Steps: wf areas Parser; fix missing anchors + test recipe from git log --stat "
+ "<Checked>..HEAD -- <paths>; wf areas --mark Parser.", shown)
+ self.assertIn("Done: wf areas Parser shows ok.", shown)
+ self.assertIn("Model: sonnet", shown)
+
+ def test_auto_task_once(self):
+ self.ok("done", "t-a", "-m", "x")
+ out = self.ok("done", "t-b", "-m", "y")
+ self.assertNotIn("added", out)
+ self.assertEqual(self.ok("list").count("t-map-parser"), 1)
+
+ def test_archived_gets_suffix(self):
+ self.ok("done", "t-a", "-m", "x")
+ out = self.ok("done", "t-map-parser", "-m", "z")
+ self.assertIn("added t-map-parser-2 (area map stale)", out)
+
+ def test_no_areas_file(self):
+ (self.root / "CLAUDE.md").unlink()
+ self.assertNotIn("added", self.ok("done", "t-a", "-m", "x"))
+
+
+class UncoveredTest(unittest.TestCase):
+ def test_uncovered_groups_two_levels(self):
+ a = areas.parse(NOTES.format(sha="abc1234"))
+ files = ["src/p.py", "lib/x/y/z.py", "lib/x/w.py", "lib/q.py", "README.md", "tests/t.py",
+ ".github/ci.yml", "TASKS.md"]
+ self.assertEqual(areas.uncovered(files, a, ["tests"]), [("lib/x", 2), ("lib", 1)])
+
+ def test_paths_globs_and_files(self):
+ a = areas.parse("## Areas\n### A\n- Paths: app/*.py tools/run.sh\n")
+ self.assertEqual(areas.uncovered(["app/m.py", "tools/run.sh", "tools/other.sh"], a, []), [("tools", 1)])
+
+ def test_no_paths_anywhere_no_nudge(self):
+ a = areas.parse("## Areas\n### A\n- Code map: `x`\n")
+ self.assertEqual(areas.uncovered(["lib/q.py"], a, []), [])
+
+
+class AnchorPathsTest(unittest.TestCase):
+ def test_anchor_folders_stand_in(self):
+ with tempfile.TemporaryDirectory() as d:
+ root = Path(d)
+ build(root)
+ a = areas.parse("## Areas\n### A\n- Code map: `parse_item`, `intro`\n### B\n- Paths: lib/\n")
+ self.assertEqual(areas.with_anchor_paths(root, a[0]).paths, ["docs.md", "src"])
+ self.assertEqual(areas.with_anchor_paths(root, a[1]).paths, ["lib/"])
+
+
+class UncoveredCliTest(Cli):
+ tasks_text = "# T\n\n## Pending\n\n- **t-a** [P1] (<1h): A. Do.\n\n- **t-b** [P1] (<1h): B. Do.\n"
+
+ def setUp(self):
+ super().setUp()
+ build(self.root)
+ notes = NOTES.format(sha="x").replace(", `NOPE_GONE`", "").replace("- Checked: x\n", "")
+ (self.root / "CLAUDE.md").write_text(notes)
+ git(self.root, "add", "-A")
+ git(self.root, "-c", "user.name=t", "-c", "user.email=t@t", "commit", "-qm", "notes")
+ (self.root / "engine" / "net").mkdir(parents=True)
+ (self.root / "engine" / "net" / "sock.py").write_text("x")
+ (self.root / "docs.md").write_text("changed")
+
+ def test_done_adds_map_task(self):
+ out = self.ok("done", "t-a", "-m", "x")
+ self.assertIn("added t-map-engine-net (uncovered area: engine/net, 1 file)", out)
+ shown = self.ok("show", "t-map-engine-net")
+ self.assertIn("- **t-map-engine-net** [P1] (<1h): Map engine/net area. "
+ "No area's Paths covers it.", shown)
+ self.assertIn("Done: wf areas lists the area ok and no longer names engine/net uncovered.", shown)
+ self.assertIn("Model: sonnet", shown)
+ self.assertNotIn("added", self.ok("done", "t-b", "-m", "y"))
+
+ def test_areas_names_uncovered(self):
+ self.assertIn("uncovered: engine/net (1 file)", self.ok("areas"))
+
+ def test_branch_diff_counts(self):
+ git(self.root, "switch", "-qc", "topic")
+ git(self.root, "add", "-A")
+ git(self.root, "-c", "user.name=t", "-c", "user.email=t@t", "commit", "-qm", "work")
+ self.assertIn("uncovered: engine/net (1 file)", self.ok("areas"))
+
+ def test_no_paths_uses_anchor_folders(self):
+ (self.root / "CLAUDE.md").write_text("## Areas\n### P\n- Code map: `parse_item`\n")
+ (self.root / "src" / "n.py").write_text("x")
+ out = self.ok("areas")
+ self.assertIn("uncovered: engine/net (1 file)", out)
+ self.assertNotIn("uncovered: src", out)
+
+ def test_ignore_config(self):
+ with open(self.root / "workflow.toml", "a") as f:
+ f.write('area_ignore = ["engine"]\n')
+ self.assertNotIn("uncovered", self.ok("areas"))
+ self.assertNotIn("added", self.ok("done", "t-a", "-m", "x"))
+
+
+if __name__ == "__main__":
+ unittest.main()
diff --git a/tests/test_batch.py b/tests/test_batch.py
new file mode 100644
index 0000000..2f0e834
--- /dev/null
+++ b/tests/test_batch.py
@@ -0,0 +1,534 @@
+import io
+import shlex
+import subprocess
+import sys
+import tempfile
+import threading
+import unittest
+from contextlib import redirect_stderr, redirect_stdout
+from pathlib import Path
+
+HERE = Path(__file__).resolve().parent.parent
+sys.path.insert(0, str(HERE))
+sys.path.insert(0, str(HERE / "tests"))
+
+import wf_res # noqa: E402
+from test_res_io import Box, hist_rec, seed_history # noqa: E402
+
+OUT = "out/wf-batch-2026-10-01-1400.md"
+
+
+class BatchBase(unittest.TestCase):
+ def setUp(self):
+ self._tmp = tempfile.TemporaryDirectory()
+ self.box = Box(Path(self._tmp.name))
+ bin_ = self.box.tmp / "bin"
+ bin_.mkdir()
+ self.claude = bin_ / "claude"
+ self.claude.write_text("#!/bin/sh\n")
+ self.claude.chmod(0o755)
+ self.box.caller["PATH"] = f"{bin_}:/usr/bin"
+
+ def tearDown(self):
+ self._tmp.cleanup()
+
+ def batch(self, *argv):
+ out, err = io.StringIO(), io.StringIO()
+ with redirect_stdout(out), redirect_stderr(err):
+ code = wf_res.batch_main(list(argv), self.box.env)
+ return code, out.getvalue(), err.getvalue()
+
+ def unit_argv(self):
+ return self.box.fake.ran("systemd-run")[0]
+
+
+class Start(BatchBase):
+ def test_starts_claude_print_unit(self):
+ code, out, err = self.batch("2", "--lanes", "fast,slow")
+ self.assertEqual((code, err), (0, ""))
+ log = self.box.env.state / "logs" / "r-1.log"
+ self.assertEqual(out, f"r-1 started; log {log}; ETA ~20:00 (estimate: not killed when over)\n"
+ f"summary: {OUT} · status: wf batch --status\n")
+ argv = self.unit_argv()
+ i = argv.index(str(self.claude))
+ self.assertEqual(argv[i - 1], "sh")
+ self.assertEqual(argv[i + 1:-1], ["-p", "--model", "opus", "--permission-mode", "auto",
+ "--permission-prompts", "none"])
+ self.assertIn("--setenv=CLAUDE_CODE_PRINT_BG_WAIT_CEILING_MS=21600000", argv)
+ self.assertIn(f"MemoryMax={4 * 1024 ** 3}", argv)
+ e = self.box.ledger().get("r-1")
+ self.assertEqual((e.title, e.est_min, e.project), ("wf-batch", 360, "proj"))
+
+ def test_prompt_from_template(self):
+ self.batch("2", "--lanes", "fast,slow")
+ prompt = self.unit_argv()[-1]
+ self.assertIn("for up to 2 tasks (lanes: fast, slow)", prompt)
+ self.assertIn(f"append one line per task to {OUT}", prompt)
+ self.assertIn("/projects/public/workflow/docs/orchestrator.md", prompt.replace(str(HERE), "/projects/public/workflow"))
+ self.assertIn("foreground", prompt)
+ self.assertNotIn("{", prompt)
+
+ def test_dry_run_writes_no_summary(self):
+ self.batch("2", "--dry-run")
+ self.assertFalse((self.box.env.root() / OUT).exists())
+
+ def test_existing_summary_not_overwritten(self):
+ f = self.box.env.root() / OUT
+ f.parent.mkdir(exist_ok=True)
+ f.write_text("live\n")
+ self.batch("2")
+ self.assertTrue(f.read_text().startswith("live\n# wf-batch"))
+
+ def test_size_lane(self):
+ code, out, err = self.batch("2", "--lanes", "fast", "--dry-run")
+ self.assertEqual((code, err), (0, ""))
+ self.assertIn("lanes: fast", out)
+
+ def test_default_lanes(self):
+ self.batch("3")
+ self.assertIn("for up to 3 tasks (lanes: every lane with ready tasks, see wf lanes)", self.unit_argv()[-1])
+
+ def test_for_sets_ceiling_and_model(self):
+ self.batch("1", "--for", "2h", "--model", "sonnet", "--mem", "2G")
+ argv = self.unit_argv()
+ self.assertIn("--setenv=CLAUDE_CODE_PRINT_BG_WAIT_CEILING_MS=7200000", argv)
+ self.assertEqual(argv[argv.index("--model") + 1], "sonnet")
+ self.assertIn(f"MemoryMax={2 * 1024 ** 3}", argv)
+
+ def test_no_claude(self):
+ self.box.caller["PATH"] = "/nonexistent"
+ code, out, err = self.batch("2")
+ self.assertEqual((code, out, err), (1, "", "wf: claude not found on PATH\n"))
+ self.assertEqual(self.box.fake.ran("systemd-run"), [])
+
+ def test_bad_count(self):
+ code, _, err = self.batch("0")
+ self.assertEqual(code, 2)
+ self.assertIn("N must be ≥ 1", err)
+
+ def test_summary_header_written(self):
+ code, out, err = self.batch("2", "--lanes", "fast")
+ self.assertEqual(code, 0)
+ text = (self.box.env.root() / OUT).read_text()
+ self.assertEqual(text, "# wf-batch proj 2026-10-01 14:00 (N=2, lanes: fast)\n")
+
+ def test_dry_run(self):
+ code, out, err = self.batch("2", "--dry-run")
+ self.assertEqual((code, err), (0, ""))
+ self.assertIn("CLAUDE_CODE_PRINT_BG_WAIT_CEILING_MS=21600000 wf res run --mem 4G --for 6h --title wf-batch -- ", out)
+ self.assertIn(f"{self.claude} -p --model opus", out)
+ self.assertEqual(self.box.fake.ran("systemd-run"), [])
+
+ def test_mem_scales_with_n(self):
+ for args, mem in ((("4",), "6G"), (("1",), "4G"), (("9",), "8G"), (("4", "--mem", "3G"), "3G")):
+ code, out, err = self.batch(*args, "--dry-run")
+ self.assertIn(f"wf res run --mem {mem} --for", out, args)
+
+ def test_defaults_from_history(self):
+ # peaks 2,3,4 → p95 4 ×1.15 = 4.6 GB; durations 20,30,40 → p90 40 ×1.5 = 60 min; ceiling stays 6h
+ seed_history(self.box, [hist_rec(f"r-{i}", title="wf-batch", peak=p, minutes=m)
+ for i, (p, m) in enumerate(((2.0, 20.0), (3.0, 30.0), (4.0, 40.0)))])
+ code, out, err = self.batch("4", "--dry-run")
+ self.assertEqual((code, err), (0, ""))
+ self.assertIn("CEILING_MS=21600000 wf res run --mem 4.6G --for 60m --title wf-batch -- ", out)
+ out = self.batch("4", "--mem", "3G", "--for", "2h", "--dry-run")[1]
+ self.assertIn("CEILING_MS=7200000 wf res run --mem 3G --for 2h --title wf-batch -- ", out)
+
+ def test_busy_exit_3(self):
+ self.box.available_gb = 1
+ code, out, _ = self.batch("2")
+ self.assertEqual(code, 3)
+ self.assertIn("busy", out)
+
+ def test_force_past_unused_claim(self):
+ self.box.wf("run", "--mem", "10G", "--for", "40m", "--title", "big", "--", "x")
+ code, out, _ = self.batch("2")
+ self.assertEqual(code, 3)
+ code, out, err = self.batch("2", "--force")
+ self.assertEqual(code, 0, (out, err))
+ self.assertIn("r-2 started", out)
+
+
+class Status(BatchBase):
+ def test_none(self):
+ code, out, err = self.batch("--status")
+ self.assertEqual((code, out, err), (0, "no batch in this project\n", ""))
+
+ def test_running_and_summary(self):
+ self.batch("2")
+ md = self.box.tmp / "proj" / OUT
+ with md.open("a") as f:
+ f.write("t-a sonnet done abc123\n")
+ (md.parent / "wf-batch-2026-09-30-0100.md").write_text("old\n")
+ code, out, _ = self.batch("--status")
+ self.assertEqual(code, 0)
+ lines = out.splitlines()
+ self.assertTrue(lines[0].startswith('r-1 proj "wf-batch" running'), lines[0])
+ self.assertIn(f"log {self.box.env.state / 'logs' / 'r-1.log'}", lines[1])
+ self.assertEqual(lines[2:], [f"{OUT}:", "# wf-batch proj 2026-10-01 14:00 (N=2, lanes: every lane with ready tasks, see wf lanes)",
+ "t-a sonnet done abc123"])
+
+ def test_orch_log_progress(self):
+ self.batch("2")
+ out_dir = self.box.tmp / "proj" / "out"
+ (out_dir / "wf-orch.log").write_text(
+ "2026-10-01 13:00 fast sonnet t-old done aaa\n2026-10-01 14:05 fast sonnet t-new done bbb\n")
+ _, out, _ = self.batch("--status")
+ self.assertIn("finished since batch start", out)
+ self.assertIn("t-new done bbb", out)
+ self.assertNotIn("t-old", out)
+
+ def test_stop_creates_file_and_status_reports(self):
+ code, out, err = self.batch("--stop")
+ self.assertEqual((code, err), (0, ""))
+ f = self.box.tmp / "proj" / "out" / "wf-batch.stop"
+ self.assertTrue(f.exists())
+ self.assertIn("out/wf-batch.stop", out)
+ code, out, _ = self.batch("--status")
+ self.assertIn("stop requested: out/wf-batch.stop", out)
+
+ def test_stale_stop_cleared_on_start(self):
+ f = self.box.tmp / "proj" / "out" / "wf-batch.stop"
+ f.parent.mkdir(exist_ok=True)
+ f.touch()
+ code, out, err = self.batch("2")
+ self.assertEqual(code, 0)
+ self.assertIn("stale out/wf-batch.stop cleared", err)
+
+ def test_stop_consumed_on_exit(self):
+ f = self.box.tmp / "proj" / "out" / "wf-batch.stop"
+ f.parent.mkdir(exist_ok=True)
+ orig = wf_res.cmd_run
+
+ def fake(env, cfg, args):
+ f.touch()
+ return 0
+ wf_res.cmd_run = fake
+ try:
+ code, _, _ = self.batch("2")
+ finally:
+ wf_res.cmd_run = orig
+ self.assertEqual(code, 0)
+ self.assertFalse(f.exists())
+
+ def test_prompt_has_stop_check(self):
+ code, out, _ = self.batch("2", "--dry-run")
+ self.assertIn("out/wf-batch.stop", out)
+ self.assertIn("stopped: stop file", out)
+
+ def test_prompt_cloud_lane(self):
+ code, out, _ = self.batch("2", "--dry-run")
+ self.assertIn("wf orch pick cloud", out)
+ self.assertIn("never run `wf cloud pull`", out)
+ self.assertIn("sidecar pulls every 10 min", out)
+ self.assertIn(f"-- {self.claude} -p", out) # not a cloud project: plain claude -p job
+
+ def test_prompt_stale_branch_once(self):
+ code, out, _ = self.batch("2", "--dry-run")
+ self.assertIn("pre-existing branch", out)
+ self.assertIn("ONCE", out)
+
+ def test_prompt_checks_stop_before_every_spawn(self):
+ code, out, _ = self.batch("2", "--dry-run")
+ self.assertIn("Before EVERY spawn", out)
+ self.assertIn(f"test -e {self.box.tmp / 'proj' / 'out' / 'wf-batch.stop'}", out)
+ self.assertNotIn("Each round: first", out)
+ code, out, _ = self.batch("--stop")
+ self.assertIn("before its next spawn", out)
+
+ def test_n_required_without_status(self):
+ code, _, err = self.batch()
+ self.assertEqual(code, 2)
+
+
+PREP_TASKS = """\
+## Awaiting your decision
+
+## Pending
+
+- **t-ready** [P1] (<1h): Ready.
+ Done: works.
+
+- **t-low** [P3] (<1h): Low, no Done.
+
+- **t-hi** [P1] (1h): High, no Done.
+
+- **t-mid** [P2] (1h): Mid, no Done.
+
+## Needs human
+
+## Deferred
+"""
+
+
+class Prep(BatchBase):
+ def setUp(self):
+ super().setUp()
+ root = self.box.env.root()
+ (root / "workflow.toml").write_text('format = 1\ntasks = "TASKS.md"\narchive = "archive.md"\n')
+ (root / "TASKS.md").write_text(PREP_TASKS)
+ (root / "archive.md").write_text("# Archive\n")
+
+ def test_prompt_lists_targets(self):
+ code, out, err = self.batch("2", "--prep")
+ self.assertEqual((code, err), (0, ""))
+ prompt = self.unit_argv()[-1]
+ self.assertIn("Tasks: t-hi t-mid.", prompt)
+ self.assertIn("never implement", prompt)
+ self.assertIn("wf set <id> --done", prompt)
+ self.assertIn("wf add -s awaiting", prompt)
+ self.assertIn(f"to {OUT}", prompt)
+ for ph in ("never invert recorded original behaviour", "wf add --parent <id> slices",
+ "concrete files/runs", "[opus: <why>]", "[Model: <m>]"):
+ self.assertIn(ph, prompt)
+ self.assertNotIn("{", prompt)
+ self.assertEqual((self.box.env.root() / OUT).read_text(),
+ "# wf-batch proj 2026-10-01 14:00 (prep K=2: t-hi t-mid)\n")
+ self.assertEqual(self.box.ledger().get("r-1").title, "wf-batch")
+
+ def test_lanes_filter(self):
+ self.batch("5", "--prep", "--lanes", "fast")
+ self.assertIn("Tasks: t-low.", self.unit_argv()[-1])
+
+ def test_prep_alias_all(self):
+ out, err = io.StringIO(), io.StringIO()
+ with redirect_stdout(out), redirect_stderr(err):
+ code = wf_res.prep_main(["all"], self.box.env)
+ self.assertEqual((code, err.getvalue()), (0, ""))
+ self.assertIn("Tasks: t-hi t-mid", self.unit_argv()[-1])
+
+ def test_nothing_to_prep_starts_nothing(self):
+ code, out, err = self.batch("3", "--prep", "--lanes", "nolane")
+ self.assertEqual((code, out, err), (0, "prep: no pending task without Done (lanes: nolane); nothing started\n", ""))
+ self.assertEqual(self.box.fake.ran("systemd-run"), [])
+ self.assertFalse((self.box.env.root() / OUT).exists())
+
+ def test_prep_mem_default_small(self):
+ for args, mem in ((("3", "--prep"), "1G"), (("3", "--prep", "--mem", "2G"), "2G")):
+ code, out, err = self.batch(*args, "--dry-run")
+ self.assertIn(f"wf res run --mem {mem} --for", out, args)
+
+ def test_dry_run(self):
+ code, out, err = self.batch("1", "--prep", "--dry-run")
+ self.assertEqual((code, err), (0, ""))
+ self.assertIn("--title wf-batch -- ", out)
+ self.assertIn("t-hi", out)
+ self.assertEqual(self.box.fake.ran("systemd-run"), [])
+
+ def test_no_project(self):
+ (self.box.env.root() / "workflow.toml").unlink()
+ code, out, err = self.batch("1", "--prep")
+ self.assertEqual(code, 1)
+ self.assertTrue(err.startswith("wf: "), err)
+
+
+class Dispatch(unittest.TestCase):
+ def test_wf_forwards_batch(self):
+ r = subprocess.run([sys.executable, str(HERE / "wf.py"), "batch", "-h"], capture_output=True, text=True)
+ self.assertEqual(r.returncode, 0)
+ self.assertIn("usage: wf batch", r.stdout)
+
+ def test_listed_in_wf_help(self):
+ r = subprocess.run([sys.executable, str(HERE / "wf.py"), "-h"], capture_output=True, text=True)
+ self.assertIn("batch", r.stdout)
+
+
+if __name__ == "__main__":
+ unittest.main()
+
+
+FAKE_WF = """import sys, pathlib
+d = pathlib.Path(sys.argv[0]).parent
+with open(d / "calls.log", "a") as f:
+ f.write(" ".join(sys.argv[1:]) + "\\n")
+if sys.argv[1:3] == ["cloud", "pull"] and (d / "clear").exists():
+ for rec in (d / ".wf" / "cloud").glob("*.json"):
+ rec.unlink()
+if sys.argv[1:3] == ["cloud", "pull"] and not (d / "pulled").exists():
+ (d / "pulled").touch()
+ print("t-a: done (session_1, $0.40 usage): merged abc1234")
+ print("archive session_1 failed: x; archive by hand (wf cloud archive --ended)")
+ print("report: commit abc1234")
+ print("t-b: running (session_2, 12 min)")
+ print("t-c: handback (session_3, $0.10 usage): cloud handback: stuck")
+ print("t-d: pulled by another wf cloud pull, skipped")
+ sys.exit(1)
+"""
+
+
+class Sidecar(unittest.TestCase):
+ """wf batch on a cloud = true project: the job runs the orchestrator under the pull sidecar."""
+
+ def setUp(self):
+ self._tmp = tempfile.TemporaryDirectory()
+ self.root = Path(self._tmp.name)
+ (self.root / "out").mkdir()
+ self.summary = self.root / OUT
+ self.summary.write_text("# wf-batch\n")
+ self.fake = self.root / "fakewf.py"
+ self.fake.write_text(FAKE_WF)
+
+ def tearDown(self):
+ self._tmp.cleanup()
+
+ def run_sidecar(self, child_s, every=0.2, rc=7, **kw):
+ child = [sys.executable, "-c", f"import time, sys; time.sleep({child_s}); sys.exit({rc})"]
+ out = io.StringIO()
+ with redirect_stdout(out):
+ code = wf_res.sidecar(child, self.root, self.summary, every, [sys.executable, str(self.fake)], **kw)
+ return code, out.getvalue()
+
+ def records(self, *ids):
+ d = self.root / ".wf" / "cloud"
+ d.mkdir(parents=True)
+ for i in ids:
+ (d / f"{i}.json").write_text("{}")
+
+ def tail_line(self):
+ return [ln[6:] for ln in self.summary.read_text().splitlines()[1:] if "tail" in ln]
+
+ def test_tail_ends_on_empty_records(self):
+ self.records("t-b")
+ (self.root / "clear").touch()
+ code, out = self.run_sidecar(0.05, every=0.1)
+ self.assertEqual(code, 7)
+ self.assertEqual(len([c for c in self.calls() if c.startswith("cloud pull")]), 1)
+ self.assertIn("orch post t-a cloud --result done --no-pick --commit abc1234", self.calls())
+ self.assertEqual(self.tail_line(), ["cloud tail ended: no cloud records left; 1 pulls; left: none (sidecar)"])
+
+ def test_tail_ends_on_stop_file(self):
+ self.records("t-b", "t-e")
+ threading.Timer(0.5, (self.root / wf_res.STOP_FILE).touch).start()
+ code, _ = self.run_sidecar(0.05, every=0.2)
+ pulls = [c for c in self.calls() if c.startswith("cloud pull")]
+ self.assertGreaterEqual(len(pulls), 1) # pulled in the tail until the stop file
+ self.assertEqual(self.tail_line(), [f"cloud tail ended: stop file; {len(pulls)} pulls; left: t-b t-e (sidecar)"])
+ self.assertEqual(code, 7)
+
+ def test_tail_ends_on_cap(self):
+ self.records("t-b")
+ shrunk = []
+ code, out = self.run_sidecar(0.05, every=0.1, cap=0.35, shrink=lambda: shrunk.append(1) or "shrunk")
+ self.assertEqual((code, shrunk), (7, [1])) # reservation shrunk once, entering the tail
+ self.assertIn("shrunk", out)
+ pulls = [c for c in self.calls() if c.startswith("cloud pull")]
+ self.assertGreaterEqual(len(pulls), 2)
+ self.assertTrue(all(c.startswith("cloud pull") or c.startswith("orch post") for c in self.calls())) # no picks
+ self.assertEqual(self.tail_line(),
+ [f"cloud tail ended: {0.35 / 3600:g}h cap; {len(pulls)} pulls; left: t-b (sidecar)"])
+
+ def test_no_tail_without_records(self):
+ shrunk = []
+ code, _ = self.run_sidecar(0.05, every=0.1, shrink=lambda: shrunk.append(1) or "")
+ self.assertEqual((code, shrunk, self.calls(), self.tail_line()), (7, [], [], []))
+
+ def test_shrink_reservation(self):
+ b = BatchBase("run")
+ b.setUp()
+ try:
+ b.batch("4", "--mem", "6G")
+ msg = wf_res.shrink_reservation(b.box.env, "r-1")
+ e = b.box.ledger().get("r-1")
+ self.assertEqual((e.mem_gb, e.cpus, e.state), (0.2, 1, "running"))
+ self.assertIn("r-1 reservation 6", msg)
+ self.assertIn("not running", wf_res.shrink_reservation(b.box.env, "r-9"))
+ finally:
+ b.tearDown()
+
+ def calls(self):
+ f = self.root / "calls.log"
+ return f.read_text().splitlines() if f.exists() else []
+
+ def test_pulls_and_posts_while_child_lives(self):
+ code, out = self.run_sidecar(1.0)
+ self.assertEqual(code, 7) # the orchestrator's exit code
+ calls = self.calls()
+ pulls = [c for c in calls if c.startswith("cloud pull")]
+ self.assertGreaterEqual(len(pulls), 2) # repeats every interval while the child lives
+ self.assertEqual(pulls[0], f"cloud pull --all --project {self.root}")
+ self.assertEqual([c for c in calls if c.startswith("orch post")],
+ ["orch post t-a cloud --result done --no-pick --commit abc1234",
+ "orch post t-c cloud --result handback --no-pick"])
+ lines = self.summary.read_text().splitlines()[1:]
+ self.assertEqual([ln[6:] for ln in lines], ["cloud t-a done abc1234 (sidecar pull)",
+ "cloud t-c handback - (sidecar pull)"])
+ self.assertIn("t-a: done", out) # pull output lands in the job log
+
+ def test_stops_with_child(self):
+ code, _ = self.run_sidecar(0.1, every=5, rc=0)
+ self.assertEqual((code, self.calls()), (0, []))
+
+ def test_stop_file_ends_pulls(self):
+ (self.root / wf_res.STOP_FILE).touch()
+ code, out = self.run_sidecar(0.7)
+ self.assertEqual((code, self.calls()), (7, []))
+ self.assertIn("sidecar: stop file, no more pulls", out)
+
+ def test_cloud_project_job_wraps_claude(self):
+ b = BatchBase("run")
+ b.setUp()
+ try:
+ root = b.box.env.root()
+ (root / "workflow.toml").write_text('format = 1\ntasks = "TASKS.md"\narchive = "a.md"\ncloud = true\n')
+ code, out, _ = b.batch("2", "--dry-run")
+ self.assertEqual(code, 0)
+ argv = shlex.split(out.split(" -- ", 1)[1])
+ self.assertEqual(argv[:9], [sys.executable, str(HERE / "wf_res.py"), "batch-sidecar", "--root", str(root),
+ "--summary", OUT, "--every", "600"])
+ self.assertEqual(argv[9:12], ["--", str(b.claude), "-p"])
+ finally:
+ b.tearDown()
+
+ def test_main_entry(self):
+ r = subprocess.run([sys.executable, str(HERE / "wf_res.py"), "batch-sidecar", "--root", str(self.root),
+ "--summary", OUT, "--every", "5", "--wf", f"{sys.executable} {self.fake}", "--",
+ sys.executable, "-c", "import sys; sys.exit(3)"], capture_output=True, text=True)
+ self.assertEqual((r.returncode, r.stderr), (3, ""))
+
+
+LOG_ROWS = "".join(f"2026-10-01T1{i}:00:00 slow opus t-{i} done abc {m}m00s\n" for i, m in enumerate((20, 25, 31)))
+
+
+class Fit(BatchBase):
+ def log(self, text):
+ (self.box.env.root() / "out").mkdir(exist_ok=True)
+ (self.box.env.root() / "out" / "wf-orch.log").write_text(text)
+
+ def test_left_shrinks_k_and_sets_for(self):
+ self.log(LOG_ROWS) # p90 of 20,25,31 = 31 min
+ code, out, err = self.batch("4", "--left", "1h40m", "--dry-run")
+ self.assertEqual((code, err), (0, ""))
+ self.assertIn("fit: 3 of 4 tasks (left 1h40m, task p90 31m from 3 runs)\n", out)
+ self.assertIn("CEILING_MS=6000000 wf res run --mem 4.5G --for 1h40m --title wf-batch -- ", out)
+ self.assertIn("for up to 3 tasks", out)
+
+ def test_left_default_30m(self):
+ code, out, err = self.batch("4", "--left", "45m", "--dry-run")
+ self.assertIn("fit: 1 of 4 tasks (left 45m, task p90 30m default (0 runs))\n", out)
+ self.assertIn("--for 45m", out)
+
+ def test_left_too_short_starts_nothing(self):
+ self.log(LOG_ROWS)
+ code, out, err = self.batch("4", "--left", "30m")
+ self.assertEqual((code, err), (0, ""))
+ self.assertEqual(out, "fit: 0 of 4 tasks (left 30m, task p90 31m from 3 runs); nothing started\n")
+ self.assertEqual(self.box.fake.ran("systemd-run"), [])
+
+ def test_prompt_has_deadline_check(self):
+ code, out, _ = self.batch("2", "--for", "2h", "--dry-run")
+ self.assertIn("wf batch --time-left 2026-10-01T16:00+00:00", out)
+ self.assertIn("stopped: deadline", out)
+
+ def test_time_left(self):
+ self.log(LOG_ROWS)
+ code, out, err = self.batch("--time-left", "2026-10-01T15:00+00:00")
+ self.assertEqual((code, out, err), (0, "time left 1h, task p90 31m (3 runs): spawn\n", ""))
+ code, out, err = self.batch("--time-left", "2026-10-01T14:30+00:00")
+ self.assertEqual((code, out), (0, "time left 30m, task p90 31m (3 runs): stop\n"))
+ code, out, err = self.batch("--time-left", "2026-10-01T13:00+00:00")
+ self.assertEqual((code, out), (0, "time left 0m, task p90 31m (3 runs): stop\n"))
+
+ def test_time_left_bad(self):
+ code, out, err = self.batch("--time-left", "soon")
+ self.assertEqual(code, 1)
+ self.assertIn("wf: --time-left", err)
diff --git a/tests/test_check.py b/tests/test_check.py
new file mode 100644
index 0000000..e4f3d3f
--- /dev/null
+++ b/tests/test_check.py
@@ -0,0 +1,304 @@
+import os
+import sys
+import time
+import tempfile
+import unittest
+from pathlib import Path
+
+sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
+from wflib import check as K
+from wflib import config as C
+
+TOML = 'format = 1\ntasks = "TASKS.md"\narchive = "tasks/archive.md"\ndocs = ["DESIGN.md", "docs/"]\n'
+ANCHORS = '[anchors]\nindex = "DESIGN.md"\nindex_section = "Subsystems"\nspecs = "docs/specs"\n'
+ARCHIVE = "# Archive (newest first)\n\n- 2026-09-01 **t-done** Done thing — ok\n"
+DESIGN = """\
+# Design
+
+## Subsystems
+
+### terrain
+
+Heightmap. [spec](docs/specs/terrain.md#terrain)
+
+### input
+
+Keys.
+
+## Notes
+
+Free text.
+"""
+SPEC = '# Terrain spec\n\n<a id="terrain"></a>\n## Terrain\n\nText.\n'
+CLEAN = """\
+# Tasks
+
+## Awaiting your decision
+
+- **a-key**: Key needed. Which one?
+
+## Pending
+
+- **t-one** [P1] (1h): One.
+ - After: [[t-done]]
+ Ref: DESIGN.md#terrain, docs/specs/terrain.md
+
+- **t-two** [P2] (5h) (blocked: [[a-key]]): Two.
+ - After: [[t-one]]
+
+## Needs human
+
+## Deferred
+"""
+
+
+class Base(unittest.TestCase):
+ toml = TOML
+
+ def setUp(self):
+ self.tmp = tempfile.TemporaryDirectory()
+ self.root = Path(self.tmp.name).resolve()
+ (self.root / "tasks").mkdir()
+ (self.root / "docs" / "specs").mkdir(parents=True)
+ (self.root / "workflow.toml").write_text(self.toml)
+ (self.root / "tasks" / "archive.md").write_text(ARCHIVE)
+ (self.root / "DESIGN.md").write_text(DESIGN)
+ (self.root / "docs" / "specs" / "terrain.md").write_text(SPEC)
+ self.tasks(CLEAN)
+
+ def tearDown(self):
+ self.tmp.cleanup()
+
+ def tasks(self, text):
+ (self.root / "TASKS.md").write_text(text)
+
+ def pending(self, items):
+ self.tasks("## Awaiting your decision\n\n- **a-key**: Key. Q?\n\n## Pending\n\n" + items +
+ "\n## Needs human\n\n## Deferred\n")
+
+ def run_check(self):
+ errors, warnings = K.check(C.load(self.root))
+ return [str(e) for e in errors], [str(w) for w in warnings]
+
+ def errors(self):
+ return self.run_check()[0]
+
+
+class TasksCheckTest(Base):
+ def test_clean(self):
+ self.assertEqual(self.run_check(), ([], []))
+
+ def test_after_prose_and_deferred(self):
+ self.tasks("## Awaiting your decision\n\n## Pending\n\n"
+ "- **t-a** [P1] (1h): A.\n After: when ready [[t-d]]\n\n"
+ "## Needs human\n\n## Deferred\n\n- **t-d** [P3] (1h): D.\n")
+ w = self.run_check()[1]
+ self.assertEqual(len(w), 2, w)
+ self.assertTrue(any("has prose" in x and ":6:" in x for x in w), w)
+ self.assertTrue(any("'t-d' is deferred" in x for x in w), w)
+
+ def test_after_clean_links(self):
+ self.pending("- **t-a** [P1] (1h): A.\n After: [[t-done]], [[t-b]]\n\n- **t-b** [P1] (1h): B.\n")
+ self.assertEqual(self.run_check()[1], [])
+
+ def test_duplicate_id(self):
+ self.pending("- **t-a** [P1] (1h): A.\n\n- **t-a** [P1] (1h): Again.\n")
+ self.assertEqual(self.errors(), ["TASKS.md:9: t-a: duplicate id (also line 7)"])
+
+ def test_id_reused_from_archive(self):
+ self.pending("- **t-done** [P1] (1h): A.\n")
+ self.assertEqual(self.errors(), ["TASKS.md:7: t-done: id already used in the archive (ids are never reused)"])
+
+ def test_bad_grammar(self):
+ self.pending("- **T-Foo** [P1] (1h): A.\n")
+ self.assertEqual(self.errors(), ["TASKS.md:7: T-Foo: bad id (want t-… or a-…, lowercase a-z 0-9 -)"])
+
+ def test_wrong_kind_for_section(self):
+ self.pending("- **a-x**: A. Q?\n")
+ self.assertEqual(self.errors(), ["TASKS.md:7: a-x: a- item in Pending (a- items belong in Awaiting, t- elsewhere)"])
+
+ def test_task_in_awaiting(self):
+ self.tasks("## Awaiting your decision\n\n- **t-x** [P1] (1h): A.\n\n## Pending\n\n## Needs human\n\n## Deferred\n")
+ self.assertEqual(self.errors(),
+ ["TASKS.md:3: t-x: t- item in Awaiting your decision (a- items belong in Awaiting, t- elsewhere)"])
+
+ def test_missing_priority_and_effort(self):
+ self.pending("- **t-a**: A.\n")
+ self.assertEqual(self.errors(), ["TASKS.md:7: t-a: no priority [P0]-[P3]",
+ "TASKS.md:7: t-a: no effort (want <1h, 1h, 5h, 10h, 100h)"])
+
+ def test_bad_priority_and_effort(self):
+ self.pending("- **t-a** [P7] (2h): A.\n")
+ self.assertEqual(self.errors(), ["TASKS.md:7: t-a: priority P7 (want P0-P3)",
+ "TASKS.md:7: t-a: effort '2h' (want <1h, 1h, 5h, 10h, 100h)"])
+
+ def test_dangling_link(self):
+ self.pending("- **t-a** [P1] (1h): A.\n - see [[t-none]]\n")
+ self.assertEqual(self.errors(), ["TASKS.md:8: [[t-none]]: no such id in TASKS or archive"])
+
+ def test_dangling_link_in_prose_section(self):
+ self.tasks(CLEAN + "\n## Notes\n\nSee [[a-gone]] and `[[t-example]]`.\n")
+ self.assertEqual(self.errors(), ["TASKS.md:22: [[a-gone]]: no such id in TASKS or archive"])
+
+ def test_blocked_on_missing_awaiting_item(self):
+ self.pending("- **t-a** [P1] (1h) (blocked: [[a-none]]): A.\n")
+ self.assertEqual(self.errors(), ["TASKS.md:7: [[a-none]]: no such id in TASKS or archive",
+ "TASKS.md:7: t-a: blocked on 'a-none', which is not an open Awaiting item"])
+
+ def test_after_cycle(self):
+ self.pending("- **t-a** [P1] (1h): A.\n - After: [[t-b]]\n\n- **t-b** [P1] (1h): B.\n - After: [[t-a]]\n")
+ self.assertIn("TASKS.md:7: t-a: After: cycle t-a → t-b → t-a", self.errors())
+
+ def test_item_before_its_open_dependency(self):
+ self.pending("- **t-a** [P1] (1h): A.\n - After: [[t-b]]\n\n- **t-b** [P1] (1h): B.\n")
+ self.assertEqual(self.errors(), ["TASKS.md:7: t-a: placed before 't-b', which it is After:"])
+
+ def test_after_bare_id_is_flagged(self):
+ self.pending("- **t-a** [P1] (1h): A.\n After: t-b\n\n- **t-b** [P1] (1h): B.\n")
+ self.assertEqual(self.errors(), ["TASKS.md:8: t-a: After: 't-b' is not a link (want [[t-b]])"])
+
+ def test_slices_bare_id_is_flagged(self):
+ self.pending("- **t-a** [P1] (1h): A.\n - Slices: [[t-b]], t-c\n\n- **t-b** [P1] (1h): B.\n")
+ self.assertEqual(self.errors(), ["TASKS.md:8: t-a: Slices: 't-c' is not a link (want [[t-c]])"])
+
+ def test_ref_path_missing(self):
+ self.pending("- **t-a** [P1] (1h): A.\n Ref: docs/none.md\n")
+ self.assertEqual(self.errors(), ["TASKS.md:8: t-a: Ref 'docs/none.md' does not exist"])
+
+ def test_ref_anchor_missing(self):
+ self.pending("- **t-a** [P1] (1h): A.\n Ref: DESIGN.md#nope\n")
+ self.assertEqual(self.errors(), ["TASKS.md:8: t-a: Ref 'DESIGN.md#nope': no such anchor"])
+
+ def test_old_numbered_items(self):
+ self.pending("1. **[P1] Old** (Effort: 1h) — goal.\n - Steps\n")
+ self.assertEqual(self.errors(), ["TASKS.md:7: old numbered item (wf migrate)"])
+
+ def test_malformed_header_has_line(self):
+ self.pending("- **t-ok** [P1] (1h): Fine.\n\n- **t-x** [P1] (1h) no colon\n")
+ self.assertEqual(self.errors(),
+ ["TASKS.md:9: t-x: bad header: want '- **id** [Pn] (effort) [(status)]: Title. Goal.'"])
+
+ def test_item_after_prose_is_flagged(self):
+ self.pending("- **t-a** [P1] (1h): A.\nstray\n- **t-b** [P1] (1h): B.\n")
+ self.assertEqual(self.errors(), ["TASKS.md:9: t-b: item after a flush-left prose line (line 8): indent or move the prose"])
+
+ def test_missing_section(self):
+ self.tasks("## Pending\n\n## Needs human\n\n## Deferred\n")
+ self.assertEqual(self.errors(), ["TASKS.md: no '## Awaiting your decision' section"])
+
+ def test_number_refs_warn(self):
+ self.pending("- **t-a** [P1] (1h): A, see #12.\n")
+ self.assertEqual(self.run_check(), ([], ["TASKS.md:7: '#12': number ref (tasks have ids: [[t-…]])"]))
+
+ def test_empty_progress_note_warns(self):
+ self.pending("- **t-a** [P1] (1h) (in progress: ): A.\n")
+ self.assertEqual(self.run_check(), ([], ["TASKS.md:7: t-a: in progress without a branch or note"]))
+
+ def test_doc_link_to_unknown_id(self):
+ (self.root / "docs" / "note.md").write_text("# N\n\nSee [[t-one]], [[t-done]], [[t-lost]], [[wiki-page]].\n")
+ self.assertEqual(self.errors(), ["docs/note.md:3: [[t-lost]]: no such id in TASKS or archive"])
+
+
+class StaleAwaitingTest(Base):
+ def commit(self, date):
+ import os
+ import subprocess
+ env = {**os.environ, "GIT_AUTHOR_DATE": date, "GIT_COMMITTER_DATE": date,
+ "GIT_AUTHOR_NAME": "t", "GIT_AUTHOR_EMAIL": "t@t", "GIT_COMMITTER_NAME": "t", "GIT_COMMITTER_EMAIL": "t@t"}
+ for cmd in (["init", "-q"], ["add", "-A"], ["commit", "-q", "-m", "x"]):
+ subprocess.run(["git", "-C", str(self.root), *cmd], check=True, env=env, capture_output=True)
+
+ def test_old_unreferenced_awaiting_item_warns(self):
+ self.tasks(CLEAN.replace("- **a-key**: Key needed. Which one?",
+ "- **a-key**: Key needed. Which one?\n\n- **a-old**: Old question. Still open?"))
+ self.commit("2020-01-01T00:00:00")
+ errors, warnings = self.run_check()
+ self.assertEqual(errors, [])
+ self.assertEqual(len(warnings), 1)
+ self.assertRegex(warnings[0], r"^TASKS.md:7: a-old: waiting \d+ days, no task references it$")
+
+ def test_fresh_item_does_not_warn(self):
+ self.tasks(CLEAN.replace("- **a-key**: Key needed. Which one?",
+ "- **a-key**: Key needed. Which one?\n\n- **a-new**: New question. Open?"))
+ import datetime
+ self.commit(datetime.datetime.now().isoformat(timespec="seconds"))
+ self.assertEqual(self.run_check(), ([], []))
+
+
+class ConfigCheckTest(Base):
+ def test_missing_tasks_file(self):
+ (self.root / "TASKS.md").unlink()
+ self.assertEqual(self.errors(), ["workflow.toml: tasks 'TASKS.md' does not exist"])
+
+ def test_missing_archive_and_docs(self):
+ (self.root / "tasks" / "archive.md").unlink()
+ (self.root / "DESIGN.md").unlink()
+ self.pending("- **t-a** [P1] (1h): A.\n")
+ self.assertEqual(self.errors(), ["workflow.toml: archive 'tasks/archive.md' does not exist",
+ "workflow.toml: docs 'DESIGN.md' does not exist"])
+
+
+class AnchorsCheckTest(Base):
+ toml = TOML + ANCHORS
+
+ def test_clean(self):
+ self.assertEqual(self.run_check(), ([], []))
+
+ def test_task_anchor_not_in_index_section(self):
+ self.pending("- **t-a** [P1] (1h): A.\n Ref: DESIGN.md#notes\n")
+ self.assertEqual(self.errors(), ["TASKS.md:8: t-a: Ref 'DESIGN.md#notes' is not a heading under 'Subsystems'"])
+
+ def test_duplicate_index_slug(self):
+ (self.root / "DESIGN.md").write_text(DESIGN.replace("### input", "### Terrain"))
+ self.assertEqual(self.errors(), ["DESIGN.md:9: two 'Subsystems' headings give anchor 'terrain' (also line 5)"])
+
+ def test_index_link_to_missing_file_and_anchor(self):
+ (self.root / "DESIGN.md").write_text(DESIGN.replace("Keys.", "Keys. [a](docs/none.md) [b](docs/specs/terrain.md#nope) [c](https://x.y/z)"))
+ self.assertEqual(self.errors(), ["DESIGN.md:11: link 'docs/none.md': file does not exist",
+ "DESIGN.md:11: link 'docs/specs/terrain.md#nope': no such anchor"])
+
+ def test_explicit_spec_anchor_without_index_entry(self):
+ (self.root / "docs" / "specs" / "combat.md").write_text('# Combat\n\n<a id="combat"></a>\n## Combat\n')
+ self.assertEqual(self.errors(), ["docs/specs/combat.md:3: explicit anchor 'combat' has no heading under 'Subsystems' in DESIGN.md"])
+
+
+class ProblemTest(unittest.TestCase):
+ def test_key_ignores_line(self):
+ a = K.Problem("TASKS.md", 7, "t-a", "x")
+ b = K.Problem("TASKS.md", 9, "t-a", "x")
+ self.assertEqual(a.key, b.key)
+ self.assertEqual(str(a), "TASKS.md:7: t-a: x")
+ self.assertEqual(str(K.Problem("workflow.toml", None, "", "y")), "workflow.toml: y")
+
+
+if __name__ == "__main__":
+ unittest.main()
+
+
+class MergedToolWorktreeTest(unittest.TestCase):
+ def git(self, root, *a):
+ import subprocess
+ subprocess.run(["git", "-C", str(root), "-c", "user.name=t", "-c", "user.email=t@t", *a],
+ check=True, capture_output=True)
+
+ def test_flags_only_merged_worktrees(self):
+ with tempfile.TemporaryDirectory() as d:
+ root = Path(d) / "tool"
+ root.mkdir()
+ self.git(root, "init", "-b", "master")
+ self.git(root, "commit", "--allow-empty", "-m", "a")
+ self.git(root, "worktree", "add", str(Path(d) / "fresh"), "-b", "fresh")
+ self.git(root, "worktree", "add", str(Path(d) / "done"), "-b", "done")
+ self.git(root, "worktree", "add", str(Path(d) / "open"), "-b", "open")
+ self.git(Path(d) / "open", "commit", "--allow-empty", "-m", "b")
+ self.git(root, "commit", "--allow-empty", "-m", "c")
+ old = time.time() - 3 * 3600
+ for n in ("done", "open"):
+ os.utime(Path(d) / n / ".git", (old, old))
+ got = K.merged_tool_worktrees(root)
+ names = sorted(p.message.split("branch ")[1].split(")")[0] for p in got)
+ self.assertEqual(names, ["done"]) # fresh (<2h) skipped
+
+ def test_not_git(self):
+ with tempfile.TemporaryDirectory() as d:
+ self.assertEqual(K.merged_tool_worktrees(Path(d)), [])
diff --git a/tests/test_claims.py b/tests/test_claims.py
new file mode 100644
index 0000000..e6b5c5d
--- /dev/null
+++ b/tests/test_claims.py
@@ -0,0 +1,182 @@
+import json
+import os
+import subprocess
+import sys
+import unittest
+from pathlib import Path
+
+sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
+from wflib import tasks as T
+from wflib import lanes as L
+from test_cli import Cli
+from test_model import LANES
+
+FOUR = LANES.replace("## Needs human", "- **t-four** [P3] (1h): Four.\n\n## Needs human")
+
+
+class HeldPickTest(unittest.TestCase):
+ def test_pick_skips_held(self):
+ item, skipped = L.pick(T.parse(FOUR), set(), L.DEFAULT_LANES, "1h", None, "opus",
+ held={"t-one": "opus session uds:/a"})
+ self.assertEqual((item.id, [(i.id, why) for i, why in skipped]),
+ ("t-two", [("t-one", "in progress by opus session uds:/a")]))
+
+
+class ClaimCliTest(Cli):
+ tasks_text = FOUR
+
+ def env(self, name, pid):
+ sock = self.root / f"{name}.sock"
+ sock.write_text("")
+ return {"CLAUDE_CODE_MESSAGING_SOCKET": str(sock), "CLAUDE_PID": str(pid), "CLAUDE_CODE_SESSION_ID": name}
+
+ def dead_pid(self):
+ p = subprocess.Popen(["true"])
+ p.wait()
+ return p.pid
+
+ def claim(self, id):
+ return json.loads((self.root / ".wf" / "claims" / f"{id}.json").read_text())
+
+ def test_progress_claims_and_other_session_skips(self):
+ a, b = self.env("a", os.getpid()), self.env("b", os.getppid())
+ self.ok("next", "--lane", "fast", "--as", "opus", "--brief", env=a)
+ self.ok("status", "t-three", "progress", "x", env=a)
+ c = self.claim("t-three")
+ self.assertEqual({k: c[k] for k in ("pid", "socket", "lane", "model")},
+ {"pid": os.getpid(), "socket": str(self.root / "a.sock"), "lane": "fast", "model": "opus"})
+ self.assertEqual(self.ok("next", "--lane", "fast", "--as", "opus", "--brief", env=a).splitlines()[0],
+ "- **t-three** [P2] (<1h) (in progress: x): Three.")
+ code, out, err = self.wf("next", "--lane", "fast", "--as", "opus", env=b)
+ self.assertEqual((code, err), (1, "wf: nothing pickable for fast (opus) in Pending\n"))
+ self.assertIn(f"===== Skipped =====\n- t-three: in progress by opus session uds:{self.root / 'a.sock'}\n", out)
+ out = self.ok("next", "--lane", "slow", "--as", "opus", env=b)
+ self.assertIn("===== Next task =====\n- **t-one**", out)
+
+ def test_dead_claim_ignored(self):
+ self.ok("status", "t-three", "progress", "x", env=self.env("a", self.dead_pid()))
+ self.assertEqual(self.ok("next", "--lane", "fast", "--as", "opus", "--brief", env=self.env("b", os.getpid())).splitlines()[0],
+ "- **t-three** [P2] (<1h) (in progress: x): Three.")
+
+ def test_claim_without_in_progress_status_ignored(self):
+ self.ok("status", "t-three", "progress", "x", env=self.env("a", os.getppid()))
+ self.ok("status", "t-three", "clear", env=self.env("b", os.getpid()))
+ self.assertFalse((self.root / ".wf" / "claims" / "t-three.json").exists())
+ self.assertEqual(self.ok("next", "--lane", "fast", "--as", "opus", "--brief", env=self.env("b", os.getpid())).splitlines()[0],
+ "- **t-three** [P2] (<1h): Three.")
+
+ def test_done_and_blocked_release(self):
+ a = self.env("a", os.getpid())
+ self.ok("status", "t-three", "progress", "x", env=a)
+ self.ok("done", "t-three", "-m", "ok", env=a)
+ self.assertFalse((self.root / ".wf" / "claims" / "t-three.json").exists())
+ self.ok("status", "t-four", "progress", "x", env=a)
+ self.ok("add", "-s", "awaiting", "Q?", env=a)
+ self.ok("status", "t-four", "blocked", "a-q", env=a)
+ self.assertFalse((self.root / ".wf" / "claims" / "t-four.json").exists())
+
+ def test_no_env_no_claim(self):
+ self.ok("status", "t-three", "progress", "x", env={"CLAUDE_PID": "", "CLAUDE_CODE_MESSAGING_SOCKET": ""})
+ self.assertFalse((self.root / ".wf" / "claims").exists())
+ self.assertFalse((self.root / ".wf" / "sessions").exists())
+
+ def test_same_lane_live_session_warns_and_keeps_registry(self):
+ a, b = self.env("a", os.getppid()), self.env("b", os.getpid())
+ self.ok("next", "--lane", "fast", "--as", "opus", "--brief", env=a)
+ out = self.ok("next", "--lane", "fast", "--as", "opus", "--brief", env=b)
+ self.assertEqual(out.splitlines()[0],
+ f"another live fast session holds this lane: uds:{self.root / 'a.sock'} "
+ "(claims keep tasks apart; tell the owner if unintended)")
+ reg = json.loads((self.root / ".wf" / "sessions" / "fast.json").read_text())
+ self.assertEqual(reg["socket"], str(self.root / "a.sock"))
+ self.ok("next", "--lane", "fast", "--as", "opus", "--brief", env=a) # own re-register: no warning
+ self.assertNotIn("another live", self.ok("next", "--lane", "fast", "--as", "opus", "--brief", env=a))
+
+ def test_lanes_unregister_drops_own_record(self):
+ a, b = self.env("a", os.getpid()), self.env("b", os.getppid())
+ self.ok("next", "--lane", "fast", "--as", "opus", "--brief", env=a)
+ self.ok("next", "--lane", "slow", "--as", "sonnet", "--brief", env=b)
+ out = self.ok("lanes", "--unregister", env=a)
+ self.assertFalse((self.root / ".wf" / "sessions" / "fast.json").exists())
+ self.assertTrue((self.root / ".wf" / "sessions" / "slow.json").exists())
+ self.assertNotIn(str(self.root / "a.sock"), out)
+
+
+def git(cwd, *args):
+ subprocess.run(["git", "-C", str(cwd), *args], check=True, capture_output=True,
+ env={**os.environ, "GIT_AUTHOR_NAME": "t", "GIT_AUTHOR_EMAIL": "t@t", "GIT_COMMITTER_NAME": "t",
+ "GIT_COMMITTER_EMAIL": "t@t"})
+
+
+class WorktreeTest(Cli):
+ tasks_text = FOUR
+
+ def setUp(self):
+ super().setUp()
+ git(self.root, "init", "-q", "-b", "master")
+ (self.root / ".gitignore").write_text(".worktrees/\n")
+ git(self.root, "add", "-A")
+ git(self.root, "commit", "-qm", "init")
+ self.wt = self.root / ".worktrees" / "fast"
+ git(self.root, "worktree", "add", "-q", str(self.wt), "-b", "fast/t-three")
+
+ def test_worktree_writes_main_tree(self):
+ none = {"CLAUDE_CODE_MESSAGING_SOCKET": ""}
+ self.wf("status", "t-three", "progress", "x", project=False, cwd=self.wt, env=none)
+ self.assertIn("(in progress: x)", (self.root / "TASKS.md").read_text())
+ self.assertNotIn("in progress", (self.wt / "TASKS.md").read_text())
+ code, out, err = self.wf("done", "t-four", "-m", "ok", project=False, cwd=self.wt / "docs", env=none)
+ self.assertEqual(code, 0, err)
+ self.assertIn("**t-four**", self.archive())
+ self.assertNotIn("t-four", (self.wt / "tasks" / "archive.md").read_text())
+
+ def test_done_in_worktree_prints_merge_steps(self):
+ code, out, err = self.wf("done", "t-four", "-m", "ok", project=False, cwd=self.wt,
+ env={"CLAUDE_CODE_MESSAGING_SOCKET": ""})
+ self.assertEqual(code, 0, err)
+ self.assertTrue(out.endswith(
+ "worktree mode (branch fast/t-three), after verify: commit your code here (explicit paths), then:\n"
+ " wf merge (rebase, ff-merge into master, commit TASKS.md tasks/archive.md, push home; "
+ "conflict → git rebase master, resolve, verify, wf merge again)\n"), out)
+ self.assertNotIn("worktree mode", self.ok("done", "t-three", "-m", "ok"))
+
+ def env(self, name, pid):
+ sock = self.root / f"{name}.sock"
+ sock.write_text("")
+ return {"CLAUDE_CODE_MESSAGING_SOCKET": str(sock), "CLAUDE_PID": str(pid), "CLAUDE_CODE_SESSION_ID": name}
+
+ def test_next_says_worktree_mode_when_two_sessions_live(self):
+ son, me = self.env("s", os.getppid()), self.env("o", os.getpid())
+ self.assertNotIn("Multi-session", self.ok("next", "--lane", "fast", "--as", "opus", env=me))
+ self.ok("next", "--lane", "slow", "--as", "sonnet", "--brief", env=son)
+ out = self.ok("next", "--lane", "fast", "--as", "opus", env=me)
+ self.assertIn("===== Multi-session =====\n"
+ "2 live sessions here: work in your lane's worktree, never on master:\n"
+ " cd .worktrees/fast && git switch -c fast/<task> master (wf there writes this TASKS.md)\n\n", out)
+ self.ok("next", "--lane", "slow", "--as", "sonnet", "--brief", env=son)
+ out = self.ok("next", "--lane", "slow", "--as", "sonnet", env=son)
+ self.assertIn(" git worktree add .worktrees/slow -b slow/<task> master (wf there writes this TASKS.md)\n", out)
+ code, out, err = self.wf("next", "--lane", "fast", "--as", "opus", project=False, cwd=self.wt, env=me)
+ self.assertIn("===== Multi-session =====\n2 live sessions here: you are in worktree .worktrees/fast "
+ "(branch fast/t-three); wf writes the main tree's TASKS.md\n", out)
+
+
+if __name__ == "__main__":
+ unittest.main()
+
+
+class StaleTest(ClaimCliTest):
+ def test_stale_list_and_clear(self):
+ self.ok("status", "t-three", "progress", "x", env=self.env("a", self.dead_pid()))
+ self.ok("status", "t-four", "progress", "y", env=self.env("b", os.getpid()))
+ self.assertEqual(self.ok("list", "--stale").splitlines()[0].split()[0], "t-three")
+ self.assertNotIn("t-four", self.ok("list", "--stale"))
+ self.assertIn("cleared t-three", self.ok("status", "--clear-stale"))
+ self.assertEqual(self.ok("list", "--progress").count("prog"), 1)
+ self.assertIn("t-four", self.ok("list", "--progress"))
+ self.assertFalse((self.root / ".wf" / "claims" / "t-three.json").exists())
+
+ def test_no_claim_listed(self):
+ self.ok("status", "t-three", "progress", "x")
+ (self.root / ".wf" / "claims" / "t-three.json").unlink(missing_ok=True)
+ self.assertIn("t-three", self.ok("list", "--stale"))
diff --git a/tests/test_cli.py b/tests/test_cli.py
new file mode 100644
index 0000000..43135c9
--- /dev/null
+++ b/tests/test_cli.py
@@ -0,0 +1,752 @@
+import datetime
+import os
+import subprocess
+import sys
+import tempfile
+import unittest
+from pathlib import Path
+
+HERE = Path(__file__).resolve().parent.parent
+WF = HERE / "wf.py"
+sys.path.insert(0, str(HERE))
+
+TOML = ('format = 1\ntasks = "TASKS.md"\narchive = "tasks/archive.md"\ndocs = ["DESIGN.md", "docs/"]\n'
+ 'verify = ["make test"]\ndone = ["update CATALOG"]\nledgers = ".superpowers/sdd"\n')
+TASKS = """\
+# Tasks — demo
+
+## Awaiting your decision
+
+- **a-key**: Key needed. Which one?
+
+## Pending
+
+- **t-one** [P1] (1h) (in progress: master): First thing. Do it.
+ - Steps: a
+ Ref: DESIGN.md#terrain, docs/plan.md
+
+- **t-two** [P2] (5h) (blocked: [[a-key]]): Second thing.
+
+- **t-three** [P2] (<1h): Third thing. Later.
+ - After: [[t-one]]
+
+## Needs human
+
+- **t-play** [P1] (<1h): Play-test feel.
+ - [ ] keyboard works
+
+## Deferred
+
+- **t-far** [P3] (100h): Far future.
+"""
+ARCHIVE = "# Archive (newest first)\n\n- 2026-09-01 **t-done** Done thing — ok\n- 2026-08-01 **t-older** Older thing — fine\n"
+TODAY = datetime.date.today().isoformat()
+
+
+class Cli(unittest.TestCase):
+ tasks_text = TASKS
+ toml = TOML
+
+ def setUp(self):
+ self.tmp = tempfile.TemporaryDirectory()
+ self.root = Path(self.tmp.name).resolve() / "demo"
+ (self.root / "tasks").mkdir(parents=True)
+ (self.root / "docs").mkdir()
+ (self.root / "workflow.toml").write_text(self.toml)
+ (self.root / "TASKS.md").write_bytes(self.tasks_text.encode())
+ (self.root / "tasks" / "archive.md").write_text(ARCHIVE)
+ (self.root / "DESIGN.md").write_text("# Design\n\n## terrain\n\nHeightmap.\n")
+ (self.root / "docs" / "plan.md").write_text("# The Plan\n\n**Goal:** Ship it.\n")
+ self.inbox = self.root.parent / "inbox.md"
+
+ def tearDown(self):
+ self.tmp.cleanup()
+
+ def wf(self, *args, stdin="", cwd=None, project=True, env=None, hints=False):
+ cmd = [sys.executable, str(WF), *(["--project", str(self.root)] if project else []), *args]
+ full = {**os.environ, "WF_INBOX": str(self.inbox), "WF_ROOT": str(self.root.parent),
+ "WF_TOOL_ROOT": str(self.root.parent / "no-tool"), "CLAUDE_CONFIG_DIR": str(self.root.parent / "no-claude"), **(env or {})}
+ r = subprocess.run(cmd, input=stdin, capture_output=True, text=True, cwd=cwd or self.root, env=full, timeout=30)
+ err = r.stderr if hints else "".join(l for l in r.stderr.splitlines(True) if not l.startswith("hint:"))
+ return r.returncode, r.stdout, err
+
+ def ok(self, *args, **kw):
+ code, out, err = self.wf(*args, **kw)
+ self.assertEqual((code, err), (0, ""), out + err)
+ return out
+
+ def fails(self, *args, code=1, **kw):
+ got, out, err = self.wf(*args, **kw)
+ self.assertEqual(got, code, out + err)
+ self.assertNotIn("Traceback", err)
+ return err
+
+ def tasks(self):
+ return (self.root / "TASKS.md").read_text()
+
+ def archive(self):
+ return (self.root / "tasks" / "archive.md").read_text()
+
+ def item(self, id):
+ return self.ok("show", id)
+
+
+class ReadTest(Cli):
+ def test_list_default_is_pending(self):
+ self.assertEqual(self.ok("list"),
+ "t-one P1 1h prog opus slow First thing\n"
+ "t-two P2 5h blkd opus fast Second thing\n"
+ "t-three P2 <1h - opus fast Third thing\n"
+ "pending 3 · human 1 · awaiting 1 · deferred 1\n")
+
+ def test_list_all(self):
+ self.assertEqual(self.ok("list", "-s", "all"),
+ "== Awaiting your decision\n"
+ "a-key -- - - - - Key needed\n"
+ "== Pending\n"
+ "t-one P1 1h prog opus slow First thing\n"
+ "t-two P2 5h blkd opus fast Second thing\n"
+ "t-three P2 <1h - opus fast Third thing\n"
+ "== Needs human\n"
+ "t-play P1 <1h - opus fast Play-test feel\n"
+ "== Deferred\n"
+ "t-far P3 100h - opus fast Far future\n"
+ "pending 3 · human 1 · awaiting 1 · deferred 1\n")
+
+ def first_column(self, *args):
+ return [l.split()[0] for l in self.ok("list", *args).splitlines()[:-1]]
+
+ def test_list_filters(self):
+ self.assertEqual(self.first_column("-p", "1"), ["t-one"])
+ self.assertEqual(self.first_column("--ready"), ["t-one"])
+ self.assertEqual(self.first_column("--blocked"), ["t-two"])
+ self.assertEqual(self.first_column("--progress"), ["t-one"])
+ self.assertEqual(self.first_column("--ref", "DESIGN.md"), ["t-one"])
+ self.assertEqual(self.first_column("--ref", "DESIGN.md#terrain"), ["t-one"])
+ self.assertEqual(self.first_column("--ref", "DESIGN.md#other"), [])
+ self.assertEqual(self.first_column("-n", "2"), ["t-one", "t-two"])
+ self.assertEqual(self.first_column("-s", "deferred"), ["t-far"])
+
+ def test_list_cuts_long_titles_to_100_columns(self):
+ self.ok("set", "t-far", "--title", "x" * 150)
+ line = self.ok("list", "-s", "deferred").splitlines()[0]
+ self.assertEqual(len(line), 100)
+ self.assertTrue(line.endswith("xxx…"))
+
+ def test_show(self):
+ self.assertEqual(self.ok("show", "t-three", "a-key"),
+ "- **t-three** [P2] (<1h): Third thing. Later.\n - After: [[t-one]]\n\n"
+ "- **a-key**: Key needed. Which one?\n")
+
+ def test_show_by_unique_prefix(self):
+ self.assertEqual(self.ok("show", "t-thr"), "- **t-three** [P2] (<1h): Third thing. Later.\n - After: [[t-one]]\n")
+
+ def test_show_ambiguous_prefix(self):
+ self.assertEqual(self.fails("show", "t-t"), "wf: 't-t' matches t-three, t-two\n")
+
+ def test_show_unknown(self):
+ self.assertRegex(self.fails("show", "t-zzz"), r"^wf: unknown id 't-zzz' \(nearest: .*\)\n$")
+
+ def test_next(self):
+ self.assertEqual(self.ok("next", "--as", "opus"),
+ "===== Awaiting your decision (mention, don't block) =====\n"
+ "- a-key: Key needed. Which one?\n\n"
+ "===== Needs human (not picked) =====\n"
+ "- t-play: Play-test feel\n\n"
+ "===== Lanes =====\n"
+ "fast: 0 pickable · 2 waiting · no session\n"
+ "slow: 1 pickable · 0 waiting · no session → orchestrator or owner: start one (wf next --lane slow)\n\n"
+ "===== Next task =====\n"
+ "- **t-one** [P1] (1h) (in progress: master): First thing. Do it.\n"
+ " - Steps: a\n"
+ " Ref: DESIGN.md#terrain, docs/plan.md\n\n"
+ "===== DESIGN.md#terrain (line 3) =====\n## terrain\n\nHeightmap.\n\n"
+ "docs/plan.md: The Plan — Ship it.\n")
+
+ def test_next_brief(self):
+ self.assertEqual(self.ok("next", "--as", "opus", "--brief"),
+ "- **t-one** [P1] (1h) (in progress: master): First thing. Do it.\n"
+ " - Steps: a\n Ref: DESIGN.md#terrain, docs/plan.md\n")
+
+ def test_next_lists_what_it_skipped(self):
+ self.ok("move", "t-one", "deferred")
+ self.ok("prio", "t-far", "3")
+ code, out, err = self.wf("next", "--as", "opus")
+ self.assertEqual(code, 1)
+ self.assertIn("===== Skipped =====\n- t-two: blocked: a-key\n- t-three: after: t-one\n", out)
+ self.assertEqual(err, "wf: nothing pickable for all lanes (opus) in Pending\n")
+
+ def test_next_shows_ledgers_and_inbox(self):
+ sdd = self.root / ".superpowers" / "sdd" / "p"
+ sdd.mkdir(parents=True)
+ (self.root / "docs" / "p.md").write_text("# P\n**Goal:** g\n### Task 1: A\n### Task 2: B\n### Task 3: C\n")
+ (sdd / "progress.md").write_text("# SDD ledger — plan: docs/p.md\nTask 1: complete (x)\nTask 2: Ruling: y\nTask 2: complete (z)\n")
+ self.inbox.write_text("- 2026-09-29 demo bug: x @abc\n- 2026-09-29 demo idea: y @abc\n")
+ out = self.ok("next", "--as", "opus")
+ self.assertIn("===== In-flight plans =====\n- docs/p.md: done 1, 2; resume at Task 3 (task-start PLAN 3) "
+ "(1 rulings; ledger .superpowers/sdd/p/progress.md)\n\n", out)
+ self.assertTrue(out.endswith("\nworkflow inbox: 2 reports (triage: workflow session)\n"), out)
+
+ def test_ctx_item(self):
+ self.assertEqual(self.ok("ctx", "t-one"),
+ "- **t-one** [P1] (1h) (in progress: master): First thing. Do it.\n"
+ " - Steps: a\n"
+ " Ref: DESIGN.md#terrain, docs/plan.md\n\n"
+ "Section: Pending\n"
+ "Needed by: t-three\n"
+ "Verify (run before wf finish): make test\n\n"
+ "===== DESIGN.md#terrain (line 3) =====\n## terrain\n\nHeightmap.\n\n"
+ "docs/plan.md: The Plan — Ship it.\n")
+
+ def test_ctx_dependencies_and_links(self):
+ out = self.ok("ctx", "t-three")
+ self.assertIn("Section: Pending\nAfter: t-one (open, Pending)\n", out)
+ self.assertEqual(self.ok("ctx", "a-key").split("\n\n", 1)[1], "Section: Awaiting your decision\nBlocks: t-two\n")
+
+ def test_ctx_archived_id(self):
+ self.assertEqual(self.ok("ctx", "t-done"), "done: - 2026-09-01 **t-done** Done thing — ok\n")
+
+ def test_ctx_anchor(self):
+ self.assertEqual(self.ok("ctx", "DESIGN.md#terrain"),
+ "===== DESIGN.md#terrain (line 3) =====\n## terrain\n\nHeightmap.\n\n"
+ "Tasks: t-one\n")
+
+ def test_search(self):
+ self.assertEqual(self.ok("search", "thing", "later"),
+ "t-three Third thing. Later.\n"
+ "t-one First thing. Do it.\n"
+ "t-two Second thing.\n"
+ "tasks/archive.md:3 t-done Done thing — ok\n"
+ "tasks/archive.md:4 t-older Older thing — fine\n")
+
+ def test_search_docs(self):
+ self.assertEqual(self.ok("search", "--docs", "height"), "DESIGN.md:3 terrain — Heightmap.\n")
+
+ def test_search_nothing(self):
+ self.assertEqual(self.fails("search", "zzz"), "wf: no hits\n")
+
+ def test_log(self):
+ self.assertEqual(self.ok("log", "-n", "1"), "- 2026-09-01 **t-done** Done thing — ok\n")
+ self.assertEqual(self.ok("log", "older"), "- 2026-08-01 **t-older** Older thing — fine\n")
+
+ def test_check_clean(self):
+ self.assertEqual(self.ok("check"), "OK: 0 errors · 0 warnings\n")
+
+ def test_check_errors(self):
+ (self.root / "TASKS.md").write_text(TASKS.replace("[[t-one]]", "[[t-gone]]"))
+ code, out, err = self.wf("check")
+ self.assertEqual((code, err), (1, ""))
+ self.assertEqual(out, "ERROR: TASKS.md:16: [[t-gone]]: no such id in TASKS or archive\n1 errors · 0 warnings\n")
+
+ def test_projects(self):
+ other = self.root.parent / "group" / "other"
+ (other / "tasks").mkdir(parents=True)
+ (other / "workflow.toml").write_text('format = 1\ntasks = "TASKS.md"\narchive = "tasks/archive.md"\n')
+ (other / "TASKS.md").write_text("## Awaiting your decision\n\n## Pending\n\n- **t-x** [P9] (1h): X.\n\n## Needs human\n\n## Deferred\n")
+ (other / "tasks" / "archive.md").write_text("# Archive\n")
+ hidden = self.root.parent / ".gate"
+ hidden.mkdir()
+ (hidden / "workflow.toml").write_text(TOML)
+ tool = self.root.parent / "workflow" # a wf checkout: its templates/ is no project
+ (tool / "wflib").mkdir(parents=True)
+ (tool / "templates").mkdir()
+ (tool / "wf.py").write_text("")
+ (tool / "templates" / "workflow.toml").write_text(TOML)
+ self.assertEqual(self.ok("projects", project=False),
+ "demo pending 3 human 1 awaiting 1 errors 0 next: t-one First thing\n"
+ "group/other pending 1 human 0 awaiting 0 errors 1 next: t-x X\n")
+
+ def test_outside_a_project(self):
+ err = self.fails("list", project=False, cwd=self.root.parent)
+ self.assertRegex(err, r"^wf: no workflow.toml in .* or above: not a wf project \(wf init makes one\)\n$")
+
+ def test_broken_config(self):
+ (self.root / "workflow.toml").write_text(TOML + "verfy = []\n")
+ self.assertEqual(self.fails("list"), "wf: workflow.toml: unknown key 'verfy'\n")
+
+ def test_finds_project_from_subfolder(self):
+ self.assertIn("t-one", self.ok("list", project=False, cwd=self.root / "docs"))
+
+ def test_no_command_is_usage_error(self):
+ self.fails(code=2)
+
+
+class AddTest(Cli):
+ def test_add_places_by_priority_and_prints_header(self):
+ out = self.ok("add", "Cave seams. Close the slit.", "-p", "1", "-e", "<1h")
+ self.assertEqual(out, "- **t-cave-seams** [P1] (<1h): Cave seams. Close the slit.\n")
+ self.assertEqual(self.first_ids(), ["t-one", "t-cave-seams", "t-two", "t-three"])
+
+ def first_ids(self, section="pending"):
+ return [l.split()[0] for l in self.ok("list", "-s", section).splitlines()[:-1]]
+
+ def test_add_with_options(self):
+ self.ok("add", "Docs pass", "-p", "2", "-e", "5h", "--sessions", "owner", "--id", "t-docs",
+ "--after", "t-one,t-done", "--ref", "docs/plan.md,DESIGN.md#terrain")
+ self.assertEqual(self.item("t-docs"),
+ "- **t-docs** [P2] (5h): Docs pass.\n"
+ " Sessions: owner\n"
+ " - After: [[t-one]], [[t-done]]\n"
+ " Ref: docs/plan.md, DESIGN.md#terrain\n")
+
+ def test_add_cloud(self):
+ self.ok("add", "Code fix", "-p", "1", "-e", "1h", "--done", "tests pass", "--model", "sonnet", "--cloud", "yes",
+ "--id", "t-cf")
+ self.assertEqual(self.item("t-cf"), "- **t-cf** [P1] (1h): Code fix.\n Done: tests pass\n"
+ " Model: sonnet\n Cloud: yes\n")
+ self.ok("add", "Gui fix", "-p", "1", "-e", "1h", "--cloud", "no", "--id", "t-gf")
+ self.assertIn(" Cloud: no\n", self.item("t-gf"))
+
+ def test_add_body_from_stdin(self):
+ self.ok("add", "With body", "-p", "3", "-e", "1h", "-b", stdin="- Steps: a\n - sub\n- Done: b\n")
+ self.assertEqual(self.item("t-with-body"),
+ "- **t-with-body** [P3] (1h): With body.\n - Steps: a\n - sub\n - Done: b\n")
+
+ def test_add_to_other_sections(self):
+ self.assertEqual(self.ok("add", "Which colour. Red or blue?", "-s", "awaiting"),
+ "- **a-which-colour** : Which colour. Red or blue?\n".replace(" :", ":"))
+ self.ok("add", "Feel check", "-p", "2", "-e", "<1h", "-s", "human")
+ self.assertEqual(self.first_ids("human"), ["t-play", "t-feel-check"])
+
+ def test_add_slice(self):
+ self.ok("add", "Map scaffold", "-e", "1h", "--parent", "t-far")
+ self.ok("add", "Combat rows", "-e", "1h", "--parent", "t-far")
+ self.assertEqual(self.ok("show", "t-far", "t-far-1", "t-far-2"),
+ "- **t-far** [P3] (100h): Far future.\n - Slices: [[t-far-1]], [[t-far-2]]\n\n"
+ "- **t-far-1** [P3] (1h): Map scaffold.\n\n"
+ "- **t-far-2** [P3] (1h): Combat rows.\n - After: [[t-far-1]]\n")
+ self.assertEqual(self.first_ids("deferred"), ["t-far", "t-far-1", "t-far-2"])
+
+ def test_add_slice_skips_deferred_previous_slice(self):
+ self.ok("add", "Map scaffold", "-e", "1h", "--parent", "t-far")
+ self.ok("add", "Combat rows", "-e", "1h", "--parent", "t-far", "-s", "pending")
+ self.ok("add", "Loot rows", "-e", "1h", "--parent", "t-far", "-s", "pending")
+ self.assertEqual(self.ok("show", "t-far-2", "t-far-3"),
+ "- **t-far-2** [P3] (1h): Combat rows.\n\n"
+ "- **t-far-3** [P3] (1h): Loot rows.\n - After: [[t-far-2]]\n")
+
+ def test_add_named_slice_chains_after_previous(self):
+ self.ok("add", "Map scaffold", "-e", "1h", "--parent", "t-far", "--id", "t-map")
+ self.ok("add", "Combat rows", "-e", "1h", "--parent", "t-far", "--id", "t-combat")
+ self.assertEqual(self.ok("show", "t-far", "t-combat"),
+ "- **t-far** [P3] (100h): Far future.\n - Slices: [[t-map]], [[t-combat]]\n\n"
+ "- **t-combat** [P3] (1h): Combat rows.\n - After: [[t-map]]\n")
+ self.assertEqual(self.fails("done", "t-far", "-m", "x"), "wf: 't-far' has open slices: t-map, t-combat\n")
+ self.ok("done", "t-map", "-m", "x")
+ self.assertIn("last slice of t-far: finish it", self.ok("done", "t-combat", "-m", "x"))
+
+ def test_add_leading_id_in_text(self):
+ self.assertEqual(self.ok("add", "a-colour: Which colour? Red or blue?", "-s", "awaiting"),
+ "- **a-colour**: Which colour? Red or blue?\n")
+ self.assertEqual(self.ok("add", "t-seams: Cave seams. Close them", "-p", "1", "-e", "1h"),
+ "- **t-seams** [P1] (1h): Cave seams. Close them.\n")
+
+ def test_add_help_says_id_works_with_parent(self):
+ self.assertIn("--parent", self.ok("add", "-h").split("--id ID", 2)[2].split("--after")[0])
+
+ def test_add_block(self):
+ self.ok("add", "-", stdin="- **t-blk** [P0] (1h): Block. Goal.\n - Steps: a\n Done: x\n")
+ self.assertEqual(self.item("t-blk"), "- **t-blk** [P0] (1h): Block. Goal.\n - Steps: a\n Done: x\n")
+ self.assertEqual(self.first_ids()[0], "t-blk")
+
+ def test_add_block_in_old_format_is_refused(self):
+ self.assertEqual(self.fails("add", "-", stdin="3. **[P2] Old** (Effort: 1h) — goal.\n"),
+ "wf: old numbered format: write '- **id** [Pn] (effort): Title. Goal.'\n")
+
+ def test_add_needs_priority_and_effort(self):
+ self.assertIn("-p", self.fails("add", "No prio", "-e", "1h", code=2))
+ self.assertIn("-e", self.fails("add", "No effort", "-p", "1", code=2))
+ self.assertEqual(self.tasks(), TASKS)
+
+ def test_add_title_without_letters(self):
+ self.assertEqual(self.fails("add", "???", "-p", "1", "-e", "1h"),
+ "wf: no id can be made from title '???': give one with --id\n")
+
+ def test_add_same_title_twice(self):
+ self.ok("add", "Cave seams", "-p", "1", "-e", "1h")
+ self.assertEqual(self.ok("add", "Cave seams", "-p", "1", "-e", "1h"), "- **t-cave-seams-2** [P1] (1h): Cave seams.\n")
+
+ def test_add_refused_when_result_fails_check(self):
+ self.assertEqual(self.fails("add", "Bad ref", "-p", "1", "-e", "1h", "--ref", "docs/none.md"),
+ "wf: refused: nothing written, the change adds problems:\n TASKS.md:14: t-bad-ref: Ref 'docs/none.md' does not exist\n")
+ self.assertEqual(self.tasks(), TASKS)
+
+ def test_dry_run(self):
+ out = self.ok("add", "Cave seams", "-p", "1", "-e", "1h", "--dry-run")
+ self.assertIn("+- **t-cave-seams** [P1] (1h): Cave seams.\n", out)
+ self.assertIn("--- TASKS.md\n+++ TASKS.md (new)\n", out)
+ self.assertEqual(self.tasks(), TASKS)
+
+
+class DoneTest(Cli):
+ def test_done_builtin_checklist_empty_cfg(self):
+ self.toml = TOML.replace('verify = ["make test"]\ndone = ["update CATALOG"]\n', '')
+ self.setUp()
+ out = self.ok("done", "t-one", "-m", "x")
+ self.assertEqual(out, "done: t-one → tasks/archive.md\nfast work now pickable: t-three (no session: tell the owner)\nchecklist:\n - re-learned anything (>3 greps to find)? one anchor line → that area's code map\n")
+
+ def test_done(self):
+ out = self.ok("done", "t-one", "-m", "shipped\nwith care")
+ self.assertEqual(out, "done: t-one → tasks/archive.md\nfast work now pickable: t-three (no session: tell the owner)\nverify:\n make test\nchecklist:\n - re-learned anything (>3 greps to find)? one anchor line → that area's code map\n - update CATALOG\n")
+ self.assertNotIn("t-one**", self.tasks())
+ self.assertEqual(self.archive().splitlines()[2], f"- {TODAY} **t-one** First thing — shipped with care")
+ self.assertEqual(self.ok("check"), "OK: 0 errors · 0 warnings\n")
+ self.assertEqual(self.ok("next", "--as", "opus", "--brief").splitlines()[0], "- **t-three** [P2] (<1h): Third thing. Later.")
+
+ def test_done_needs_entry(self):
+ self.assertEqual(self.fails("done", "t-one", code=2), "wf: done: -m \"<entry>\" is required for a single task\n")
+
+ def test_done_several_uses_goals(self):
+ self.ok("done", "t-one", "t-three")
+ self.assertEqual(self.archive().splitlines()[2:4],
+ [f"- {TODAY} **t-three** Third thing — Later.", f"- {TODAY} **t-one** First thing — Do it."])
+
+ def test_done_refuses_parent_with_open_slices(self):
+ self.ok("add", "Map scaffold", "-e", "1h", "--parent", "t-far")
+ self.assertEqual(self.fails("done", "t-far", "-m", "x"), "wf: 't-far' has open slices: t-far-1\n")
+
+ def test_done_awaiting_item_unblocks(self):
+ self.assertEqual(self.ok("done", "a-key"), "removed: a-key\nunblocked: t-two\n")
+ self.assertEqual(self.item("t-two"), "- **t-two** [P2] (5h): Second thing.\n")
+ self.assertEqual(self.archive(), ARCHIVE)
+
+ def test_done_unknown(self):
+ self.assertRegex(self.fails("done", "t-on", "-m", "x"), r"^wf: unknown id 't-on' \(nearest: t-one")
+ self.assertEqual(self.archive(), ARCHIVE)
+
+
+class EditTest(Cli):
+ def order(self):
+ return [l.split()[0] for l in self.ok("list").splitlines()[:-1]]
+
+ def test_prio(self):
+ self.assertEqual(self.ok("prio", "t-three", "0"), "- **t-three** [P0] (<1h): Third thing. Later.\n")
+
+ def test_prio_refused_when_it_jumps_a_dependency(self):
+ # t-three is After: t-one, the insert rule keeps it behind t-one
+ self.ok("prio", "t-three", "0")
+ self.assertEqual(self.order(), ["t-one", "t-three", "t-two"])
+
+ def test_move_section(self):
+ self.assertEqual(self.ok("move", "t-two", "deferred"), "- **t-two** [P2] (5h) (blocked: [[a-key]]): Second thing.\n")
+ self.assertEqual(self.order(), ["t-one", "t-three"])
+
+ def test_move_relative(self):
+ self.ok("move", "t-three", "--before", "t-two")
+ self.assertEqual(self.order(), ["t-one", "t-three", "t-two"])
+ self.assertEqual(self.fails("move", "t-two", "--before", "t-one"),
+ "wf: moving 't-two' there breaks priority order (wf prio, or --force)\n")
+ self.ok("move", "t-two", "--before", "t-one", "--force")
+ self.assertEqual(self.order(), ["t-two", "t-one", "t-three"])
+
+ def test_move_needs_one_target(self):
+ self.fails("move", "t-two", code=2)
+ self.fails("move", "t-two", "deferred", "--before", "t-one", code=2)
+
+ def test_status(self):
+ self.assertEqual(self.ok("status", "t-three", "progress", "feature/x"),
+ "- **t-three** [P2] (<1h) (in progress: feature/x): Third thing. Later.\n")
+ self.assertEqual(self.ok("status", "t-three", "blocked", "a-key"),
+ "- **t-three** [P2] (<1h) (blocked: [[a-key]]): Third thing. Later.\n")
+ self.assertEqual(self.ok("status", "t-three", "clear"), "- **t-three** [P2] (<1h): Third thing. Later.\n")
+ self.assertEqual(self.fails("status", "t-three", "blocked", "a-none"), "wf: 'a-none' is not an open Awaiting item\n")
+ self.fails("status", "t-three", "progress", code=2)
+
+ def test_set(self):
+ self.ok("set", "t-two", "--title", "Second: renamed", "--effort", "10h", "--sessions", "owner",
+ "--after", "t-one", "--ref", "docs/plan.md (goal)")
+ self.assertEqual(self.item("t-two"),
+ "- **t-two** [P2] (10h) (blocked: [[a-key]]): Second: renamed.\n"
+ " Sessions: owner\n - After: [[t-one]]\n Ref: docs/plan.md (goal)\n")
+ self.ok("set", "t-two", "--after", "", "--ref", "", "--sessions", "")
+ self.assertEqual(self.item("t-two"), "- **t-two** [P2] (10h) (blocked: [[a-key]]): Second: renamed.\n")
+
+ def test_set_done(self):
+ self.ok("set", "t-two", "--done", "two works")
+ self.assertEqual(self.item("t-two"), "- **t-two** [P2] (5h) (blocked: [[a-key]]): Second thing.\n Done: two works\n")
+ self.ok("set", "t-two", "--done", "")
+ self.assertEqual(self.item("t-two"), "- **t-two** [P2] (5h) (blocked: [[a-key]]): Second thing.\n")
+
+ def test_rename(self):
+ self.assertEqual(self.ok("rename", "a-key", "a-which-key"), "renamed: a-key → a-which-key (1 link)\n")
+ self.assertEqual(self.item("t-two"), "- **t-two** [P2] (5h) (blocked: [[a-which-key]]): Second thing.\n")
+ self.assertEqual(self.fails("rename", "t-one", "t-done"), "wf: id 't-done' already used in the archive (ids are never reused)\n")
+
+ def test_set_refused_on_bad_ref(self):
+ self.assertIn("Ref 'missing.md' does not exist", self.fails("set", "t-two", "--ref", "missing.md"))
+ self.assertEqual(self.tasks(), TASKS)
+
+ def test_set_without_fields(self):
+ self.fails("set", "t-two", code=2)
+
+ def test_note(self):
+ self.ok("note", "t-one", "cause found")
+ self.assertEqual(self.item("t-one"),
+ "- **t-one** [P1] (1h) (in progress: master): First thing. Do it.\n"
+ " - Steps: a\n - cause found\n Ref: DESIGN.md#terrain, docs/plan.md\n")
+
+ def test_body(self):
+ self.ok("body", "t-one", stdin="- Steps: b\n- Done: c\n")
+ self.assertEqual(self.item("t-one"),
+ "- **t-one** [P1] (1h) (in progress: master): First thing. Do it.\n"
+ " - Steps: b\n - Done: c\n Ref: DESIGN.md#terrain, docs/plan.md\n")
+
+ def test_body_refusal_says_refused_and_writes_nothing(self):
+ before = self.tasks()
+ for _ in range(2):
+ err = self.fails("body", "t-one", stdin="- After: [[t-three]]\n")
+ self.assertTrue(err.startswith("wf: refused: nothing written"), err)
+ self.assertIn("placed before 't-three'", err)
+ self.assertEqual(self.tasks(), before)
+
+ def test_body_positional_text_hints_stdin(self):
+ self.assertIn("stdin", self.fails("body", "t-one", "some", "text"))
+ self.assertIn("stdin", self.ok("body", "-h"))
+
+ def test_tick(self):
+ self.assertEqual(self.ok("tick", "t-play", "keyboard"), "ticked: keyboard works\n")
+ self.assertIn(" - [x] keyboard works\n", self.tasks())
+ self.assertEqual(self.fails("tick", "t-play", "1"), "wf: box already ticked: keyboard works\n")
+
+ def test_write_needs_exact_id(self):
+ self.assertRegex(self.fails("prio", "t-thr", "1"), r"^wf: unknown id 't-thr' \(nearest: t-three")
+
+ def test_untouched_text_survives_a_write(self):
+ self.ok("note", "t-far", "x")
+ self.assertEqual(self.tasks(), TASKS.replace("Far future.\n", "Far future.\n - x\n"))
+
+
+class OldProblemsTest(Cli):
+ tasks_text = TASKS + "\n## Notes\n\nSee [[t-lost]].\n"
+
+ def test_existing_problems_do_not_block_other_edits(self):
+ self.ok("note", "t-far", "x")
+ self.assertIn(" - x\n", self.tasks())
+
+
+class CrlfTest(Cli):
+ tasks_text = TASKS.replace("\n", "\r\n")
+
+ def test_write_keeps_crlf(self):
+ self.ok("add", "Cave seams", "-p", "1", "-e", "1h")
+ raw = (self.root / "TASKS.md").read_bytes()
+ self.assertIn(b"- **t-cave-seams** [P1] (1h): Cave seams.\r\n", raw)
+ self.assertEqual(raw.count(b"\n"), raw.count(b"\r\n"))
+ self.assertTrue(raw.endswith(b"Far future.\r\n"))
+
+ def test_read_commands_work(self):
+ self.assertEqual(self.ok("show", "a-key"), "- **a-key**: Key needed. Which one?\n")
+
+
+class FormatGateTest(Cli):
+ toml = TOML.replace("format = 1", "format = 0")
+
+ def test_write_refused(self):
+ self.assertEqual(self.fails("note", "t-one", "x"),
+ "wf: TASKS format 0, wf needs 1: run wf migrate --write (idle project, one commit)\n")
+ self.assertEqual(self.tasks(), TASKS)
+
+ def test_read_works(self):
+ self.assertIn("t-one", self.ok("list"))
+
+
+class NewerFormatTest(Cli):
+ toml = TOML.replace("format = 1", "format = 2")
+
+ def test_refused(self):
+ self.assertEqual(self.fails("list"), "wf: project format 2 is newer than this wf (1): update /projects/public/workflow\n")
+
+
+class ReportTest(Cli):
+ def test_report_line(self):
+ self.assertEqual(self.ok("report", "prio change\nneeds two commands", "--kind", "friction", "--cmd", "wf prio x 1"),
+ "reported (workflow inbox); carry on\n")
+ self.assertRegex(self.inbox.read_text(),
+ rf"^- {TODAY} demo friction: prio change needs two commands \(cmd: wf prio x 1\) @[0-9a-f]{{7}}\n$")
+
+ def test_default_kind_and_outside_project(self):
+ self.ok("report", "something", project=False, cwd=self.root.parent)
+ self.assertRegex(self.inbox.read_text(), rf"^- {TODAY} {self.root.parent.name} bug: something @")
+
+ def test_parallel_reports_stay_whole(self):
+ env = {**os.environ, "WF_INBOX": str(self.inbox)}
+ procs = [subprocess.Popen([sys.executable, str(WF), "--project", str(self.root), "report", f"report {n} " + "x" * 3000],
+ env=env, stdout=subprocess.DEVNULL) for n in range(8)]
+ for p in procs:
+ self.assertEqual(p.wait(timeout=30), 0)
+ lines = self.inbox.read_text().splitlines()
+ self.assertEqual(len(lines), 8)
+ for line in lines:
+ self.assertRegex(line, rf"^- {TODAY} demo bug: report \d x{{3000}} @[0-9a-f]{{7}}$")
+
+ def test_empty_report(self):
+ self.fails("report", " ", code=2)
+
+
+class InitTest(Cli):
+ def test_init(self):
+ new = self.root.parent / "fresh"
+ new.mkdir()
+ self.assertEqual(self.ok("init", project=False, cwd=new),
+ "wrote workflow.toml\nwrote TASKS.md\nwrote tasks/archive.md\nwrote CLAUDE.md\n")
+ self.assertTrue((new / "TASKS.md").read_text().startswith("# Tasks — fresh\n"))
+ self.assertTrue((new / "CLAUDE.md").read_text().startswith("# CLAUDE.md — fresh\n"))
+ self.assertEqual(self.ok("check", project=False, cwd=new), "OK: 0 errors · 0 warnings\n")
+ self.assertEqual(self.ok("add", "First", "-p", "1", "-e", "1h", project=False, cwd=new), "- **t-first** [P1] (1h): First.\n")
+
+ def test_init_twice(self):
+ self.assertEqual(self.fails("init"), f"wf: {self.root}/workflow.toml exists already\n")
+
+ def test_init_keeps_existing_tasks_file(self):
+ new = self.root.parent / "old"
+ new.mkdir()
+ (new / "TASKS.md").write_text("mine\n")
+ self.assertEqual(self.ok("init", project=False, cwd=new),
+ "wrote workflow.toml\nkept TASKS.md (exists; old format → wf migrate)\nwrote tasks/archive.md\nwrote CLAUDE.md\n")
+ self.assertEqual((new / "TASKS.md").read_text(), "mine\n")
+
+ def test_init_keeps_existing_claude_md(self):
+ new = self.root.parent / "oldc"
+ new.mkdir()
+ (new / "CLAUDE.md").write_text("mine\n")
+ out = self.ok("init", project=False, cwd=new)
+ self.assertIn("kept CLAUDE.md (exists)\n", out)
+ self.assertEqual((new / "CLAUDE.md").read_text(), "mine\n")
+
+ def test_init_inside_a_project_subfolder_makes_a_new_project_there(self):
+ self.assertEqual(self.ok("init", project=False, cwd=self.root / "docs").splitlines()[0], "wrote workflow.toml")
+ self.assertTrue((self.root / "docs" / "workflow.toml").is_file())
+
+
+class ProjectsRootTest(unittest.TestCase):
+ def setUp(self):
+ import wf
+ self.wf = wf
+ self.tmp = tempfile.TemporaryDirectory()
+ self.top = Path(self.tmp.name)
+ self.tool = self.top / "public" / "workflow"
+ self.tool.mkdir(parents=True)
+
+ def tearDown(self):
+ self.tmp.cleanup()
+
+ def project(self, rel):
+ (self.top / rel).mkdir(parents=True)
+ (self.top / rel / "workflow.toml").write_text("")
+
+ def test_parent_with_projects_wins(self):
+ self.project("public/a")
+ self.project("games/b")
+ self.assertEqual(self.wf.default_root(self.tool), self.top / "public")
+
+ def test_parent_without_projects_falls_back_to_grandparent(self):
+ self.project("games/b")
+ self.assertEqual(self.wf.default_root(self.tool), self.top)
+
+ def test_no_projects_anywhere_keeps_parent(self):
+ self.assertEqual(self.wf.default_root(self.tool), self.top / "public")
+
+
+class WriteSafetyTest(unittest.TestCase):
+ def setUp(self):
+ import wf
+ self.wf = wf
+ self.tmp = tempfile.TemporaryDirectory()
+ self.path = Path(self.tmp.name) / "TASKS.md"
+ self.path.write_text("one\n")
+
+ def tearDown(self):
+ self.tmp.cleanup()
+
+ def test_write(self):
+ stamp = self.wf.stamp(self.path)
+ self.wf.write_if_unchanged(self.path, "two\n", stamp)
+ self.assertEqual(self.path.read_text(), "two\n")
+ self.assertEqual(sorted(p.name for p in self.path.parent.iterdir()), ["TASKS.md"])
+
+ def test_stops_when_the_file_changed_since_read(self):
+ stamp = self.wf.stamp(self.path)
+ self.path.write_text("someone else wrote this\n")
+ with self.assertRaisesRegex(self.wf.Failure, "TASKS.md changed on disk since it was read: nothing written, run again"):
+ self.wf.write_if_unchanged(self.path, "two\n", stamp)
+ self.assertEqual(self.path.read_text(), "someone else wrote this\n")
+
+ def test_keeps_file_mode(self):
+ self.path.chmod(0o640)
+ self.wf.write_if_unchanged(self.path, "two\n", self.wf.stamp(self.path))
+ self.assertEqual(self.path.stat().st_mode & 0o777, 0o640)
+
+
+if __name__ == "__main__":
+ unittest.main()
+
+
+OLD_TASKS = """\
+# Tasks — demo
+
+## Pending
+
+1. **[P1] First thing** (Effort: 1h) — do it.
+ - Steps: a
+ Reference: [DESIGN.md#terrain](DESIGN.md#terrain)
+
+2. **[P2] Second thing** (Effort: 5h) — later.
+ - After: First thing
+
+## Needs human
+
+## Awaiting your decision
+
+- Which key to use.
+"""
+MIGRATED = """\
+# Tasks — demo
+
+## Pending
+
+- **t-first-thing** [P1] (1h): First thing. do it.
+ - Steps: a
+ Ref: DESIGN.md#terrain
+
+- **t-second-thing** [P2] (5h): Second thing. later.
+ - After: [[t-first-thing]]
+
+## Needs human
+
+## Awaiting your decision
+
+- **a-which-key-to-use**: Which key to use.
+
+## Deferred
+"""
+
+
+class MigrateCliTest(Cli):
+ tasks_text = OLD_TASKS
+ toml = TOML.replace("format = 1", "format = 0")
+
+ def test_dry_run_prints_diff_and_map_and_writes_nothing(self):
+ out = self.ok("migrate")
+ self.assertIn("+- **t-first-thing** [P1] (1h): First thing. do it.\n", out)
+ self.assertIn("\nids:\n t-first-thing First thing\n t-second-thing Second thing\n a-which-key-to-use Which key to use\n", out)
+ self.assertTrue(out.endswith("check after migrate: 0 errors\ndry run: nothing written (wf migrate --write)\n"), out)
+ self.assertEqual(self.tasks(), OLD_TASKS)
+ self.assertIn("format = 0", (self.root / "workflow.toml").read_text())
+
+ def test_write(self):
+ out = self.ok("migrate", "--write")
+ self.assertTrue(out.endswith("check after migrate: 0 errors\nwrote TASKS.md, workflow.toml (format = 1)\n"), out)
+ self.assertEqual(self.tasks(), MIGRATED)
+ self.assertEqual((self.root / "workflow.toml").read_text(), TOML)
+ self.assertEqual(self.ok("check"), "OK: 0 errors · 0 warnings\n")
+ self.assertEqual(self.ok("next", "--as", "opus", "--brief").splitlines()[0], "- **t-first-thing** [P1] (1h): First thing. do it.")
+
+ def test_second_run_changes_nothing(self):
+ self.ok("migrate", "--write")
+ self.assertEqual(self.ok("migrate", "--write"), "nothing to migrate\n")
+ self.assertEqual(self.tasks(), MIGRATED)
+
+ def test_ids_avoid_archived_ones(self):
+ (self.root / "tasks" / "archive.md").write_text("# A\n\n- 2026-09-01 **t-first-thing** First thing — old\n")
+ self.ok("migrate", "--write")
+ self.assertIn("- **t-first-thing-2** [P1] (1h): First thing. do it.\n", self.tasks())
diff --git a/tests/test_cloud.py b/tests/test_cloud.py
new file mode 100644
index 0000000..e922f9c
--- /dev/null
+++ b/tests/test_cloud.py
@@ -0,0 +1,1174 @@
+import datetime as dt
+import io
+import json
+import os
+import shutil
+import sys
+import tempfile
+import unittest
+import subprocess
+import textwrap
+from contextlib import redirect_stderr, redirect_stdout
+from pathlib import Path
+
+HERE = Path(__file__).resolve().parent.parent
+sys.path.insert(0, str(HERE))
+import wf_cloud # noqa: E402
+from wflib import cloud as C, usage as U # noqa: E402
+
+T0 = dt.datetime(2026, 10, 6, 12, 0, tzinfo=dt.timezone.utc)
+
+
+def led_with(n_running, **kw):
+ led = {**C.new(), **kw}
+ for i in range(n_running):
+ C.add(led, f"t{i}", "p", f"s{i}", "claude-sonnet-5-5", T0)
+ return led
+
+
+class Pure(unittest.TestCase):
+ def test_default_budget(self):
+ self.assertEqual(C.new()["budget"], 240.0)
+
+ def test_balance(self):
+ # 240 - 10 - 4*2 = 222
+ self.assertEqual(C.balance(led_with(2, spent=10.0)), 222.0)
+ self.assertEqual(C.balance(C.new()), 240.0)
+
+ def test_refuse_low_balance(self):
+ # 20 - 17 - 0 = 3 < 4
+ self.assertIn("balance", C.refusal(led_with(0, budget=20.0, spent=17.0)))
+ # exactly 4 left: ok
+ self.assertIsNone(C.refusal(led_with(0, budget=20.0, spent=16.0)))
+
+ def test_refuse_max_parallel(self):
+ self.assertIn("max_parallel", C.refusal(led_with(3)))
+ self.assertIsNone(C.refusal(led_with(2)))
+
+ def test_balance_counts_reserve_toward_refusal(self):
+ # 10 - 0 - 4*2 = 2 < 4
+ self.assertIn("balance", C.refusal(led_with(2, budget=10.0)))
+
+ def test_set_balance(self):
+ led = led_with(0, spent=50.0)
+ row = C.set_balance(led, 200.0, T0)
+ self.assertEqual(led["spent"], 40.0)
+ self.assertEqual((row["usd_source"], row["usd"]), ("owner", -10.0))
+
+ def test_charge_from_usage(self):
+ led = led_with(1)
+ u = U.Usage(turns=1, inp=1_000_000, out=1_000_000) # sonnet-5-5: 2 + 10 = 12 ; x1.15 = 13.8
+ e = C.end(led, "s0", "done", u)
+ self.assertAlmostEqual(e["usd"], 13.8)
+ self.assertEqual(e["usd_source"], "self")
+ self.assertAlmostEqual(led["spent"], 13.8)
+ self.assertEqual(C.running(led), 0)
+
+ def test_charge_missing_usage_is_reserve(self):
+ led = led_with(1)
+ e = C.end(led, "s0", "handback", None)
+ self.assertEqual((e["usd"], e["usd_source"]), (4.0, "est"))
+
+ def test_end_twice_refused(self):
+ led = led_with(1)
+ C.end(led, "s0", "done", None)
+ with self.assertRaises(C.CloudError):
+ C.end(led, "s0", "done", None)
+
+ def test_expire_lost_after_24h(self):
+ led = led_with(2)
+ led["entries"][1]["sent"] = (T0 + dt.timedelta(hours=20)).isoformat()
+ lost = C.expire(led, T0 + dt.timedelta(hours=25))
+ self.assertEqual([e["sid"] for e in lost], ["s0"])
+ self.assertEqual(led["entries"][0]["state"], "lost")
+ self.assertEqual(led["spent"], 4.0)
+ self.assertEqual(C.running(led), 1)
+
+ def test_roundtrip_and_bad(self):
+ led = led_with(1)
+ self.assertEqual(C.loads(C.dumps(led)), led)
+ with self.assertRaises(C.CloudError):
+ C.loads("{nope")
+
+
+class ArchivePure(unittest.TestCase):
+ CREDS = '{"claudeAiOauth": {"accessToken": "tok-1", "refreshToken": "r"}, "other": 1}'
+
+ def test_request_url_and_headers(self):
+ url, h = C.archive_request("session_01AbCdEf", self.CREDS, "2.1.291")
+ self.assertEqual(url, "https://api.anthropic.com/v1/code/sessions/session_01AbCdEf/archive")
+ self.assertEqual(h, {"Authorization": "Bearer tok-1", "Content-Type": "application/json",
+ "anthropic-version": "2023-06-01", "User-Agent": "claude-code/2.1.291"})
+
+ def test_trusted_device_token_sent(self):
+ creds = '{"claudeAiOauth": {"accessToken": "tok-1"}, "trustedDeviceToken": "dev-9"}'
+ _, h = C.archive_request("session_01AbCdEf", creds, "2.1.291")
+ self.assertEqual(h["X-Trusted-Device-Token"], "dev-9")
+
+ def test_no_token_or_bad_sid(self):
+ for creds in ("", "{}", '{"claudeAiOauth": {}}', "not json"):
+ with self.assertRaisesRegex(C.CloudError, "no claude.ai login token"):
+ C.archive_request("session_01AbCdEf", creds, "1")
+ for sid in ("pending:x", "-", "session_01/../x"):
+ with self.assertRaisesRegex(C.CloudError, "not a cloud session id"):
+ C.archive_request(sid, self.CREDS, "1")
+
+ def test_problem(self):
+ self.assertIsNone(C.archive_problem(200, ""))
+ self.assertIsNone(C.archive_problem(409, "already"))
+ self.assertEqual(C.archive_problem(401, "x"), "HTTP 401 (login expired? run claude once)")
+ self.assertEqual(C.archive_problem(404, "no\nsuch " + "y" * 200), "HTTP 404: no such " + "y" * 102)
+
+ def test_unarchived(self):
+ led = led_with(3)
+ C.end(led, "s0", "done")
+ C.end(led, "s1", "lost")
+ led["entries"][1]["archived"] = True
+ for e in led["entries"]:
+ e["sid"] = "session_" + e["sid"]
+ C.set_balance(led, 200, T0)
+ self.assertEqual(C.unarchived(led), ["session_s0"])
+
+
+class Cli(unittest.TestCase):
+ def run_wf(self, st, *argv):
+ out = io.StringIO()
+ with redirect_stdout(out):
+ code = wf_cloud.main(list(argv), state=st, now=T0)
+ return code, out.getvalue()
+
+ def test_ledger_flow(self):
+ with tempfile.TemporaryDirectory() as d:
+ st = Path(d)
+ code, out = self.run_wf(st, "ledger")
+ self.assertEqual(code, 0)
+ self.assertIn("budget $240.00 spent $0.00 balance $240.00 running 0/3", out)
+ _, out = self.run_wf(st, "ledger", "--budget", "100", "--set-balance", "70")
+ self.assertIn("budget $100.00 spent $30.00 balance $70.00", out)
+ _, out = self.run_wf(st, "ledger") # persisted
+ self.assertIn("spent $30.00", out)
+
+
+# Synthetic, same shape as a real `claude --cloud` run in a pty (escape codes, CRLF, title line).
+SEND_OUT = ("\x1b[?2004l\x1b]3008;start=x;type=command\x1b\\\x1b7\x1b[r\x1b8\x1b[?25h\x1b[>4;2m\x1b[c"
+ "Created cloud session: Some title\r\n"
+ "View: https://claude.ai/code/session_01AbCdEfGhIjKlMnOpQrStUv?from=cli&m=0\r\n"
+ "Resume with: claude --teleport session_01AbCdEfGhIjKlMnOpQrStUv\r\n")
+
+FAKE = r"""#!/usr/bin/env python3
+import os, sys, time, tty
+mode, st = os.environ.get("FAKE_MODE", ""), os.environ["FAKE_STATE"]
+def say(s): sys.stdout.write(s); sys.stdout.flush()
+def line():
+ return sys.stdin.readline().strip()
+n = int(open(st).read()) if os.path.exists(st) else 0
+open(st, "w").write(str(n + 1))
+if "trust" in mode and n == 0:
+ say("\x1b[2J Do\x1b[4Gyou trust the files in this folder?\r\n 1. No, exit\r\n 2. Yes, proceed\r\n")
+ if line() != "\x1b[B":
+ say("refused\r\n"); sys.exit(1)
+if sys.argv[1] == "--cloud":
+ if os.environ.get("FAKE_ARGS"):
+ open(os.environ["FAKE_ARGS"], "w").write(os.getcwd() + "\n" + sys.argv[2])
+ if "nosid" in mode:
+ say("Error: something\r\nline2\r\n"); sys.exit(1)
+ say(%r); sys.exit(0)
+assert sys.argv[1] == "--teleport", sys.argv
+say("\x1b[3G◯\x1b[5GChecking\x1b[14Gout\x1b[18Gbranch\r\n")
+if "noresume" in mode or ("flaky" in mode and n == 0):
+ sys.exit(0)
+say("●\x1b[3GSession\x1b[11Gresumed\r\n❯ ")
+while True:
+ cmd = line()
+ if cmd.startswith("/export "):
+ if line() == "": # 2nd Enter (autocomplete) needed
+ open(cmd.split(" ", 1)[1], "w").write("● WF-RESULT done\n")
+ say("Conversation exported\r\n")
+ elif cmd == "/exit":
+ sys.exit(0)
+""" % SEND_OUT
+
+
+class Parse(unittest.TestCase):
+ def test_sid_from_send_output(self):
+ self.assertEqual(C.parse_sid(SEND_OUT), "session_01AbCdEfGhIjKlMnOpQrStUv")
+
+ def test_sid_from_view_only(self):
+ self.assertEqual(C.parse_sid("View: https://claude.ai/code/session_01Zz9Zz9Zz9Zz9?from=cli\n"),
+ "session_01Zz9Zz9Zz9Zz9")
+
+ def test_sid_absent(self):
+ self.assertIsNone(C.parse_sid("Error: not logged in\nclaude --teleport\n"))
+
+ def test_strip_ansi_gaps(self):
+ # CSI n G (column) and CSI n C (forward) are word gaps in the TUI
+ self.assertEqual(C.strip_ansi("\x1b[39m●\x1b[3GSession\x1b[11Gresumed\r\n"), "● Session resumed\n")
+ self.assertEqual(C.strip_ansi("a\x1b[2Cb"), "a b")
+
+ def test_last_lines(self):
+ self.assertEqual(C.last_lines("a\r\n\r\nb\nc\n", 2), ["b", "c"])
+
+ def test_trust_keeps_other_keys(self):
+ out = json.loads(C.trust('{"x": 1, "projects": {"/a": {"k": 2}}}', "/b"))
+ self.assertEqual(out, {"x": 1, "projects": {"/a": {"k": 2}, "/b": {"hasTrustDialogAccepted": True}}})
+ out = json.loads(C.trust('{"projects": {"/a": {"k": 2}}}', "/a"))
+ self.assertEqual(out["projects"]["/a"], {"k": 2, "hasTrustDialogAccepted": True})
+ self.assertEqual(json.loads(C.trust("", "/c")), {"projects": {"/c": {"hasTrustDialogAccepted": True}}})
+ with self.assertRaises(C.CloudError):
+ C.trust("[1]", "/c")
+
+
+class PtyDriver(unittest.TestCase):
+ """send/export against a fake claude on a real pty."""
+
+ def setUp(self):
+ self.d = Path(tempfile.mkdtemp())
+ fake = self.d / "claude"
+ fake.write_text(FAKE)
+ fake.chmod(0o755)
+ self.repo = self.d / "repo"
+ self.repo.mkdir()
+ self.cj = self.d / "claude.json"
+ self.cj.write_text('{"keep": true}')
+ self.env = {"WF_CLAUDE_JSON": str(self.cj), "FAKE_STATE": str(self.d / "n")}
+ self.old = (wf_cloud.CLAUDE, wf_cloud.SETTLE, dict(os.environ))
+ wf_cloud.CLAUDE, wf_cloud.SETTLE = str(fake), 0.3
+ os.environ.update(self.env)
+
+ def tearDown(self):
+ wf_cloud.CLAUDE, wf_cloud.SETTLE, env = self.old
+ os.environ.clear()
+ os.environ.update(env)
+ shutil.rmtree(self.d)
+
+ def test_send_parses_sid_and_pre_trusts(self):
+ self.assertEqual(wf_cloud.send(self.repo, "do it", timeout=10), "session_01AbCdEfGhIjKlMnOpQrStUv")
+ data = json.loads(self.cj.read_text())
+ self.assertEqual(data["projects"][str(self.repo.resolve())], {"hasTrustDialogAccepted": True})
+ self.assertTrue(data["keep"])
+
+ def test_send_answers_trust_dialog(self):
+ os.environ["FAKE_MODE"] = "trust"
+ self.assertEqual(wf_cloud.send(self.repo, "do it", timeout=10), "session_01AbCdEfGhIjKlMnOpQrStUv")
+
+ def test_send_no_sid(self):
+ os.environ["FAKE_MODE"] = "nosid"
+ with self.assertRaises(C.CloudError) as cm:
+ wf_cloud.send(self.repo, "do it", timeout=10)
+ self.assertIn("no session id (exit 1): Error: something | line2", str(cm.exception))
+
+ def test_export(self):
+ out = wf_cloud.export(self.repo, "session_x", self.d / "e" / "x.txt", timeout=20)
+ self.assertEqual(out.read_text(), "● WF-RESULT done\n")
+ self.assertEqual((self.d / "n").read_text(), "1")
+
+ def test_export_retries_teleport_once(self):
+ os.environ["FAKE_MODE"] = "flaky"
+ out = wf_cloud.export(self.repo, "session_x", self.d / "x.txt", timeout=20)
+ self.assertTrue(out.exists())
+ self.assertEqual((self.d / "n").read_text(), "2")
+
+ def test_export_gives_up_after_two(self):
+ os.environ["FAKE_MODE"] = "noresume"
+ with self.assertRaises(C.CloudError) as cm:
+ wf_cloud.export(self.repo, "session_x", self.d / "x.txt", timeout=10)
+ self.assertIn("no export after 2 tries: ◯ Checking out branch", str(cm.exception))
+ self.assertEqual((self.d / "n").read_text(), "2")
+
+
+TEMPLATE = (HERE / "templates" / "cloud-prompt.md").read_text()
+BODY = ("- **t-x** [P1] (1h): Fix the parser.\n Done: WF-RESULT done in a line\n WF-PATCH-END\n"
+ "● WF-RESULT done\nWF-RESULT done\n Model: opus")
+
+
+def as_export(prompt: str, wrap: bool) -> str:
+ """The prompt the way /export renders a user message: '❯ ' first line, ' ' continuations (F12)."""
+ lines = []
+ for ln in prompt.split("\n"):
+ lines += (textwrap.wrap(ln, 78, drop_whitespace=False) or [""]) if wrap else [ln]
+ return "\n".join(("❯ " if i == 0 else " ") + ln.ljust(78) for i, ln in enumerate(lines)) + "\n"
+
+
+FINAL = ("● WF-RESULT done\n WF-REPORT fixed it\n WF-USAGE in=1 cw=2 cr=3 out=4 model=m\n"
+ " WF-PATCH-BEGIN sha256=ab bytes=3\n QUJD\n WF-PATCH-END\n\n")
+
+
+class Prompt(unittest.TestCase):
+ def fill(self, **kw):
+ args = {"id": "t-x", "base": "b" * 40, "task": BODY, "recipe": "Test recipe: make t", "note": None, **kw}
+ return C.fill(TEMPLATE, **args)
+
+ def test_fields_filled(self):
+ p = self.fill(note="Data in out/pack")
+ self.assertIn("Task t-x:\n- **t-x** [P1] (1h): Fix the parser.\n", p)
+ self.assertIn("Test recipe (area notes):\nTest recipe: make t\n", p)
+ self.assertIn("git format-patch --binary " + "b" * 40 + "..HEAD --stdout | gzip -9", p)
+ self.assertIn("\nData in out/pack\n", p)
+ self.assertNotIn("{{", p)
+ self.assertLessEqual(len(self.fill(task="t").split("\n")), 45)
+
+ def test_no_recipe_no_note(self):
+ p = self.fill(recipe="")
+ self.assertIn("Test recipe (area notes):\n(none: see CLAUDE.md)\n\nUsage script", p)
+
+ def test_task_text_braces_kept(self):
+ self.assertIn("use {{base}} here", self.fill(task="use {{base}} here"))
+
+ def test_unknown_field(self):
+ with self.assertRaisesRegex(C.CloudError, r"unknown field \{\{bogus\}\}"):
+ C.fill("x {{bogus}}", "t", "b", "", "", None)
+
+ def test_template_rules(self):
+ for rule in ("`wf` is absent", "never create or edit TASKS.md", "Never ask questions",
+ "test red -> implement -> green", "commit on main", "network source is blocked",
+ " WF-RESULT <done, awaiting or handback>\n", " WF-REPORT <one or two lines",
+ 'print("WF-USAGE in=%d cw=%d cr=%d out=%d model=%s"', 'echo "WF-PATCH-BEGIN sha256=$(',
+ "echo WF-PATCH-END", "base64 -w 76", "~/.claude/projects", "never -A", "__pycache__"):
+ self.assertIn(rule, TEMPLATE)
+ # live run 2026-10-06: '[WF-RESULT R]' placeholders made the session drop every key word
+ self.assertNotIn("[WF-", TEMPLATE)
+ self.assertIn("key word", TEMPLATE)
+
+ def test_parser_never_matches_prompt(self):
+ p = self.fill()
+ for wrap in (False, True):
+ self.assertIsNone(C.final_message(as_export(p, wrap)), wrap)
+
+ def test_parser_takes_final_message_after_prompt(self):
+ exp = as_export(self.fill(), True) + "\n● Working on it.\n Ran 3 shell commands\n\n" + FINAL
+ self.assertEqual(C.final_message(exp), ["WF-RESULT done", "WF-REPORT fixed it",
+ "WF-USAGE in=1 cw=2 cr=3 out=4 model=m",
+ "WF-PATCH-BEGIN sha256=ab bytes=3", "QUJD", "WF-PATCH-END"])
+
+ def test_parser_last_result_wins_and_stops_at_next_message(self):
+ exp = "● WF-RESULT handback\n WF-REPORT old\n\n● WF-RESULT awaiting\n WF-REPORT q?\n\n❯ more\n"
+ self.assertEqual(C.final_message(exp), ["WF-RESULT awaiting", "WF-REPORT q?"])
+
+ def test_include_problem(self):
+ for bad in ("/abs", "../x", "a/../../b", "", ".git", ".git/config"):
+ self.assertIsNotNone(C.include_problem(bad), bad)
+ for ok in ("out/pack", "data.bin", "out/a/b/"):
+ self.assertIsNone(C.include_problem(ok), ok)
+
+ def test_size_refusal(self):
+ self.assertIsNone(C.size_refusal(90_000_000))
+ self.assertEqual(C.size_refusal(91_400_000), "snapshot 91 MB > 90 MB; trim cloud_include")
+
+
+from test_claims import FOUR, git # noqa: E402
+from test_setup import GitCli # noqa: E402
+
+AREAS = "# Demo\n\n## Areas\n\n### app\n- Test recipe: RECIPE-MARK python3 -m unittest\n- Paths: src/app.py\n"
+
+
+
+def stub_archive(tc, d: Path, status=200):
+ """No network: wf_cloud.post records (url, headers) and answers `status`; fake credentials file."""
+ creds = d / "credentials.json"
+ creds.write_text('{"claudeAiOauth": {"accessToken": "tok-1"}}')
+ tc.posts, tc.answer = [], status
+ saved = (wf_cloud.post, wf_cloud.cli_version, os.environ.get("WF_CLAUDE_CREDENTIALS"))
+
+ def post(url, headers, timeout=10):
+ tc.posts.append((url, headers))
+ if isinstance(tc.answer, Exception):
+ raise tc.answer
+ return tc.answer, "body"
+
+ def restore():
+ wf_cloud.post, wf_cloud.cli_version, env = saved
+ os.environ.pop("WF_CLAUDE_CREDENTIALS", None)
+ if env is not None:
+ os.environ["WF_CLAUDE_CREDENTIALS"] = env
+ wf_cloud.post, wf_cloud.cli_version = post, lambda: "9.9.9"
+ os.environ["WF_CLAUDE_CREDENTIALS"] = str(creds)
+ tc.addCleanup(restore)
+
+
+class CloudProject(GitCli):
+ """A cloud-opted git project + fake claude; wf cloud in-process."""
+ tasks_text = FOUR.replace("Four.", "Four: fix src/app.py.")
+ toml = GitCli.toml + 'cloud = true\ncloud_include = ["out/pack"]\ncloud_note = "NOTE-MARK data in out/pack"\n'
+
+ def setUp(self):
+ super().setUp()
+ (self.root / "src").mkdir()
+ (self.root / "src" / "app.py").write_text("v1\n")
+ (self.root / "CLAUDE.md").write_text(AREAS)
+ (self.root / "out").mkdir()
+ (self.root / "out" / "tracked.txt").write_text("tracked under out/\n")
+ git(self.root, "add", "-A")
+ git(self.root, "add", "-f", "out/tracked.txt")
+ git(self.root, "commit", "-qm", "app")
+ (self.root / "src" / "app.py").write_text("dirty\n") # working tree: not in the snapshot
+ (self.root / "out" / "pack").mkdir(parents=True)
+ (self.root / "out" / "pack" / "data.bin").write_text("pack\n")
+ (self.root / "out" / "other.txt").write_text("not included\n")
+ (self.root / ".wf").mkdir(exist_ok=True)
+ (self.root / ".wf" / "x").write_text("state\n")
+ self.d = Path(tempfile.mkdtemp())
+ fake = self.d / "claude"
+ fake.write_text(FAKE)
+ fake.chmod(0o755)
+ self.st = self.d / "state"
+ self.old = (wf_cloud.CLAUDE, wf_cloud.CAP, dict(os.environ))
+ wf_cloud.CLAUDE = str(fake)
+ os.environ.update({"WF_CLAUDE_JSON": str(self.d / "claude.json"), "FAKE_STATE": str(self.d / "n"),
+ "FAKE_ARGS": str(self.d / "args"), "CLAUDE_CODE_MESSAGING_SOCKET": "",
+ "WF_INBOX": str(self.d / "inbox.md")})
+ self.snap = self.root / "out" / "cloud" / "t-four"
+ self.rec = self.root / ".wf" / "cloud" / "t-four.json"
+ stub_archive(self, self.d)
+
+ def tearDown(self):
+ wf_cloud.CLAUDE, wf_cloud.CAP, env = self.old
+ os.environ.clear()
+ os.environ.update(env)
+ shutil.rmtree(self.d)
+ super().tearDown()
+
+ def send(self, *extra):
+ out, err = io.StringIO(), io.StringIO()
+ with redirect_stdout(out), redirect_stderr(err):
+ code = wf_cloud.main(["send", "t-four", "--project", str(self.root), *extra], state=self.st, now=T0)
+ return code, out.getvalue(), err.getvalue()
+
+ def ledger(self):
+ return json.loads((self.st / "cloud.json").read_text()) if (self.st / "cloud.json").exists() else None
+
+ def sh(self, cwd, *args):
+ return subprocess.run(["git", "-C", str(cwd), *args], capture_output=True, text=True).stdout.strip()
+
+class Send(CloudProject):
+ """wf cloud send against the fake claude."""
+
+ def test_send_snapshot_prompt_record_claim(self):
+ master = self.sh(self.root, "rev-parse", "master")
+ code, out, err = self.send()
+ self.assertEqual((code, err), (0, ""), out)
+ sid = "session_01AbCdEfGhIjKlMnOpQrStUv"
+ files = sorted(str(p.relative_to(self.snap)) for p in self.snap.rglob("*")
+ if p.is_file() and ".git" not in p.relative_to(self.snap).parts)
+ self.assertEqual(files, [".gitignore", "CLAUDE.md", "DESIGN.md", "docs/plan.md", "out/pack/data.bin",
+ "src/app.py", "workflow.toml"])
+ self.assertEqual((self.snap / "src" / "app.py").read_text(), "v1\n")
+ self.assertEqual(self.sh(self.snap, "log", "--format=%s"), f"base {master}")
+ self.assertEqual(self.sh(self.snap, "ls-files").split("\n"), files) # all committed (-f)
+ self.assertEqual(self.sh(self.snap, "status", "--porcelain"), "")
+ self.assertEqual(self.sh(self.snap, "branch", "--show-current"), "main")
+ base = self.sh(self.snap, "rev-parse", "HEAD")
+ cwd, prompt = (self.d / "args").read_text().split("\n", 1)
+ self.assertEqual(Path(cwd), self.snap.resolve())
+ self.assertIn("Task t-four:\n- **t-four** [P3] (1h): Four: fix src/app.py.\n", prompt)
+ self.assertIn("RECIPE-MARK", prompt)
+ self.assertIn("NOTE-MARK data in out/pack", prompt)
+ self.assertIn(f"format-patch --binary {base}..HEAD", prompt)
+ rec = json.loads(self.rec.read_text())
+ self.assertEqual({k: rec[k] for k in ("id", "sid", "master", "base", "lane", "sent", "model")},
+ {"id": "t-four", "sid": sid, "master": master, "base": base, "lane": "slow",
+ "sent": "2026-10-06T12:00:00+00:00", "model": "claude-opus-5-5"})
+ self.assertIn(f"**t-four** [P3] (1h) (in progress: cloud:{sid}): Four", self.tasks())
+ led = self.ledger()
+ self.assertEqual([(e["id"], e["sid"], e["state"]) for e in led["entries"]], [("t-four", sid, "running")])
+ self.assertIn(f"sent t-four: {sid}", out)
+ code, _, err = self.send() # twice: refused
+ self.assertEqual(code, 1)
+ self.assertIn(f"t-four already sent ({sid})", err)
+
+ def test_dry_run_sends_nothing(self):
+ code, out, err = self.send("--dry-run")
+ self.assertEqual((code, err), (0, ""))
+ self.assertRegex(out, r"^snapshot 0\.0 MB \(master [0-9a-f]{12}, base [0-9a-f]{12}\)\n\nYou are a remote")
+ self.assertIn("Task t-four:", out)
+ self.assertFalse(self.snap.exists())
+ self.assertFalse(self.snap.parent.exists()) # out/cloud gone too, out/ kept
+ self.assertFalse(self.rec.exists())
+ self.assertIsNone(self.ledger())
+ self.assertFalse((self.d / "n").exists()) # claude never ran
+ self.assertNotIn("in progress", self.tasks())
+
+ def test_dry_run_keeps_live_snapshot(self):
+ self.assertEqual(self.send()[0], 0)
+ head = self.sh(self.snap, "rev-parse", "HEAD")
+ code, out, err = self.send("--dry-run")
+ self.assertEqual((code, err), (0, ""))
+ self.assertEqual(self.sh(self.snap, "rev-parse", "HEAD"), head) # live session's folder intact
+ self.assertEqual(sorted(p.name for p in self.snap.parent.iterdir()), ["t-four"]) # dry-run folder dropped
+
+ def test_size_refused(self):
+ wf_cloud.CAP = 10
+ code, _, err = self.send()
+ self.assertEqual(code, 1)
+ self.assertIn("MB > 0 MB; trim cloud_include", err)
+ self.assertFalse(self.snap.exists())
+ self.assertIsNone(self.ledger())
+ self.assertFalse((self.d / "n").exists())
+
+ def test_ledger_refused_exit_3(self):
+ with wf_cloud.locked(self.st) as led:
+ for i in range(3):
+ C.add(led, f"t{i}", "p", f"s{i}", "claude-opus-5-5", T0)
+ code, out, err = self.send()
+ self.assertEqual(code, 3)
+ self.assertEqual(err.count("\n"), 1)
+ self.assertIn("cloud refused: ", err)
+ self.assertIn("max_parallel", err)
+ self.assertEqual(len(self.ledger()["entries"]), 3)
+ self.assertFalse(self.snap.exists())
+ self.assertFalse(self.rec.exists())
+ self.assertFalse((self.d / "n").exists())
+
+ def test_send_failure_releases_reserve(self):
+ os.environ["FAKE_MODE"] = "nosid"
+ code, _, err = self.send()
+ self.assertEqual(code, 1)
+ self.assertIn("no session id", err)
+ self.assertEqual(self.ledger()["entries"], [])
+ self.assertFalse(self.snap.exists())
+ self.assertFalse(self.rec.exists())
+
+ def test_not_opted_in(self):
+ (self.root / "workflow.toml").write_text(GitCli.toml)
+ code, _, err = self.send()
+ self.assertEqual(code, 1)
+ self.assertIn("not opted in", err)
+
+ def test_include_outside_refused(self):
+ (self.root / "workflow.toml").write_text(GitCli.toml + 'cloud = true\ncloud_include = ["../x"]\n')
+ code, _, err = self.send("--dry-run")
+ self.assertEqual(code, 1)
+ self.assertIn("cloud_include '../x': must be a path inside the project", err)
+
+
+if __name__ == "__main__":
+ unittest.main()
+
+
+class PullPure(unittest.TestCase):
+ def result(self, raw: bytes, sha=None, n=None):
+ import base64, gzip, hashlib
+ gz = gzip.compress(raw)
+ b64 = base64.b64encode(gz).decode()
+ return ["WF-RESULT done", "WF-REPORT fixed the parser,", "tests green",
+ "WF-USAGE in=1000 cw=2000 cr=3000 out=4000 model=claude-opus-5-5",
+ f"WF-PATCH-BEGIN sha256={sha or hashlib.sha256(gz).hexdigest()} bytes={n or len(gz)}",
+ *[b64[i:i + 76] for i in range(0, len(b64), 76)], "WF-PATCH-END"]
+
+ def test_parse_and_decode(self):
+ r = C.parse_result(self.result(b"diff --git a/x b/x\n"))
+ self.assertEqual((r.state, r.report, r.model), ("done", "fixed the parser, tests green", "claude-opus-5-5"))
+ self.assertEqual((r.usage.inp, r.usage.cw, r.usage.cr, r.usage.out), (1000, 2000, 3000, 4000))
+ self.assertEqual(C.decode_patch(r), b"diff --git a/x b/x\n")
+
+ def test_header_wrapped_anywhere(self):
+ lines = self.result(b"abc")
+ head = lines[4]
+ lines[4:5] = [head[:30], head[30:61], head[61:]] # sha split over three lines
+ self.assertEqual(C.decode_patch(C.parse_result(lines)), b"abc")
+
+ def test_mismatches(self):
+ with self.assertRaisesRegex(C.CloudError, "sha256 mismatch"):
+ C.decode_patch(C.parse_result(self.result(b"abc", sha="0" * 64)))
+ with self.assertRaisesRegex(C.CloudError, "bytes .* != 7 announced"):
+ C.decode_patch(C.parse_result(self.result(b"abc", n=7)))
+ with self.assertRaisesRegex(C.CloudError, "bad WF-RESULT 'maybe'"):
+ C.parse_result(["WF-RESULT maybe"])
+ with self.assertRaisesRegex(C.CloudError, "without WF-PATCH-END"):
+ C.parse_result(self.result(b"abc")[:-1])
+ self.assertIsNone(C.parse_result(["WF-RESULT handback", "WF-REPORT no"]).usage)
+
+ def keyless(self, raw: bytes) -> str:
+ """The keys-dropped shape seen live: '● done', report, usage, header split, base64, bare WF-PATCH-END."""
+ import base64, gzip, hashlib
+ gz = gzip.compress(raw)
+ b64 = base64.b64encode(gz).decode()
+ msg = ["done", "Added sub and a test.", "usage in=4 cw=5 cr=6 out=7 model=claude-opus-5-5",
+ f"sha256={hashlib.sha256(gz).hexdigest()}", f"bytes={len(gz)}",
+ *[b64[i:i + 76] for i in range(0, len(b64), 76)], "WF-PATCH-END"]
+ return "\n".join(("● " if i == 0 else " ") + ln for i, ln in enumerate(msg)) + "\n\n"
+
+ def test_keyless_final_message(self):
+ prompt = as_export("do it\nend with\n [WF-PATCH-END]\nWF-PATCH-END", wrap=False)
+ text = prompt + "● working\n\n" + self.keyless(b"diff --git a/x b/x\n") + "● Session resumed\n"
+ self.assertIsNone(C.final_message(text))
+ lines = C.keyless_message(text)
+ self.assertEqual((lines[0], lines[-1]), ("done", "WF-PATCH-END"))
+ r = C.keyless_result(lines)
+ self.assertEqual((r.state, r.model, r.usage.inp, r.usage.out), ("handback", "claude-opus-5-5", 4, 7))
+ self.assertEqual(C.decode_patch(r), b"diff --git a/x b/x\n")
+ self.assertIsNone(C.keyless_message(prompt)) # prompt only: running
+ self.assertIsNone(C.keyless_message(prompt + "● still working\n"))
+ self.assertIsNone(C.keyless_message(text + as_export("redo with the keys", wrap=False))) # redo sent
+ self.assertIsNotNone(C.keyless_message(text + "❯ \n")) # empty input line
+ r = C.keyless_result(["done", "no patch here", "WF-PATCH-END"])
+ self.assertEqual((r.has_patch, r.usage), (False, None))
+
+ def test_patch_files_and_problem(self):
+ patch = ("diff --git a/src/a.py b/src/a.py\n--- a/src/a.py\n+++ b/src/a.py\n"
+ "diff --git a/old.txt b/new.txt\nrename from old.txt\nrename to new.txt\n")
+ self.assertEqual(C.patch_files(patch), ["src/a.py", "old.txt", "new.txt"])
+ never = ["TASKS.md", "tasks/archive.md", ".wf", "out", "data/pack"]
+ self.assertIsNone(C.patch_problem(["src/a.py", "outish.txt"], "", never))
+ self.assertIn("TASKS.md", C.patch_problem(["TASKS.md"], "", never))
+ self.assertIn("data/pack", C.patch_problem(["data/pack/x"], "", never))
+ self.assertIn("outside the tree", C.patch_problem(["../x"], "", never))
+ self.assertIn("outside the tree", C.patch_problem([".git/hooks/x"], "", never))
+ self.assertIn("outside the project folder", C.patch_problem(["other/x"], "proj", never))
+ self.assertIsNone(C.patch_problem(["proj/src/a.py"], "proj", never))
+ self.assertIn("out", C.patch_problem(["proj/out/x"], "proj", never))
+
+
+class Pull(CloudProject):
+ """wf cloud pull --export FIXTURE on a git project with a recorded send."""
+ toml = None # set in setUp: no worktree_setup (it writes log.txt), a quick_gate
+
+ def setUp(self):
+ from test_cli import TOML
+ self.toml = (TOML + 'cloud = true\ncloud_include = ["out/pack"]\n'
+ 'quick_gate = ["grep -q fixed src/app.py"]\n')
+ super().setUp()
+ os.environ.update({"GIT_AUTHOR_NAME": "t", "GIT_AUTHOR_EMAIL": "t@t", "GIT_COMMITTER_NAME": "t",
+ "GIT_COMMITTER_EMAIL": "t@t"})
+ (self.root / "src" / "app.py").write_text("v1\n")
+ import wf
+ from wflib import config
+ cfg = config.load(self.root)
+ self.master, self.base, _ = wf_cloud.snapshot(cfg, self.snap)
+ self.sid = "session_01PullPullPullPull"
+ with wf_cloud.locked(self.st) as led:
+ C.add(led, "t-four", str(self.root), self.sid, C.MODEL, T0)
+ self.rec.parent.mkdir(parents=True, exist_ok=True)
+ self.rec.write_text(json.dumps({"id": "t-four", "sid": self.sid, "project": str(self.root), "lane": "slow",
+ "master": self.master, "base": self.base, "folder": str(self.snap),
+ "sent": T0.isoformat(), "model": C.MODEL, "bytes": 1}))
+ with redirect_stdout(io.StringIO()):
+ self.assertEqual(wf.main(["--project", str(self.root), "status", "t-four", "progress",
+ f"cloud:{self.sid}"]), 0)
+ self.exp = self.d / "export.txt"
+
+ def change(self, files: dict[str, str]):
+ """Commit files in the snapshot (the cloud's work) -> the gzip patch bytes."""
+ import gzip
+ for rel, text in files.items():
+ (self.snap / rel).parent.mkdir(parents=True, exist_ok=True)
+ (self.snap / rel).write_text(text)
+ git(self.snap, "add", "-A", "-f")
+ git(self.snap, "-c", "user.name=c", "-c", "user.email=c@x", "commit", "-qm", "cloud work")
+ p = subprocess.run(["git", "-C", str(self.snap), "format-patch", "--binary", f"{self.base}..HEAD", "--stdout"],
+ capture_output=True, check=True).stdout
+ return gzip.compress(p)
+
+ def write_export(self, state="done", gz=b"", report="fixed app", sha=None, wrap=False, final=True):
+ import base64, hashlib
+ b64 = base64.b64encode(gz).decode()
+ msg = [f"WF-RESULT {state}", f"WF-REPORT {report}", "WF-USAGE in=1000 cw=0 cr=0 out=1000 model=claude-opus-5-5",
+ f"WF-PATCH-BEGIN sha256={sha or hashlib.sha256(gz).hexdigest()} bytes={len(gz)}",
+ *[b64[i:i + 76] for i in range(0, len(b64), 76)], "WF-PATCH-END"]
+ if wrap: # the export soft-wraps at 60 columns: the header too
+ msg = [part for ln in msg for part in textwrap.wrap(ln, 60, break_long_words=True)]
+ text = as_export("prompt with ● WF-RESULT done inside\nWF-PATCH-BEGIN", wrap=False) + "\n"
+ if final:
+ text += "\n".join(("● " if i == 0 else " ") + ln for i, ln in enumerate(msg)) + "\n\n❯ \n"
+ self.exp.write_text(text)
+
+ def pull(self, *extra, now=T0 + dt.timedelta(hours=1), export=True):
+ out, err = io.StringIO(), io.StringIO()
+ argv = ["pull", "t-four", "--project", str(self.root), "--no-push",
+ *(["--export", str(self.exp)] if export else []), *extra]
+ with redirect_stdout(out), redirect_stderr(err):
+ code = wf_cloud.main(argv, state=self.st, now=now)
+ return code, out.getvalue(), err.getvalue()
+
+ def entry(self):
+ return self.ledger()["entries"][0]
+
+ def assert_ended(self, state):
+ self.assertEqual(self.entry()["state"], state)
+ self.assertFalse(self.snap.exists())
+ self.assertFalse(self.rec.exists())
+
+ def assert_handback(self, why, kept: bool):
+ self.assert_ended("handback")
+ self.assertIn(f"Recovery: cloud attempt {self.sid} — {why}", self.tasks())
+ self.assertNotIn("in progress", self.tasks())
+ self.assertEqual(self.sh(self.root, "rev-parse", "master"), self.master)
+ self.assertEqual(self.sh(self.root, "branch", "--list", "slow/*"), "")
+ self.assertEqual((self.root / "out" / "cloud" / "t-four.patch").exists(), kept)
+
+ def test_done_archives_session(self):
+ self.write_export(gz=self.change({"src/app.py": "fixed\n"}))
+ code, out, err = self.pull()
+ self.assertEqual((code, err), (0, ""), out)
+ self.assertEqual(self.posts, [(f"https://api.anthropic.com/v1/code/sessions/{self.sid}/archive",
+ {"Authorization": "Bearer tok-1", "Content-Type": "application/json",
+ "anthropic-version": "2023-06-01", "User-Agent": "claude-code/9.9.9"})])
+ self.assertIs(self.entry()["archived"], True)
+ self.assertNotIn("archive", out.replace("archive.md", ""))
+
+ def test_second_pull_same_id_skips(self):
+ """Two pulls of one id at once (batch sidecar + orchestrator): the one without the record lock skips
+ with rc 0, no 'branch … exists (local WIP?)'; the holder's pull still ends the task."""
+ import fcntl
+ self.write_export(gz=self.change({"src/app.py": "fixed\n"}))
+ fd = os.open(self.rec, os.O_RDONLY)
+ try:
+ fcntl.flock(fd, fcntl.LOCK_EX | fcntl.LOCK_NB) # the other pull, mid-flight
+ code, out, err = self.pull()
+ finally:
+ os.close(fd)
+ self.assertEqual((code, err), (0, ""), out)
+ self.assertEqual(out, "t-four: pulled by another wf cloud pull, skipped\n")
+ self.assertTrue(self.rec.exists())
+ self.assertEqual(self.entry()["state"], "running")
+ self.assertEqual(self.sh(self.root, "branch", "--list", "slow/*"), "")
+ code, out, err = self.pull()
+ self.assertEqual((code, err), (0, ""), out)
+ self.assert_ended("done")
+
+ def test_pull_claim_record_gone(self):
+ with wf_cloud.pull_claim(self.d / "gone.json") as mine:
+ self.assertFalse(mine)
+ with wf_cloud.pull_claim(self.rec) as mine:
+ self.assertTrue(mine)
+ self.rec.unlink() # winner ended it while we waited: re-checked on the next claim
+ with wf_cloud.pull_claim(self.rec) as mine:
+ self.assertFalse(mine)
+
+ def archive_fails(self, answer, why):
+ self.answer = answer
+ self.write_export(state="handback", report="cannot")
+ code, out, err = self.pull()
+ self.assertEqual((code, err), (0, ""), out)
+ self.assert_ended("handback")
+ self.assertIn(f"archive {self.sid} failed: {why}; archive by hand (wf cloud archive --ended)", out)
+ self.assertNotIn("archived", self.entry())
+
+ def test_archive_http_error_warns_pull_still_ends(self):
+ self.archive_fails(403, "HTTP 403: body")
+
+ def test_archive_network_error_warns_pull_still_ends(self):
+ self.archive_fails(C.CloudError("timed out"), "timed out")
+
+ def test_running_not_archived(self):
+ self.write_export(final=False)
+ code, out, err = self.pull()
+ self.assertEqual(code, 4, out + err)
+ self.assertEqual(self.posts, [])
+
+ def test_archive_command_ended(self):
+ self.write_export(state="handback", report="cannot")
+ self.answer = 500
+ self.pull()
+ self.answer = 200
+ out = io.StringIO()
+ with redirect_stdout(out):
+ code = wf_cloud.main(["archive", "--ended"], state=self.st, now=T0)
+ self.assertEqual((code, out.getvalue()), (0, f"archived {self.sid}\n"))
+ self.assertIs(self.entry()["archived"], True)
+ with redirect_stdout(out := io.StringIO()):
+ code = wf_cloud.main(["archive", "--ended"], state=self.st, now=T0)
+ self.assertEqual((code, out.getvalue()), (0, "no ended cloud session left to archive\n"))
+
+ def test_archive_command_one_sid_failure_exit_1(self):
+ self.answer = 401
+ err = io.StringIO()
+ with redirect_stdout(io.StringIO()), redirect_stderr(err):
+ code = wf_cloud.main(["archive", "session_01Other"], state=self.st, now=T0)
+ self.assertEqual(code, 1)
+ self.assertEqual(err.getvalue(), "wf: archive session_01Other failed: HTTP 401 (login expired? run claude once)\n")
+
+ def test_done_applies_gates_merges(self):
+ self.write_export(gz=self.change({"src/app.py": "fixed\n", "src/new.py": "new\n"}))
+ code, out, err = self.pull()
+ self.assertEqual((code, err), (0, ""), out)
+ self.assertEqual(self.sh(self.root, "show", "master:src/app.py"), "fixed")
+ self.assertEqual(self.sh(self.root, "show", "master:src/new.py"), "new")
+ self.assertIn("cloud work", self.sh(self.root, "log", "--format=%s", "master"))
+ self.assertNotIn("t-four", self.tasks())
+ self.assertIn(f"fixed app (cloud {self.sid})", self.archive())
+ self.assertEqual(self.sh(self.root, "status", "--porcelain", "TASKS.md", "tasks"), "") # bookkeeping committed
+ self.assert_ended("done")
+ e = self.entry()
+ self.assertEqual(e["usd_source"], "self")
+ self.assertAlmostEqual(e["usd"], U.cost(C.MODEL, U.Usage(inp=1000, out=1000)) * C.OVERHEAD)
+ self.assertFalse((self.root / "out" / "cloud" / "t-four.patch").exists())
+ self.assertRegex(out, r"report: commit [0-9a-f]{7,}\n$")
+ self.assertIn("$ grep -q fixed src/app.py", out)
+
+ def test_wrapped_indented_base64_rejoined(self):
+ self.write_export(gz=self.change({"src/app.py": "fixed\n"}), wrap=True)
+ code, out, err = self.pull()
+ self.assertEqual((code, err), (0, ""), out)
+ self.assertEqual(self.sh(self.root, "show", "master:src/app.py"), "fixed")
+
+ def test_sha_mismatch_hands_back(self):
+ self.write_export(gz=self.change({"src/app.py": "fixed\n"}), sha="0" * 64)
+ code, out, err = self.pull()
+ self.assertEqual(code, 0, err)
+ self.assert_handback("patch sha256 mismatch", kept=False)
+
+ def test_gate_red_hands_back_keeps_patch(self):
+ self.write_export(gz=self.change({"src/app.py": "broken\n"}))
+ code, out, err = self.pull()
+ self.assertEqual(code, 0, err)
+ self.assert_handback("quick_gate 'grep -q fixed src/app.py' red (exit 1)", kept=True)
+ wt = self.root / ".worktrees" / "slow"
+ self.assertEqual(self.sh(wt, "status", "--porcelain"), "")
+ self.assertEqual(self.sh(wt, "branch", "--show-current"), "")
+
+ def test_patch_touching_cloud_include_refused(self):
+ self.write_export(gz=self.change({"src/app.py": "fixed\n", "out/pack/data.bin": "x\n"}))
+ code, out, err = self.pull()
+ self.assertEqual(code, 0, err)
+ self.assert_handback("patch touches out/pack/data.bin", kept=True)
+
+ def test_patch_touching_tasks_refused(self):
+ self.write_export(gz=self.change({"src/app.py": "fixed\n", "TASKS.md": "x\n"}))
+ code, out, err = self.pull()
+ self.assertEqual(code, 0, err)
+ self.assert_handback("patch touches TASKS.md", kept=True)
+
+ def test_awaiting_adds_question_and_blocks(self):
+ self.write_export(state="awaiting", report="Which format: csv or json?")
+ code, out, err = self.pull()
+ self.assertEqual(code, 0, err)
+ self.assert_ended("awaiting")
+ m = __import__("re").search(r"\*\*(a-[^*]+)\*\*: Which format: csv or json\? \[\[t-four\]\]", self.tasks())
+ self.assertTrue(m, self.tasks())
+ self.assertIn(f"(blocked: [[{m[1]}]])", self.tasks())
+ self.assertEqual(self.sh(self.root, "rev-parse", "master"), self.master)
+
+ def test_cloud_handback(self):
+ self.write_export(state="handback", report="needs the LAN host")
+ code, out, err = self.pull()
+ self.assertEqual(code, 0, err)
+ self.assert_handback("cloud handback: needs the LAN host", kept=False)
+
+ def test_no_marker_running_exit_4(self):
+ self.write_export(final=False)
+ code, out, err = self.pull()
+ self.assertEqual((code, err), (4, ""))
+ self.assertIn(f"t-four: running ({self.sid}, sent 1h00m ago)", out)
+ self.assertTrue(self.rec.exists() and self.snap.exists())
+ self.assertEqual(self.entry()["state"], "running")
+ self.assertIn(f"in progress: cloud:{self.sid}", self.tasks())
+
+ def test_keyless_final_message_hands_back_keeps_patch(self):
+ import gzip
+ self.write_export(final=False)
+ msg = PullPure().keyless(gzip.decompress(self.change({"src/app.py": "fixed\n"})))
+ self.exp.write_text(self.exp.read_text() + msg + "● Session resumed\n")
+ code, out, err = self.pull()
+ self.assertEqual(code, 0, err)
+ self.assert_handback("bad result: no WF-RESULT key; patch out/cloud/t-four.patch", kept=True)
+ self.assertIn(b"+fixed", (self.root / "out" / "cloud" / "t-four.patch").read_bytes())
+ self.assertEqual(self.entry()["usd_source"], "self")
+
+ def test_teleport_export_used_and_removed(self):
+ self.write_export(final=False)
+ seen = []
+
+ def fake_export(folder, sid, out, timeout=300):
+ seen.append((Path(folder), sid))
+ Path(out).parent.mkdir(parents=True, exist_ok=True)
+ Path(out).write_text(self.exp.read_text())
+ return Path(out)
+
+ old, wf_cloud.export = wf_cloud.export, fake_export
+ try:
+ code, out, err = self.pull(export=False)
+ finally:
+ wf_cloud.export = old
+ self.assertEqual((code, seen), (4, [(self.snap, self.sid)]))
+ self.assertFalse((self.root / "out" / "cloud" / "t-four.export.txt").exists())
+
+ def test_over_24h_lost(self):
+ self.write_export(final=False)
+ code, out, err = self.pull(now=T0 + dt.timedelta(hours=25))
+ self.assertEqual(code, 0, err)
+ self.assert_ended("lost")
+ self.assertEqual((self.entry()["usd"], self.entry()["usd_source"]), (4.0, "est"))
+ self.assertIn(f"Recovery: cloud attempt {self.sid} — lost: no WF-RESULT after 25h00m", self.tasks())
+ self.assertNotIn("in progress", self.tasks())
+
+ def test_all_and_not_sent(self):
+ os.environ["FAKE_MODE"] = "noresume" # teleport never resumes: export fails
+ out, err = io.StringIO(), io.StringIO()
+ with redirect_stdout(out), redirect_stderr(err):
+ code = wf_cloud.main(["pull", "--all", "--project", str(self.root)], state=self.st,
+ now=T0 + dt.timedelta(hours=25))
+ self.assertEqual(code, 0, err.getvalue()) # no export, > 24h: lost
+ self.assert_ended("lost")
+ code, _, err = self.pull()
+ self.assertEqual(code, 1)
+ self.assertIn("t-four is not out in the cloud", err)
+
+
+# ------------------------------------------------------------------ fit (spec §4.5)
+
+FIT_DOC = """\
+# Tasks — demo
+
+## Awaiting your decision
+
+## Pending
+
+- **t-ok** [P1] (1h): Plain code. Goal.
+ Done: tests pass
+- **t-sonnet** [P1] (1h): Plain code. Goal.
+ Done: tests pass
+ Model: sonnet
+- **t-slice** [P1] (5h): Big. Goal.
+ Done: tests pass
+- **t-owner** [P1] (1h): Plain. Goal.
+ Done: tests pass
+ Sessions: owner
+- **t-solo** [P1] (1h): Plain. Goal.
+ Done: tests pass
+ Sessions: solo
+- **t-nodone** [P1] (1h): Plain. Goal.
+- **t-no** [P1] (1h): Plain. Goal.
+ Done: tests pass
+ Cloud: no
+- **t-wf** [P1] (1h): Plain. Goal.
+ Steps: run wf set x --model opus then check
+ Done: tests pass
+- **t-res** [P1] (1h): Plain. Goal.
+ Steps: wf res run the build
+ Done: tests pass
+- **t-gui** [P1] (1h): Plain. Goal.
+ Steps: click through the GUI
+ Done: tests pass
+- **t-lan** [P1] (1h): Plain. Goal.
+ Steps: copy to 10.0.0.20
+ Done: tests pass
+- **t-srv** [P1] (1h): Plain. Goal.
+ Steps: hit the live server
+ Done: tests pass
+- **t-yes-wf** [P1] (1h): Plain. Goal.
+ Steps: run wf set x --model opus then check
+ Done: tests pass
+ Cloud: yes
+- **t-yes-sonnet** [P1] (1h): Plain. Goal.
+ Done: tests pass
+ Model: sonnet
+ Cloud: yes
+- **t-yes-owner** [P1] (1h): Plain. Goal.
+ Done: tests pass
+ Sessions: owner
+ Cloud: yes
+- **t-yes-haiku** [P1] (<1h): Plain. Goal.
+ Done: tests pass
+ Model: haiku
+ Cloud: yes
+- **t-no-opus** [P1] (1h): Plain. Goal.
+ Done: tests pass
+ Model: opus
+ Cloud: no
+- **t-yes-slice** [P1] (5h): Big. Goal.
+ Done: tests pass
+ Cloud: yes
+
+## Needs human
+
+## Deferred
+"""
+
+
+class FitTest(unittest.TestCase):
+ def setUp(self):
+ from wflib import tasks as TK
+ self.TK = TK
+ self.doc = TK.parse(FIT_DOC)
+ self.cfg = type("Cfg", (), {"cloud": True, "slice_above": "1h"})()
+
+ def fit(self, id):
+ return C.fit(self.doc.item(id), self.cfg)
+
+ def test_fits(self):
+ self.assertIsNone(self.fit("t-ok"))
+
+ def test_exclusions(self):
+ for id, why in [("t-sonnet", "Model sonnet"), ("t-slice", "slice"), ("t-owner", "not runner-ready"),
+ ("t-solo", "Sessions: solo"), ("t-nodone", "not runner-ready"), ("t-no", "Cloud: no"),
+ ("t-wf", "`wf` commands"), ("t-res", "wf res"), ("t-gui", "GUI"), ("t-lan", "LAN"),
+ ("t-srv", "live server")]:
+ with self.subTest(id):
+ self.assertIn(why, self.fit(id) or "")
+
+ def test_not_opted_in(self):
+ self.cfg.cloud = False
+ self.assertIn("not opted in", self.fit("t-ok"))
+
+ def test_cloud_yes_skips_regex_only_opus_only(self):
+ self.assertIsNone(self.fit("t-yes-wf"))
+ self.assertEqual(self.fit("t-yes-sonnet"), "Model sonnet stays local")
+ self.assertEqual(self.fit("t-yes-haiku"), "Model haiku stays local")
+ self.assertEqual(self.fit("t-no-opus"), "Cloud: no")
+ self.assertEqual(self.fit("t-sonnet"), "Model sonnet stays local") # unset: old rule
+ self.assertIn("not runner-ready", self.fit("t-yes-owner"))
+ self.assertIn("slice", self.fit("t-yes-slice"))
+
+ def test_cloud_property_and_set(self):
+ self.assertEqual(self.doc.item("t-no").cloud, "no")
+ self.assertIsNone(self.doc.item("t-ok").cloud)
+ self.TK.set_fields(self.doc, "t-ok", cloud="yes")
+ self.assertEqual(self.doc.item("t-ok").body, [" Done: tests pass", " Cloud: yes"])
+ self.TK.set_fields(self.doc, "t-ok", cloud="no", model="sonnet")
+ self.assertEqual(self.doc.item("t-ok").body, [" Done: tests pass", " Model: sonnet", " Cloud: no"])
+ self.TK.set_fields(self.doc, "t-ok", cloud="")
+ self.assertEqual(self.doc.item("t-ok").body, [" Done: tests pass", " Model: sonnet"])
+ with self.assertRaises(self.TK.TaskError):
+ self.TK.set_fields(self.doc, "t-ok", cloud="maybe")
+
+
+class PickKeyTest(unittest.TestCase):
+ def test_order(self):
+ from wflib import tasks as TK
+ doc = TK.parse("""# Tasks
+
+## Pending
+
+- **t-p1-small** [P1] (<1h): A. G.
+ Done: x
+- **t-p1-big** [P1] (1h): A. G.
+ Done: x
+- **t-p2-yes** [P2] (<1h): A. G.
+ Done: x
+ Cloud: yes
+- **t-p0** [P0] (<1h): A. G.
+ Done: x
+
+## Needs human
+
+## Deferred
+""")
+ items = [doc.item(i) for i in ("t-p1-small", "t-p1-big", "t-p2-yes", "t-p0")]
+ got = [i.id for i in sorted(items, key=C.pick_key)]
+ # yes first; then prio; then 1h before <1h
+ self.assertEqual(got, ["t-p2-yes", "t-p0", "t-p1-big", "t-p1-small"])
+
+
+class CloudLineCheck(unittest.TestCase):
+ def test_check(self):
+ sys.path.insert(0, str(HERE / "tests"))
+ from test_check import Base
+
+ class B(Base):
+ def runTest(self):
+ self.pending("- **t-a** [P1] (1h): A. G.\n Done: x\n Cloud: yes\n")
+ self.assertEqual(self.run_check(), ([], []))
+ self.pending("- **t-a** [P1] (1h): A. G.\n Done: x\n Cloud: maybe\n")
+ self.assertTrue(any("Cloud 'maybe'" in e for e in self.errors()))
+ self.pending("- **t-a** [P1] (1h): A. G.\n Done: x\n Cloud: yes\n Cloud: no\n")
+ self.assertTrue(any("two Cloud lines" in e for e in self.errors()))
+ r = unittest.TextTestRunner(stream=io.StringIO()).run(B())
+ self.assertTrue(r.wasSuccessful(), r.failures + r.errors)
+
+
+ORCH_TASKS = """\
+# Tasks — demo
+
+## Awaiting your decision
+
+## Pending
+
+- **t-fast** [P1] (<1h): Fast sonnet task.
+ - Done: fast works
+ - Model: sonnet
+
+- **t-gui** [P1] (<1h): Check the GUI screenshot.
+ - Done: gui works
+
+- **t-hai** [P1] (<1h): Haiku forced to the cloud.
+ - Done: hai works
+ - Model: haiku
+ - Cloud: yes
+
+- **t-slow** [P2] (1h): Slow opus task.
+ - Done: slow works
+
+- **t-yes** [P3] (<1h): Opus task marked for the cloud.
+ - Done: yes works
+ - Cloud: yes
+
+## Needs human
+
+## Deferred
+"""
+
+
+class OrchCloud(CloudProject):
+ """wf orch pick cloud (subprocess, fake claude via WF_CLAUDE, ledger via XDG_STATE_HOME)."""
+ tasks_text = ORCH_TASKS
+
+ def setUp(self):
+ super().setUp()
+ from test_merge import IDENT
+ self.env = {**IDENT, "WF_CLAUDE": wf_cloud.CLAUDE, "XDG_STATE_HOME": str(self.d / "xdg")}
+
+ def orch(self, *args):
+ out = self.ok("orch", *args, env=self.env)
+ return out
+
+ def put_ledger(self, **kw):
+ n = kw.pop("running", 0)
+ led = {**C.new(), **kw}
+ for i in range(n):
+ C.add(led, f"t-x{i}", "/x", f"session_x{i}", "opus", T0)
+ (self.d / "xdg" / "wf").mkdir(parents=True, exist_ok=True)
+ (self.d / "xdg" / "wf" / "cloud.json").write_text(C.dumps(led))
+
+ def status(self, id):
+ return next(l for l in (self.root / "TASKS.md").read_text().splitlines() if f"**{id}**" in l)
+
+ def test_pick_cloud_yes_first_then_opus_then_none(self):
+ out = self.orch("pick", "cloud")
+ self.assertIn("pick: t-yes (lane cloud, model opus", out) # Cloud: yes beats P2 1h; haiku never
+ out = self.orch("pick", "cloud")
+ self.assertIn("pick: t-slow (lane cloud, model opus", out)
+ self.assertIn("in progress: cloud:session_01AbCdEfGhIjKlMnOpQrStUv", self.status("t-slow"))
+ rec = json.loads((self.root / ".wf" / "orch" / "t-slow.json").read_text())
+ self.assertEqual((rec["lane"], rec["task_lane"], rec["branch"]), ("cloud", "slow", "slow/t-slow"))
+ self.assertTrue((self.root / ".wf" / "cloud" / "t-slow.json").exists())
+ out = self.orch("pick", "cloud")
+ self.assertEqual(out.strip().splitlines()[0][:26], "stop lane cloud: none fit ")
+ led = json.loads((self.d / "xdg" / "wf" / "cloud.json").read_text())
+ self.assertEqual([e["id"] for e in led["entries"]], ["t-yes", "t-slow"])
+
+ def test_local_lane_skips_cloud_claim(self):
+ self.orch("pick", "cloud", "--id", "t-slow")
+ out = self.orch("pick", "slow")
+ self.assertNotIn("t-slow", out.splitlines()[0])
+ self.assertIn("pick: t-fast", out)
+
+ def test_stop_ledger(self):
+ self.put_ledger(budget=3.0)
+ out = self.orch("pick", "cloud")
+ self.assertIn("stop lane cloud: ledger (balance $3.00 < reserve $4.00)", out)
+ self.assertNotIn("in progress", self.status("t-slow"))
+
+ def test_stop_max_parallel(self):
+ self.put_ledger(max_parallel=2, running=2)
+ out = self.orch("pick", "cloud")
+ self.assertIn("stop lane cloud: max parallel (2 running >= max_parallel 2)", out)
+ self.assertFalse((self.root / ".wf" / "cloud" / "t-slow.json").exists())
+
+ def test_id_unfit_refused(self):
+ out = self.orch("pick", "cloud", "--id", "t-gui")
+ self.assertIn("stop lane cloud: none fit (t-gui: needs a GUI)", out)
+
+ def test_not_opted_in(self):
+ (self.root / "workflow.toml").write_text(GitCli.toml)
+ out = self.orch("pick", "cloud")
+ self.assertIn("stop lane cloud: none fit (project not opted in", out)
+
+ def test_post_handback_commits_and_stops(self):
+ self.orch("pick", "cloud", "--id", "t-slow")
+ self.ok("note", "t-slow", "Recovery: cloud attempt x — red") # what wf cloud pull leaves behind
+ self.ok("status", "t-slow", "clear")
+ out = self.orch("post", "t-slow", "cloud", "--result", "handback")
+ self.assertIn("post: t-slow handback", out)
+ self.assertIn("committed leftover", out)
+ self.assertIn("stop lane cloud: handback", out)
+ self.assertIn(" cloud opus t-slow handback ", (self.root / "out" / "wf-orch.log").read_text())
diff --git a/tests/test_config.py b/tests/test_config.py
new file mode 100644
index 0000000..e51d2bc
--- /dev/null
+++ b/tests/test_config.py
@@ -0,0 +1,134 @@
+import sys
+import tempfile
+import unittest
+from pathlib import Path
+
+sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
+from wflib import config, lanes as L
+from wflib import config as C
+
+
+class ConfigTest(unittest.TestCase):
+ def setUp(self):
+ self.tmp = tempfile.TemporaryDirectory()
+ self.root = Path(self.tmp.name).resolve()
+
+ def tearDown(self):
+ self.tmp.cleanup()
+
+ def write(self, text):
+ (self.root / "workflow.toml").write_text(text)
+
+ def test_find_root_from_nested_dir(self):
+ self.write('format = 1\ntasks = "TASKS.md"\narchive = "a.md"\n')
+ deep = self.root / "a" / "b"
+ deep.mkdir(parents=True)
+ self.assertEqual(C.find_root(deep), self.root)
+
+ def test_find_root_none(self):
+ with self.assertRaisesRegex(C.ConfigError, "no workflow.toml in .* or above"):
+ C.find_root(self.root)
+
+ def test_minimal(self):
+ self.write('format = 1\ntasks = "TASKS.md"\narchive = "tasks/archive.md"\n')
+ cfg = C.load(self.root)
+ self.assertEqual((cfg.format, cfg.tasks, cfg.archive), (1, self.root / "TASKS.md", self.root / "tasks/archive.md"))
+ self.assertEqual((cfg.docs, cfg.verify, cfg.done, cfg.ledgers, cfg.anchors_index), ([], [], [], None, None))
+
+ def test_full(self):
+ self.write('format = 1\ntasks = "docs/TASKS.md"\narchive = "docs/DONE.md"\ndocs = ["DESIGN.md", "docs/"]\n'
+ 'verify = ["make test"]\ndone = ["update CATALOG"]\nledgers = ".superpowers/sdd"\n'
+ '[anchors]\nindex = "DESIGN.md"\nindex_section = "Subsystems"\nspecs = "docs/specs"\n')
+ cfg = C.load(self.root)
+ self.assertEqual(cfg.docs, [self.root / "DESIGN.md", self.root / "docs"])
+ self.assertEqual((cfg.verify, cfg.done), (["make test"], ["update CATALOG"]))
+ self.assertEqual(cfg.ledgers, self.root / ".superpowers/sdd")
+ self.assertEqual((cfg.anchors_index, cfg.anchors_section, cfg.anchors_specs),
+ (self.root / "DESIGN.md", "Subsystems", self.root / "docs/specs"))
+
+ def test_cloud_keys(self):
+ self.write('format = 1\ntasks = "T.md"\narchive = "a.md"\n')
+ cfg = C.load(self.root)
+ self.assertEqual((cfg.cloud, cfg.cloud_include, cfg.cloud_note), (False, [], None))
+ self.write('format = 1\ntasks = "T.md"\narchive = "a.md"\ncloud = true\ncloud_include = ["out/p"]\n'
+ 'cloud_note = "data in out/p"\n')
+ cfg = C.load(self.root)
+ self.assertEqual((cfg.cloud, cfg.cloud_include, cfg.cloud_note), (True, ["out/p"], "data in out/p"))
+ self.write('format = 1\ntasks = "T.md"\narchive = "a.md"\ncloud = "yes"\n')
+ with self.assertRaisesRegex(C.ConfigError, "'cloud' must be true or false"):
+ C.load(self.root)
+
+ def test_unknown_key(self):
+ self.write('format = 1\ntasks = "T.md"\narchive = "a.md"\nverfy = []\n')
+ with self.assertRaisesRegex(C.ConfigError, "workflow.toml: unknown key 'verfy'"):
+ C.load(self.root)
+
+ def test_unknown_anchors_key(self):
+ self.write('format = 1\ntasks = "T.md"\narchive = "a.md"\n[anchors]\nindx = "D.md"\n')
+ with self.assertRaisesRegex(C.ConfigError, r"unknown key 'anchors.indx'"):
+ C.load(self.root)
+
+ def test_missing_required(self):
+ self.write('format = 1\narchive = "a.md"\n')
+ with self.assertRaisesRegex(C.ConfigError, "workflow.toml: missing 'tasks'"):
+ C.load(self.root)
+
+ def test_wrong_type(self):
+ self.write('format = 1\ntasks = "T.md"\narchive = "a.md"\nverify = "make"\n')
+ with self.assertRaisesRegex(C.ConfigError, "'verify' must be a list of strings"):
+ C.load(self.root)
+
+ def test_bad_toml(self):
+ self.write('format = \n')
+ with self.assertRaisesRegex(C.ConfigError, "workflow.toml: "):
+ C.load(self.root)
+
+
+
+
+
+class LaneConfigTest(unittest.TestCase):
+ def load(self, extra: str):
+ d = Path(tempfile.mkdtemp())
+ self.addCleanup(__import__("shutil").rmtree, d)
+ (d / "workflow.toml").write_text('format = 1\ntasks = "TASKS.md"\narchive = "a.md"\n' + extra)
+ return config.load(d)
+
+ def test_default_lanes(self):
+ cfg = self.load("")
+ self.assertEqual(cfg.slice_above, "1h")
+ self.assertEqual(cfg.lanes, (L.Lane("fast", ("<1h",), "unblock", None, True),
+ L.Lane("slow", ("1h",), "priority", "fast", False)))
+ self.assertEqual(cfg.area_stale_commits, 20)
+ self.assertEqual(cfg.areas_file, cfg.root / "CLAUDE.md")
+
+ def test_custom_lanes_replace_default(self):
+ cfg = self.load('slice_above = "5h"\n[lanes.quick]\nefforts = ["<1h", "1h"]\norder = "unblock"\n'
+ '[lanes.long]\nefforts = ["5h"]\nfallback = "quick"\n')
+ self.assertEqual(cfg.lanes, (L.Lane("quick", ("<1h", "1h"), "unblock", None, True),
+ L.Lane("long", ("5h",), "priority", "quick", False)))
+
+ def test_lane_errors(self):
+ bad = {
+ '[lanes.a]\nefforts = ["<1h", "1h"]\n[lanes.b]\nefforts = ["1h"]\n': "effort '1h' in lanes a and b",
+ '[lanes.a]\nefforts = ["<1h"]\n': "effort '1h' (≤ slice_above) in no lane",
+ '[lanes.a]\nefforts = ["<1h"]\nslices = true\n[lanes.b]\nefforts = ["1h"]\nslices = true\n':
+ "slices = true on more than one lane",
+ '[lanes.a]\nefforts = ["<1h"]\nfallback = "b"\n[lanes.b]\nefforts = ["1h"]\nfallback = "a"\n':
+ "fallback cycle: a ↔ b",
+ '[lanes.a]\nefforts = ["<1h", "1h", "5h"]\n': "lane a: effort '5h' above slice_above '1h'",
+ '[lanes.a]\nefforts = ["<1h", "1h"]\norder = "fifo"\n': "lane a: order 'fifo' (want unblock, priority)",
+ '[lanes.all]\nefforts = ["<1h", "1h"]\n': "lane name 'all' is reserved",
+ '[lanes.A]\nefforts = ["<1h", "1h"]\n': "bad lane name 'A'",
+ '[lanes.a]\nefforts = ["<1h", "1h"]\nspeed = 1\n': "unknown key 'lanes.a.speed'",
+ 'slice_above = "2h"\n': "slice_above '2h' (want <1h, 1h, 5h, 10h, 100h)",
+ 'area_stale_commits = 0\n': "'area_stale_commits' must be a number ≥ 1",
+ }
+ for extra, msg in bad.items():
+ with self.subTest(msg=msg), self.assertRaises(config.ConfigError) as cm:
+ self.load(extra)
+ self.assertIn(msg, str(cm.exception))
+
+
+if __name__ == "__main__":
+ unittest.main()
diff --git a/tests/test_ctx_hint.py b/tests/test_ctx_hint.py
new file mode 100644
index 0000000..7750284
--- /dev/null
+++ b/tests/test_ctx_hint.py
@@ -0,0 +1,98 @@
+import json
+import sys
+import unittest
+from pathlib import Path
+
+HERE = Path(__file__).resolve().parent.parent
+sys.path.insert(0, str(HERE))
+sys.path.insert(0, str(HERE / "tests"))
+
+import test_cli # noqa: E402
+from test_usage import entry # noqa: E402
+from wflib import usage # noqa: E402
+
+SID = "99999999-2222-3333-4444-555555555555"
+USER = json.dumps({"type": "user", "timestamp": "2026-10-05T09:00:00Z", "message": {"role": "user", "content": "hi"}})
+
+
+def transcript(*sizes):
+ """One request per size (inp 10, cw 1000, rest cache read), each streamed as 2 entries."""
+ lines = [USER]
+ for n, size in enumerate(sizes):
+ e = entry(f"m{n}", f"r{n}", "claude-opus-5-5", f"2026-10-05T09:0{n}:01Z", inp=10, cw5=1000, cr=size - 1010)
+ lines += [e, e, USER]
+ return "\n".join(lines) + "\n"
+
+
+class ContextTokens(unittest.TestCase):
+ def test_last_request_prompt_size(self):
+ self.assertEqual(usage.context_tokens(transcript(50_000, 152_000)), 152_000)
+
+ def test_skips_synthetic_and_sidechain(self):
+ text = transcript(40_000)
+ synth = json.loads(entry("s", "rs", "<synthetic>", "2026-10-05T10:00:00Z", inp=1))
+ side = json.loads(entry("x", "rx", "claude-opus-5-5", "2026-10-05T10:00:01Z", cr=999_000))
+ side["isSidechain"] = True
+ text += json.dumps(synth) + "\n" + json.dumps(side) + "\n"
+ self.assertEqual(usage.context_tokens(text), 40_000)
+
+ def test_partial_first_line_and_empty(self):
+ self.assertEqual(usage.context_tokens('ens": 5}}}\n' + transcript(30_000)), 30_000)
+ self.assertIsNone(usage.context_tokens(USER + "\n"))
+
+ def test_hint_line(self):
+ self.assertEqual(usage.ctx_hint(152_000, 100_000),
+ "context ~152k tokens (> 100k): ask the owner to /clear, then continue "
+ "(subagent: ignore, this is the main session)")
+ self.assertIsNone(usage.ctx_hint(99_999, 100_000))
+ self.assertIsNone(usage.ctx_hint(None, 100_000))
+ self.assertIsNone(usage.ctx_hint(500_000, 0))
+
+
+class HintCli(test_cli.Cli):
+ def setUp(self):
+ super().setUp()
+ self.home = self.root.parent / "claude-home"
+ (self.home / "projects" / "-work-demo").mkdir(parents=True)
+ self.jsonl = self.home / "projects" / "-work-demo" / f"{SID}.jsonl"
+
+ def env(self, sid=SID):
+ return {"CLAUDE_CONFIG_DIR": str(self.home), "CLAUDE_CODE_SESSION_ID": sid}
+
+ HINT = "context ~152k tokens (> 100k)"
+
+ def test_done_prints_hint_when_big(self):
+ self.jsonl.write_text(transcript(152_000))
+ out = self.ok("done", "t-one", "-m", "ok", env=self.env())
+ self.assertIn(self.HINT, out.splitlines()[-1])
+
+ def test_next_prints_hint_when_big(self):
+ self.jsonl.write_text(transcript(152_000))
+ out = self.ok("next", "--as", "opus", env=self.env())
+ self.assertIn(self.HINT, out.splitlines()[-1])
+
+ def test_silent_when_small_or_missing(self):
+ self.jsonl.write_text(transcript(60_000))
+ self.assertNotIn("context ~", self.ok("next", "--as", "opus", env=self.env()))
+ self.assertNotIn("context ~", self.ok("next", "--as", "opus", env=self.env(sid="")))
+ self.assertNotIn("context ~", self.ok("next", "--as", "opus", env=self.env(sid="nope")))
+
+ def test_threshold_from_toml(self):
+ (self.root / "workflow.toml").write_text(self.toml + "ctx_hint = 200000\n")
+ self.jsonl.write_text(transcript(152_000))
+ self.assertNotIn("context ~", self.ok("next", "--as", "opus", env=self.env()))
+ (self.root / "workflow.toml").write_text(self.toml + "ctx_hint = 0\n")
+ self.jsonl.write_text(transcript(900_000))
+ self.assertNotIn("context ~", self.ok("next", "--as", "opus", env=self.env()))
+
+ def test_bad_threshold(self):
+ (self.root / "workflow.toml").write_text(self.toml + 'ctx_hint = "big"\n')
+ self.assertIn("'ctx_hint' must be a number", self.fails("list"))
+
+ def test_reads_only_the_tail(self):
+ self.jsonl.write_text(transcript(152_000) + (USER + "\n") * 40_000 + transcript(160_000))
+ self.assertIn("context ~160k tokens", self.ok("next", "--as", "opus", env=self.env()))
+
+
+if __name__ == "__main__":
+ unittest.main()
diff --git a/tests/test_finish.py b/tests/test_finish.py
new file mode 100644
index 0000000..38fceb1
--- /dev/null
+++ b/tests/test_finish.py
@@ -0,0 +1,210 @@
+import subprocess
+import unittest
+
+from test_cli import TOML, Cli
+from test_claims import git
+from test_merge import IDENT
+
+
+class FinishTest(Cli):
+ def setUp(self):
+ super().setUp()
+ git(self.root, "init", "-q", "-b", "master")
+ (self.root / ".gitignore").write_text(".worktrees/\n.wf/\n")
+ git(self.root, "add", "-A")
+ git(self.root, "commit", "-qm", "init")
+ self.wt = self.root / ".worktrees" / "sonnet"
+ git(self.root, "worktree", "add", "-q", str(self.wt), "-b", "sonnet/t-three")
+
+ def out(self, *args, cwd=None):
+ return subprocess.run(["git", *args], cwd=cwd or self.root, capture_output=True, text=True).stdout
+
+ def finish(self, *args, cwd=None):
+ return self.wf("finish", *args, project=False, cwd=cwd or self.wt, env=IDENT)
+
+ def write_code(self):
+ (self.wt / "code.txt").write_text("x\n")
+
+ def test_report_line_tool_commit(self):
+ self.write_code()
+ code, out, err = self.finish("t-three", "-m", "ok", "--commit", "impl", "code.txt", "--no-push",
+ "--tool-commit", "abc1234")
+ self.assertEqual(code, 0, err)
+ self.assertRegex(out, r"report: commit [0-9a-f]+ tool abc1234\n$")
+
+ def test_report_line_no_worktree(self):
+ (self.root / "code.txt").write_text("x\n")
+ code, out, err = self.wf("finish", "t-three", "-m", "ok", "--commit", "impl", "code.txt", "--no-push",
+ project=False, cwd=self.root, env=IDENT)
+ self.assertEqual(code, 0, err)
+ sha = self.out("rev-parse", "--short", "HEAD").strip()
+ self.assertTrue(out.endswith(f"report: commit {sha}\n"), out)
+
+ def test_commit_done_merge_in_one(self):
+ self.write_code()
+ code, out, err = self.finish("t-three", "-m", "ok", "--commit", "impl", "code.txt", "--no-push")
+ self.assertEqual((code, err), (0, ""), out)
+ self.assertIn("done: t-three → tasks/archive.md\n", out)
+ self.assertIn("committed code.txt\n", out)
+ self.assertIn("merged sonnet/t-three into master\n", out)
+ sha = self.out("log", "--format=%h", "-n1", "--grep=^impl$", "master").strip()
+ self.assertTrue(out.endswith(f"report: commit {sha}\n"), out)
+ self.assertNotIn("wf merge", out) # no merge-step reminder: finish merged
+ self.assertEqual(self.out("log", "--format=%s", "master"), "t-three done\nimpl\ninit\n")
+ self.assertEqual(self.out("status", "--porcelain"), "")
+ self.assertEqual((self.root / "code.txt").read_text(), "x\n")
+ self.assertIn("t-three", (self.root / "tasks" / "archive.md").read_text())
+
+ def test_wip_commits_notes_clears(self):
+ self.write_code()
+ self.wf("status", "t-three", "progress", "br", project=False, cwd=self.wt, env=IDENT)
+ code, out, err = self.wf("wip", "t-three", "-m", "half done, next: tests", "--commit", "wip", "code.txt",
+ project=False, cwd=self.wt, env=IDENT)
+ self.assertEqual(code, 0, err)
+ self.assertIn("committed code.txt", out)
+ self.assertEqual(self.out("log", "--format=%s", "-n1", cwd=self.wt), "wip\n")
+ self.assertEqual(self.out("status", "--porcelain", cwd=self.wt, ).count("code.txt"), 0)
+ code, out, err = self.wf("show", "t-three", project=False, cwd=self.wt)
+ self.assertIn("half done, next: tests", out)
+ self.assertNotIn("in progress", out)
+
+ def test_no_commit_bookkeeping_only(self):
+ git(self.wt, "switch", "-q", "--detach", "master")
+ code, out, err = self.finish("t-three", "-m", "ok", "--no-push")
+ self.assertEqual((code, err), (0, ""), out)
+ self.assertTrue(out.endswith("committed TASKS.md tasks/archive.md\n"), out)
+ self.assertEqual(self.out("log", "--format=%s", "master"), "bookkeeping\ninit\n")
+
+ def test_commit_tasks_only_no_path_lists_once(self):
+ code, out, err = self.finish("t-three", "-m", "ok", "--commit", "bk", "--no-push")
+ self.assertEqual((code, err), (0, ""), out)
+ self.assertEqual(out.count("TASKS.md"), 1, out)
+ self.assertIn("committed TASKS.md tasks/archive.md\n", out)
+ self.assertEqual(self.out("log", "--format=%s", "master"), "t-three done\ninit\n")
+
+ def test_commit_tasks_path_named_lists_once(self):
+ code, out, err = self.finish("t-three", "-m", "ok", "--commit", "bk", "TASKS.md", "--no-push", cwd=self.root)
+ self.assertEqual((code, err), (0, ""), out)
+ self.assertEqual(out.count("TASKS.md"), 1, out)
+ self.assertEqual(self.out("log", "--format=%s", "master"), "bk\ninit\n")
+
+ def test_dirty_outside_paths_refused_before_done(self):
+ self.write_code()
+ (self.wt / "DESIGN.md").write_text("stray\n")
+ code, out, err = self.finish("t-three", "-m", "ok", "--commit", "impl", "code.txt", "--no-push")
+ self.assertEqual((code, err), (1, "wf: uncommitted changes outside the --commit paths: DESIGN.md\n"))
+ self.assertIn("t-three", (self.root / "TASKS.md").read_text())
+ self.assertEqual(self.out("log", "--format=%s", "master"), "init\n")
+
+ def test_gate_red_refused_before_done(self):
+ (self.root / "workflow.toml").write_text(TOML + 'quick_gate = ["exit 3"]\n')
+ git(self.root, "commit", "-qam", "gate")
+ git(self.wt, "rebase", "-q", "master")
+ code, out, err = self.finish("t-three", "-m", "ok", "--no-push")
+ self.assertEqual(code, 1, out + err)
+ self.assertIn("quick_gate 'exit 3' red", err)
+ self.assertIn("wf add -p 0", err)
+ self.assertIn("done+gate-red", err)
+ self.assertIn("t-three", (self.root / "TASKS.md").read_text())
+
+ def test_paths_need_commit_message(self):
+ code, out, err = self.finish("t-three", "-m", "ok", "code.txt")
+ self.assertEqual(code, 2, out + err)
+ self.assertIn("paths need --commit", err)
+
+ def test_main_tree_commits_code_and_bookkeeping(self):
+ (self.root / "code.txt").write_text("y\n")
+ code, out, err = self.finish("t-three", "-m", "ok", "--commit", "impl", "code.txt", cwd=self.root)
+ self.assertEqual((code, err), (0, ""), out)
+ self.assertEqual(self.out("log", "--format=%s", "master"), "impl\ninit\n")
+ self.assertEqual(self.out("show", "--stat", "--format=", "master").split("|")[0].strip(), "TASKS.md")
+ self.assertEqual(self.out("status", "--porcelain"), "")
+
+ def test_main_tree_pushes_home(self):
+ bare = self.root.parent / "home-finish.git"
+ subprocess.run(["git", "init", "-q", "--bare", str(bare)], check=True)
+ git(self.root, "remote", "add", "home", str(bare))
+ (self.root / "code.txt").write_text("y\n")
+ code, out, err = self.finish("t-three", "-m", "ok", "--commit", "impl", "code.txt", cwd=self.root)
+ self.assertEqual((code, err), (0, ""), out)
+ self.assertIn("pushed home\n", out)
+ self.assertEqual(self.out("log", "--format=%s", "master", cwd=bare), "impl\ninit\n")
+
+ def test_main_tree_no_push_flag(self):
+ bare = self.root.parent / "home-finish2.git"
+ subprocess.run(["git", "init", "-q", "--bare", str(bare)], check=True)
+ git(self.root, "remote", "add", "home", str(bare))
+ (self.root / "code.txt").write_text("y\n")
+ code, out, err = self.finish("t-three", "-m", "ok", "--commit", "impl", "code.txt", "--no-push", cwd=self.root)
+ self.assertEqual((code, err), (0, ""), out)
+ self.assertNotIn("pushed home", out)
+
+ # books stay with the task: no half-done finish, no main-tree done of a worktree's task
+
+ def test_path_outside_repo_refused_before_done(self):
+ other = self.root.parent / "other-repo-file.md"
+ other.write_text("x\n")
+ code, out, err = self.finish("t-three", "-m", "ok", "--commit", "impl", str(other), "--no-push")
+ self.assertEqual(code, 1, out + err)
+ self.assertIn("is outside this repo", err)
+ self.assertIn("t-three", (self.root / "TASKS.md").read_text())
+ self.assertEqual(self.out("status", "--porcelain"), "")
+
+ def test_missing_path_refused_before_done(self):
+ code, out, err = self.finish("t-three", "-m", "ok", "--commit", "impl", "shared/CLAUDE.md", "--no-push")
+ self.assertEqual(code, 1, out + err)
+ self.assertIn("does not exist", err)
+ self.assertIn("t-three", (self.root / "TASKS.md").read_text())
+
+ def test_unchanged_paths_refused_before_done(self):
+ code, out, err = self.finish("t-three", "-m", "ok", "--commit", "impl", "workflow.toml", "--no-push")
+ self.assertEqual(code, 1, out + err)
+ self.assertIn("nothing to commit", err)
+ self.assertIn("t-three", (self.root / "TASKS.md").read_text())
+
+ def test_worktree_books_copy_refused(self):
+ (self.wt / "TASKS.md").write_text((self.wt / "TASKS.md").read_text() + "\n")
+ code, out, err = self.finish("t-three", "-m", "ok", "--commit", "bk", "TASKS.md", "--no-push")
+ self.assertEqual(code, 1, out + err)
+ self.assertIn("copy of the books", err)
+ self.assertIn("t-three", (self.root / "TASKS.md").read_text())
+
+ def test_already_done_resumes_commit_and_merge(self):
+ self.write_code()
+ code, out, err = self.wf("done", "t-three", "-m", "ok", project=False, cwd=self.wt, env=IDENT)
+ self.assertEqual(code, 0, err)
+ code, out, err = self.finish("t-three", "-m", "ok", "--commit", "impl", "code.txt", "--no-push")
+ self.assertEqual((code, err), (0, ""), out)
+ self.assertIn("t-three already done (tasks/archive.md): resuming", out)
+ self.assertEqual(self.out("log", "--format=%s", "master"), "t-three done\nimpl\ninit\n")
+ self.assertEqual(self.out("status", "--porcelain"), "")
+ self.assertEqual((self.root / "tasks" / "archive.md").read_text().count("**t-three**"), 1)
+
+ def test_main_tree_done_of_worktree_task_refused(self):
+ self.wf("status", "t-three", "progress", "sonnet/t-three", project=False, cwd=self.wt, env=IDENT)
+ for cmd in (["done", "t-three", "-m", "ok"], ["finish", "t-three", "-m", "ok", "--no-push"]):
+ code, out, err = self.wf(*cmd, project=False, cwd=self.root, env=IDENT)
+ self.assertEqual(code, 1, out + err)
+ self.assertIn("in progress in worktree", err)
+ self.assertIn("branch sonnet/t-three", err)
+ self.assertIn("t-three", (self.root / "TASKS.md").read_text())
+ # from the worktree it goes through, books committed on master by the merge
+ self.write_code()
+ code, out, err = self.finish("t-three", "-m", "ok", "--commit", "impl", "code.txt", "--no-push")
+ self.assertEqual((code, err), (0, ""), out)
+ self.assertEqual(self.out("log", "--format=%s", "master"), "t-three done\nimpl\ninit\n")
+
+ def test_main_tree_done_after_status_clear(self):
+ self.wf("status", "t-three", "progress", "sonnet/t-three", project=False, cwd=self.wt, env=IDENT)
+ self.wf("status", "t-three", "clear", project=False, cwd=self.root, env=IDENT)
+ code, out, err = self.wf("done", "t-three", "-m", "ok", project=False, cwd=self.root, env=IDENT)
+ self.assertEqual(code, 0, out + err)
+
+ def test_main_tree_done_branch_without_worktree_ok(self):
+ self.wf("status", "t-three", "progress", "fast/t-three", project=False, cwd=self.root, env=IDENT)
+ code, out, err = self.wf("done", "t-three", "-m", "ok", project=False, cwd=self.root, env=IDENT)
+ self.assertEqual(code, 0, out + err)
+
+
+if __name__ == "__main__":
+ unittest.main()
diff --git a/tests/test_gate.py b/tests/test_gate.py
new file mode 100644
index 0000000..7a33c1f
--- /dev/null
+++ b/tests/test_gate.py
@@ -0,0 +1,74 @@
+import sys
+import tempfile
+import unittest
+from pathlib import Path
+
+sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
+from wflib import config as C
+from test_claims import git
+from test_cli import TOML
+from test_setup import GitCli
+
+GATE = TOML + 'quick_gate = ["echo \\"$WF_MAIN\\" > gate.txt", "echo second >> gate.txt"]\n'
+
+
+class GateConfigTest(unittest.TestCase):
+ def setUp(self):
+ self.tmp = tempfile.TemporaryDirectory()
+ self.root = Path(self.tmp.name).resolve()
+
+ def tearDown(self):
+ self.tmp.cleanup()
+
+ def load(self, extra):
+ (self.root / "workflow.toml").write_text('format = 1\ntasks = "T.md"\narchive = "a.md"\n' + extra)
+ return C.load(self.root)
+
+ def test_default_empty(self):
+ self.assertEqual(self.load("").quick_gate, [])
+
+ def test_list(self):
+ self.assertEqual(self.load('quick_gate = ["make quick"]\n').quick_gate, ["make quick"])
+
+ def test_wrong_type(self):
+ with self.assertRaisesRegex(C.ConfigError, "'quick_gate' must be a list of strings"):
+ self.load('quick_gate = "make quick"\n')
+
+
+class GateCliTest(GitCli):
+ toml = GATE
+
+ def setUp(self):
+ super().setUp()
+ self.wt = self.root / ".worktrees" / "opus"
+ git(self.root, "worktree", "add", "-q", str(self.wt), "-b", "opus/t-three")
+
+ def test_runs_in_worktree_with_wf_main(self):
+ code, out, err = self.wf("gate", project=False, cwd=self.wt / "docs")
+ self.assertEqual((code, err), (0, ""), out)
+ self.assertEqual((self.wt / "gate.txt").read_text(), f"{self.root}\nsecond\n")
+ self.assertFalse((self.root / "gate.txt").exists())
+ self.assertIn("$ echo second >> gate.txt\n", out)
+ self.assertTrue(out.endswith("quick_gate: 2 commands green\n"), out)
+
+ def test_runs_in_main_tree(self):
+ code, out, err = self.wf("gate", project=False, cwd=self.root)
+ self.assertEqual((code, err), (0, ""), out)
+ self.assertEqual((self.root / "gate.txt").read_text(), f"{self.root}\nsecond\n")
+
+ def test_red_exit_1_rest_skipped(self):
+ (self.wt / "workflow.toml").write_text(TOML + 'quick_gate = ["exit 4", "echo c > c.txt"]\n')
+ code, out, err = self.wf("gate", project=False, cwd=self.wt)
+ self.assertEqual(code, 1, out + err)
+ self.assertTrue(err.startswith("wf: quick_gate 'exit 4' red (exit 4): fix it before wf done; later commands skipped\n"), err)
+ self.assertIn("done+gate-red", err)
+ self.assertFalse((self.wt / "c.txt").exists())
+
+ def test_none_configured(self):
+ (self.wt / "workflow.toml").write_text(TOML)
+ code, out, err = self.wf("gate", project=False, cwd=self.wt)
+ self.assertEqual((code, out, err), (0, "no quick_gate in workflow.toml: nothing to do\n", ""))
+
+
+if __name__ == "__main__":
+ unittest.main()
diff --git a/tests/test_human_done.py b/tests/test_human_done.py
new file mode 100644
index 0000000..e6affee
--- /dev/null
+++ b/tests/test_human_done.py
@@ -0,0 +1,40 @@
+import sys
+import unittest
+from pathlib import Path
+
+sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
+from wflib import tasks as T
+
+
+def ready(done):
+ doc = T.parse(f"# T\n\n## Pending\n\n- **t-x** [P2] (<1h): X.\n Done: {done}\n")
+ return doc.sections[0].items[0].runner_ready if hasattr(doc, "sections") else None
+
+
+class HumanDoneTest(unittest.TestCase):
+ def test_owner_bound_subject(self):
+ self.assertTrue(ready("owner-bound task with Done not counted"))
+
+ def test_owner_actions(self):
+ self.assertFalse(ready("owner confirms X"))
+ self.assertFalse(ready("owner approves X"))
+ self.assertFalse(ready("report to owner"))
+ self.assertFalse(ready("confirm it works"))
+
+ def test_word_alone_is_fine(self):
+ self.assertTrue(ready("owner-facing docs updated; owner-private names scrubbed"))
+ self.assertTrue(ready("tests confirm the parser handles X"))
+ self.assertTrue(ready("the owner file is read"))
+
+ def test_more_actions(self):
+ self.assertFalse(ready("owner says it is fine"))
+ self.assertFalse(ready("Owner tests it"))
+ self.assertFalse(ready("confirm with the owner"))
+
+ def test_match_exposed(self):
+ doc = T.parse("# T\n\n## Pending\n\n- **t-x** [P2] (<1h): X.\n Done: owner approves X\n")
+ self.assertEqual(doc.sections[0].items[0].human_done_match, "owner approves")
+
+
+if __name__ == "__main__":
+ unittest.main()
diff --git a/tests/test_lanes.py b/tests/test_lanes.py
new file mode 100644
index 0000000..a9bcf93
--- /dev/null
+++ b/tests/test_lanes.py
@@ -0,0 +1,408 @@
+import json
+import os
+import sys
+import unittest
+from pathlib import Path
+
+sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
+from wflib import lanes as L
+from wflib import tasks as T
+from test_cli import Cli
+
+ITEMS = """\
+## Awaiting your decision
+
+## Pending
+
+- **t-a** [P1] (1h): A.
+
+- **t-b** [P1] (1h): B.
+ - After: [[t-s]]
+
+- **t-s** [P1] (1h): S.
+ Model: sonnet
+
+- **t-s2** [P2] (1h): S2.
+ Model: sonnet
+ - After: [[t-x]]
+
+- **t-c** [P2] (1h): C.
+ - After: [[t-s2]]
+
+## Needs human
+
+## Deferred
+"""
+SOCK = "/run/user/1000/cc-socks/1.sock"
+
+
+class LanesTest(unittest.TestCase):
+ def setUp(self):
+ self.doc = T.parse(ITEMS)
+
+ def test_counts(self):
+ self.assertEqual(L.counts(self.doc, set(), L.DEFAULT_LANES, "1h"),
+ {"fast": (0, 0, 0), "slow": (2, 3, 0)})
+
+ def test_counts_held(self):
+ self.assertEqual(L.counts(self.doc, set(), L.DEFAULT_LANES, "1h", {"t-a": "x"}),
+ {"fast": (0, 0, 0), "slow": (1, 4, 0)})
+
+ def test_counts_slice_jobs(self):
+ self.assertEqual(L.counts(T.parse(PICK), {"t-old1"}, L.DEFAULT_LANES, "1h"),
+ {"fast": (3, 2, 1), "slow": (1, 1, 0)})
+
+ def test_lanes_block(self):
+ counts = {"fast": (3, 1, 1), "slow": (0, 2, 0)}
+ sessions = {"slow": {"socket": SOCK, "alive": True, "model": "opus"}}
+ self.assertEqual(L.lanes_block(counts, sessions, "fast"), [
+ "fast (you): 3 pickable (1 slice) · 1 waiting",
+ f"slow: 0 pickable · 2 waiting · session uds:{SOCK} (alive, opus)",
+ ])
+
+ def test_lanes_block_no_session(self):
+ self.assertEqual(L.lanes_block({"fast": (2, 0, 2), "slow": (0, 0, 0)}, {}, None), [
+ "fast: 2 pickable (2 slices) · 0 waiting · no session → orchestrator or owner: start one (wf next --lane fast)",
+ "slow: 0 pickable · 0 waiting · no session",
+ ])
+
+ def test_cross_waits_by_lane(self):
+ doc = T.parse(PICK)
+ self.assertEqual([(m.id, b.id) for m, b in L.cross_waits(doc, {"t-old1"}, "slow", L.DEFAULT_LANES, "1h")],
+ [("t-w1", "t-blocker")])
+ self.assertEqual(L.cross_waits(doc, {"t-old1"}, "fast", L.DEFAULT_LANES, "1h"), [])
+
+ def test_waiting_block(self):
+ doc = T.parse(PICK)
+ waits = L.cross_waits(doc, {"t-old1"}, "slow", L.DEFAULT_LANES, "1h")
+ self.assertEqual(L.waiting_block(waits, {"fast": {"socket": SOCK, "alive": True}}, L.DEFAULT_LANES, "1h"),
+ [f'- t-w1 waits on t-blocker (fast lane) → message uds:{SOCK}: '
+ '"t-blocker blocks my t-w1, please take it"'])
+ self.assertEqual(L.waiting_block(waits, {}, L.DEFAULT_LANES, "1h"),
+ ["- t-w1 waits on t-blocker (fast lane) → no fast session: tell the owner"])
+
+ def test_notify_block_by_lane(self):
+ doc = T.parse(PICK)
+ freed = [doc.item("t-w1"), doc.item("t-w2")]
+ self.assertEqual(L.notify_block(freed, {"slow": {"socket": SOCK, "alive": True}}, L.DEFAULT_LANES, "1h"), [
+ "fast work now pickable: t-w2 (no session: tell the owner)",
+ f"notify slow uds:{SOCK}: now pickable t-w1",
+ ])
+
+ def test_solo_done_block_any_session_name(self):
+ sessions = {"fast": {"socket": SOCK, "alive": True, "pid": 1}, "all": {"socket": "/b", "alive": True, "pid": 2},
+ "slow": {"socket": "/c", "alive": False, "pid": 3}}
+ self.assertEqual(L.solo_done_block(["t-s"], sessions, "2"),
+ [f"notify fast uds:{SOCK}: solo t-s done, run wf next"])
+
+
+PICK = """\
+## Awaiting your decision
+
+## Pending
+
+- **t-big** [P2] (5h): Big unsliced.
+
+- **t-p0** [P0] (<1h): Small P0.
+
+- **t-blocker** [P3] (<1h): Small, two wait on it.
+
+- **t-w1** [P2] (1h): Waits.
+ - After: [[t-blocker]]
+
+- **t-w2** [P3] (<1h): Waits too.
+ - After: [[t-blocker]]
+ Model: sonnet
+
+- **t-mid** [P1] (1h): Mid.
+ Model: sonnet
+
+- **t-done-parent** [P1] (10h): Sliced, slices archived.
+ - Slices: [[t-old1]]
+
+## Needs human
+
+## Deferred
+"""
+
+
+class PickTest(unittest.TestCase):
+ def setUp(self):
+ self.doc = T.parse(PICK)
+
+ def pick(self, lane=None, model=None, archived=frozenset({"t-old1"})):
+ item, _ = L.pick(self.doc, set(archived), L.DEFAULT_LANES, "1h", lane, model)
+ return item.id if item else None
+
+ def test_lane_of(self):
+ got = {i.id: L.lane_of(i, L.DEFAULT_LANES, "1h") for i in self.doc.section("pending").items}
+ self.assertEqual(got, {"t-big": "fast", "t-p0": "fast", "t-blocker": "fast", "t-w1": "slow",
+ "t-w2": "fast", "t-mid": "slow", "t-done-parent": "fast"})
+
+ def test_no_effort_is_fast(self):
+ item = T.Item(id="t-x", prio=1)
+ self.assertEqual(L.lane_of(item, L.DEFAULT_LANES, "1h"), "fast")
+
+ def test_fast_unblock_order(self):
+ # unblock: waited-on first (t-blocker: 2 waiters), then the slice job t-big (0 waiters) vs t-p0 by prio
+ self.assertEqual([i.id for i in L.ranked(self.doc, {"t-old1"}, L.DEFAULT_LANES, "1h", "fast")],
+ ["t-blocker", "t-p0", "t-big"])
+
+ def test_slow_priority_order_then_fallback(self):
+ self.assertEqual(self.pick("slow"), "t-mid")
+ self.assertEqual([i.id for i in L.ranked(self.doc, {"t-old1"}, L.DEFAULT_LANES, "1h", "slow")],
+ ["t-mid", "t-blocker", "t-p0", "t-big"])
+
+ def test_fallback_when_lane_empty(self):
+ doc = T.parse(PICK.replace("(1h): Mid.", "(<1h): Mid.")) # slow has only t-w1 (blocked)
+ item, _ = L.pick(doc, {"t-old1"}, L.DEFAULT_LANES, "1h", "slow")
+ self.assertEqual(item.id, "t-blocker")
+
+ def test_all_lanes_priority_order(self):
+ self.assertEqual(self.pick(None), "t-p0")
+
+ def test_model_ceiling(self):
+ # haiku session: no task has Model haiku; opus (default) and sonnet are above it
+ self.assertIsNone(self.pick("fast", "haiku"))
+ self.assertEqual(self.pick("slow", "sonnet"), "t-mid")
+ self.assertEqual(self.pick("fast", "sonnet"), None) # t-w2 is blocked; others are opus
+ self.assertEqual(self.pick("fast", "opus"), "t-blocker")
+
+ def test_slice_job(self):
+ big = self.doc.item("t-big")
+ self.assertTrue(L.is_slice_job(big, "1h"))
+ self.assertFalse(L.is_slice_job(self.doc.item("t-p0"), "1h"))
+
+ def test_sliced_parent_not_slice_job(self):
+ parent = self.doc.item("t-done-parent")
+ self.assertFalse(L.is_slice_job(parent, "1h"))
+ self.assertTrue(L.sliced_out(parent, "1h"))
+ # skipped lists items ranked before the pick: drop the others so the parent is first
+ doc = T.parse(PICK[:PICK.index("- **t-big**")] + PICK[PICK.index("- **t-done-parent**"):])
+ _, skipped = L.pick(doc, {"t-old1"}, L.DEFAULT_LANES, "1h", "fast")
+ self.assertIn(("t-done-parent", "sliced: all slices done → wf done t-done-parent or add slices"),
+ [(i.id, why) for i, why in skipped])
+
+
+PICK_DONE = PICK.replace("Small P0.\n", "Small P0.\n Done: x\n").replace(
+ "two wait on it.\n", "two wait on it.\n Done: x\n").replace("Big unsliced.\n", "Big unsliced.\n Done: x\n")
+
+
+class LanesCliTest(Cli):
+ tasks_text = "# Tasks — demo\n\n" + PICK_DONE
+
+ def env(self, name="me"):
+ sock = self.root / f"{name}.sock"
+ sock.write_text("")
+ return {"CLAUDE_CODE_MESSAGING_SOCKET": str(sock), "CLAUDE_PID": str(os.getpid()), "CLAUDE_CODE_SESSION_ID": name}
+
+ def session(self, name):
+ return json.loads((self.root / ".wf" / "sessions" / f"{name}.json").read_text())
+
+ def test_next_lane_registers_lane(self):
+ out = self.ok("next", "--lane", "slow", "--as", "opus", env=self.env())
+ self.assertIn("===== Next task =====\n- **t-mid**", out)
+ s = self.session("slow")
+ self.assertEqual((s["lane"], s["model"], s["socket"]), ("slow", "opus", str(self.root / "me.sock")))
+ self.assertIn("===== Lanes =====\nfast: 3 pickable (1 slice) · 2 waiting · no session → "
+ "orchestrator or owner: start one (wf next --lane fast)\n"
+ "slow (you): 1 pickable · 1 waiting\n\n"
+ "===== Waiting on other lanes =====\n"
+ "- t-w1 waits on t-blocker (fast lane) → no fast session: tell the owner\n", out)
+
+ def test_next_without_lane_registers_all(self):
+ out = self.ok("next", "--as", "opus", env=self.env())
+ self.assertIn("===== Next task =====\n- **t-p0**", out)
+ self.assertEqual(self.session("all")["lane"], "all")
+ self.assertNotIn("Waiting on other lanes", out)
+
+ def test_slice_job_output(self):
+ (self.root / "TASKS.md").write_text("# Tasks — demo\n\n## Awaiting your decision\n\n## Pending\n\n"
+ "- **t-big** [P2] (5h): Big unsliced.\n\n## Needs human\n\n## Deferred\n")
+ out = self.ok("next", "--lane", "fast", "--as", "opus", "--brief")
+ self.assertEqual(out, "- **t-big** [P2] (5h): Big unsliced.\n\n"
+ "===== Slice job (no code) =====\n"
+ "effort 5h > slice_above 1h: split it, don't implement.\n"
+ ' wf add "<title>. <goal>" -e <1h|1h> --parent t-big --model haiku|sonnet|opus'
+ " (then wf body: Steps/Done/Ref)\n"
+ ' wf note t-big "sliced into …" · wf status t-big clear · not wf done'
+ " (the parent waits on its slices)\n")
+
+ def test_unknown_lane(self):
+ self.assertEqual(self.fails("next", "--lane", "nope", code=2), "wf: unknown lane 'nope' (fast, slow)\n")
+ self.assertEqual(self.fails("list", "--lane", "nope", code=2), "wf: unknown lane 'nope' (fast, slow)\n")
+
+ def test_nothing_pickable_text(self):
+ self.assertEqual(self.fails("next", "--lane", "fast", "--as", "haiku", "--brief"),
+ "wf: nothing pickable for fast (haiku) in Pending\n")
+ self.assertEqual(self.fails("next", "--as", "haiku", "--brief").splitlines()[-1],
+ "wf: nothing pickable for all lanes (haiku) in Pending")
+
+ def test_list_runner_lane_pick_order(self):
+ out = self.ok("list", "--runner", "--lane", "fast")
+ self.assertEqual([l.split()[0] for l in out.splitlines()[:-1]], ["t-blocker", "t-p0", "t-big"])
+
+ def test_list_lane_filter_and_column(self):
+ out = self.ok("list", "--lane", "slow")
+ self.assertEqual(out.splitlines()[:2], ["t-w1 P2 1h - opus slow Waits",
+ "t-mid P1 1h - sonnet slow Mid"])
+
+ def test_lanes_command(self):
+ out = self.ok("lanes", "--lane", "fast", "--as", "sonnet", env=self.env())
+ self.assertEqual(out, "fast (you): 3 pickable (1 slice) · 2 waiting\n"
+ "slow: 0 pickable · 1 waiting · 1 not runner-ready (no Done) · no session\n")
+ self.assertEqual((self.session("fast")["lane"], self.session("fast")["model"]), ("fast", "sonnet"))
+ self.ok("lanes", "--unregister", env=self.env())
+ self.assertFalse((self.root / ".wf" / "sessions" / "fast.json").exists())
+
+ def test_done_notifies_by_lane(self):
+ sock = self.root / "s.sock"
+ sock.write_text("")
+ d = self.root / ".wf" / "sessions"
+ d.mkdir(parents=True)
+ (d / "slow.json").write_text(json.dumps({"lane": "slow", "model": "opus", "socket": str(sock),
+ "pid": os.getpid()}))
+ out = self.ok("done", "t-blocker", "-m", "ok", env={"CLAUDE_CODE_MESSAGING_SOCKET": ""})
+ self.assertIn(f"notify slow uds:{sock}: now pickable t-w1\n", out)
+ self.assertNotIn("t-w2", out) # same lane as t-blocker: the doer sees it
+
+
+
+
+WAIT_NONE = """\
+## Awaiting your decision
+
+## Pending
+
+- **t-x** [P1] (1h): X.
+ - After: [[t-nope]]
+
+## Needs human
+
+## Deferred
+"""
+
+
+class LanesWaitTest(Cli):
+ tasks_text = WAIT_NONE
+ env = {"WF_LANES_POLL": "0.1"}
+
+ def test_pickable_exits_0_at_once(self):
+ (self.root / "TASKS.md").write_text(DONE_ITEMS)
+ out = self.ok("lanes", "--wait", "5", env=self.env)
+ self.assertIn("slow: 2 pickable", out)
+
+ def test_timeout_exits_1(self):
+ self.assertEqual(self.fails("lanes", "--wait", "1", env=self.env), "")
+
+ def test_added_mid_wait(self):
+ import threading
+ import time
+
+ def add():
+ time.sleep(0.6)
+ (self.root / "TASKS.md").write_text(DONE_ITEMS)
+ th = threading.Thread(target=add)
+ th.start()
+ out = self.ok("lanes", "--wait", "10", env=self.env)
+ th.join()
+ self.assertIn("slow: 2 pickable", out)
+
+ def test_stop_file_exits_2_at_once(self):
+ (self.root / "out").mkdir()
+ (self.root / "out" / "wf-batch.stop").touch()
+ code, out, err = self.wf("lanes", "--wait", "30", env=self.env)
+ self.assertEqual((code, err), (2, ""))
+ self.assertIn("stop requested: out/wf-batch.stop", out)
+
+ def test_no_done_not_pickable(self):
+ (self.root / "TASKS.md").write_text(NO_DONE)
+ code, out, err = self.wf("lanes", "--wait", "1", env=self.env)
+ self.assertEqual((code, err), (1, ""))
+ self.assertEqual(out,
+ "fast: 0 pickable · 0 waiting · 2 not runner-ready (no Done) · no session\n"
+ "slow: 0 pickable · 0 waiting · no session\n")
+
+ def test_lanes_plain_shows_not_ready(self):
+ (self.root / "TASKS.md").write_text(NO_DONE)
+ self.assertIn("fast: 0 pickable · 0 waiting · 2 not runner-ready (no Done)", self.ok("lanes"))
+
+
+NO_DONE = """\
+## Awaiting your decision
+
+## Pending
+
+- **t-a** [P1] (<1h): A.
+
+- **t-b** [P1] (<1h): B.
+
+## Needs human
+
+## Deferred
+"""
+
+DONE_ITEMS = ITEMS.replace("Model: sonnet\n", "Model: sonnet\n Done: ok.\n").replace("(1h): A.", "(1h): A.\n Done: ok.") \
+ .replace("(1h): B.", "(1h): B.\n Done: ok.")
+
+
+class LibRunnerTest(unittest.TestCase):
+ def test_counts_runner_and_not_ready(self):
+ doc = T.parse(NO_DONE.replace("A.", "A.\n Done: ok."))
+ a = (doc, set(), L.DEFAULT_LANES, "1h")
+ self.assertEqual(L.counts(*a), {"fast": (2, 0, 0), "slow": (0, 0, 0)})
+ self.assertEqual(L.counts(*a, runner=True), {"fast": (1, 0, 0), "slow": (0, 0, 0)})
+ self.assertEqual(L.not_ready(*a), {"fast": 1, "slow": 0})
+
+ def test_not_ready_skips_owner_bound_with_done(self):
+ doc = T.parse(NO_DONE.replace("A.", "A.\n Done: ok.\n Sessions: owner"))
+ a = (doc, set(), L.DEFAULT_LANES, "1h")
+ self.assertEqual(L.not_ready(*a), {"fast": 1, "slow": 0})
+ self.assertEqual(len(L.prep_targets(*a)), 1)
+
+
+PREP = """\
+## Pending
+
+- **t-ok** [P1] (<1h): Has Done.
+ Done: x works.
+
+- **t-low** [P3] (<1h): Low, no Done.
+
+- **t-wait** [P0] (1h): Waits, no Done.
+ - After: [[t-low]]
+
+- **t-hi** [P1] (1h): High, no Done.
+
+- **t-own** [P0] (<1h): Owner, no Done.
+ Sessions: owner
+
+- **t-prog** [P0] (<1h) (in progress: w): Running.
+
+- **t-blk** [P0] (<1h) (blocked: a-q): Blocked.
+
+- **t-par** [P0] (5h): Parent.
+
+- **t-par-1** [P2] (<1h): Slice.
+ Done: y.
+
+## Needs human
+
+- **t-h** [P0] (<1h): Human.
+"""
+
+
+class PrepTargetsTest(unittest.TestCase):
+ def ids(self, **kw):
+ return [i.id for i in L.prep_targets(T.parse(PREP), set(), L.DEFAULT_LANES, "1h", **kw)]
+
+ def test_pickable_first_then_prio(self):
+ self.assertEqual(self.ids(), ["t-hi", "t-low", "t-wait"])
+
+ def test_lane_filter(self):
+ self.assertEqual(self.ids(only={"fast"}), ["t-low"])
+ self.assertEqual(self.ids(only={"slow"}), ["t-hi", "t-wait"])
+
+
+if __name__ == "__main__":
+ unittest.main()
diff --git a/tests/test_ledgers.py b/tests/test_ledgers.py
new file mode 100644
index 0000000..1a1add1
--- /dev/null
+++ b/tests/test_ledgers.py
@@ -0,0 +1,26 @@
+import sys
+import unittest
+from pathlib import Path
+
+sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
+from wflib import ledgers as L
+
+PLAN = "# P\n**Goal:** g\n### Task 1: A\nx\n### Task 2: B\n### Task 3: C\n"
+
+
+class SummaryTest(unittest.TestCase):
+ def test_resumes_at_first_task_without_complete_line(self):
+ ledger = "# SDD ledger — plan: p.md\nTask 1: complete (x)\nTask 2: Ruling: y\nTask 2: complete (z)\n"
+ self.assertEqual(L.summary(PLAN, ledger), "done 1, 2; resume at Task 3 (task-start PLAN 3)")
+
+ def test_ruling_line_is_not_completion(self):
+ self.assertEqual(L.summary(PLAN, "Task 1: Ruling: complete rewrite\n"),
+ "done none; resume at Task 1 (task-start PLAN 1)")
+
+ def test_all_done_reports_last_final_review_line(self):
+ ledger = "Task 1: complete\nTask 2: complete\nTask 3: complete\nFinal review: running\nFinal review: clean\n"
+ self.assertEqual(L.summary(PLAN, ledger), "done 1, 2, 3; all tasks done; Final review: clean")
+
+
+if __name__ == "__main__":
+ unittest.main()
diff --git a/tests/test_lock.py b/tests/test_lock.py
new file mode 100644
index 0000000..15d5e88
--- /dev/null
+++ b/tests/test_lock.py
@@ -0,0 +1,30 @@
+import fcntl
+import subprocess
+import sys
+import time
+import unittest
+
+from test_cli import Cli, WF
+
+
+class LockTest(Cli):
+ def test_write_waits_for_lock(self):
+ lock_path = self.root / ".wf" / "lock"
+ lock_path.parent.mkdir(exist_ok=True)
+ with open(lock_path, "w") as held:
+ fcntl.flock(held, fcntl.LOCK_EX)
+ p = subprocess.Popen([sys.executable, str(WF), "--project", str(self.root), "note", "t-three", "first"],
+ stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True)
+ time.sleep(0.5)
+ self.assertIsNone(p.poll(), "write ran while the lock was held")
+ self.ok("show", "t-three") # read commands never wait
+ out, err = p.communicate(timeout=10)
+ self.assertEqual((p.returncode, err), (0, ""))
+ self.ok("note", "t-three", "second")
+ body = self.item("t-three")
+ self.assertIn("first", body)
+ self.assertIn("second", body)
+
+ def test_lock_file_is_git_ignored_folder(self):
+ self.ok("note", "t-three", "x")
+ self.assertEqual((self.root / ".wf" / ".gitignore").read_text(), "*\n")
diff --git a/tests/test_merge.py b/tests/test_merge.py
new file mode 100644
index 0000000..cb14a20
--- /dev/null
+++ b/tests/test_merge.py
@@ -0,0 +1,159 @@
+import fcntl
+import subprocess
+import sys
+import time
+import unittest
+
+from test_cli import Cli, WF
+from test_claims import git
+
+IDENT = {"GIT_AUTHOR_NAME": "t", "GIT_AUTHOR_EMAIL": "t@t", "GIT_COMMITTER_NAME": "t", "GIT_COMMITTER_EMAIL": "t@t",
+ "CLAUDE_CODE_MESSAGING_SOCKET": ""}
+
+
+class MergeTest(Cli):
+ def setUp(self):
+ super().setUp()
+ git(self.root, "init", "-q", "-b", "master")
+ (self.root / ".gitignore").write_text(".worktrees/\n.wf/\n")
+ git(self.root, "add", "-A")
+ git(self.root, "commit", "-qm", "init")
+ self.wt = self.root / ".worktrees" / "sonnet"
+ git(self.root, "worktree", "add", "-q", str(self.wt), "-b", "sonnet/t-three")
+
+ def merge(self, *args):
+ return self.wf("merge", *args, project=False, cwd=self.wt, env=IDENT)
+
+ def out(self, *args, cwd=None):
+ return subprocess.run(["git", *args], cwd=cwd or self.root, capture_output=True, text=True).stdout
+
+ def commit_code(self):
+ (self.wt / "code.txt").write_text("x\n")
+ git(self.wt, "add", "code.txt")
+ git(self.wt, "commit", "-qm", "code")
+
+ def test_merge_ff_commits_bookkeeping_and_detaches(self):
+ self.commit_code()
+ self.wf("done", "t-three", "-m", "ok", project=False, cwd=self.wt, env=IDENT)
+ code, out, err = self.merge("--no-push")
+ self.assertEqual((code, err), (0, ""), out)
+ self.assertEqual(out, "rebased onto master\nfast-forwarded master\n"
+ "committed TASKS.md tasks/archive.md\nmerged sonnet/t-three into master\n")
+ self.assertEqual(self.out("log", "--format=%s", "master"), "t-three done\ncode\ninit\n")
+ self.assertEqual(self.out("branch", "--list", "sonnet/t-three"), "")
+ self.assertEqual(self.out("status", "--porcelain"), "")
+ self.assertEqual((self.root / "code.txt").read_text(), "x\n")
+
+ def test_merge_commits_notes_and_followups_after_done(self):
+ self.commit_code()
+ self.wf("done", "t-three", "-m", "ok", project=False, cwd=self.wt, env=IDENT)
+ r = self.wf("add", "-p", "2", "-e", "1h", "follow up", project=False, cwd=self.wt, env=IDENT); self.assertEqual(r[0], 0, r)
+ code, out, err = self.merge("--no-push")
+ self.assertEqual((code, err), (0, ""), out)
+ self.assertEqual(self.out("status", "--porcelain"), "")
+ self.assertIn("follow up", self.out("show", "master:TASKS.md"))
+
+ def test_merge_rebases_onto_moved_master(self):
+ self.commit_code()
+ (self.root / "other.txt").write_text("o\n")
+ git(self.root, "add", "other.txt")
+ git(self.root, "commit", "-qm", "other")
+ code, out, err = self.merge("--no-push")
+ self.assertEqual((code, err), (0, ""), out)
+ self.assertEqual(self.out("log", "--format=%s", "master"), "code\nother\ninit\n")
+
+ def test_merge_conflict_aborts_and_keeps_branch(self):
+ (self.root / "DESIGN.md").write_text("main side\n")
+ git(self.root, "commit", "-qam", "main edit")
+ (self.wt / "DESIGN.md").write_text("branch side\n")
+ git(self.wt, "commit", "-qam", "branch edit")
+ code, out, err = self.merge("--no-push")
+ self.assertEqual((code, err), (1, "wf: rebase onto master conflicts: git rebase master, resolve, verify, "
+ "then wf merge again\n"))
+ self.assertEqual((self.wt / "DESIGN.md").read_text(), "branch side\n")
+ self.assertEqual(self.out("status", "--porcelain", cwd=self.wt), "")
+ self.assertEqual(self.out("branch", "--show-current", cwd=self.wt), "sonnet/t-three\n")
+
+ def test_merge_refuses_dirty_worktree(self):
+ (self.wt / "DESIGN.md").write_text("dirty\n")
+ code, out, err = self.merge("--no-push")
+ self.assertEqual((code, err), (1, "wf: worktree has uncommitted changes: commit them first\n"))
+
+ def test_merge_outside_worktree_fails(self):
+ self.assertEqual(self.fails("merge"), "wf: merge runs inside a linked git worktree (lane worktree)\n")
+
+ def test_merge_detached_fails(self):
+ git(self.wt, "switch", "-q", "--detach", "master")
+ code, out, err = self.merge("--no-push")
+ self.assertEqual((code, err), (1, "wf: worktree is on a detached HEAD: nothing to merge\n"))
+
+ def test_merge_detached_commits_leftover_bookkeeping(self):
+ self.commit_code()
+ self.wf("done", "t-three", "-m", "ok", project=False, cwd=self.wt, env=IDENT)
+ self.merge("--no-push")
+ self.wf("add", "-p", "2", "-e", "1h", "late follow up", project=False, cwd=self.wt, env=IDENT)
+ code, out, err = self.merge("--no-push")
+ self.assertEqual((code, err, out), (0, "", "committed TASKS.md tasks/archive.md\n"))
+ self.assertEqual(self.out("status", "--porcelain"), "")
+ self.assertIn("late follow up", self.out("show", "master:TASKS.md"))
+
+ def test_merge_detached_with_impl_commit_merges_it(self):
+ git(self.wt, "switch", "-q", "--detach", "master")
+ (self.root / "other.txt").write_text("o\n")
+ git(self.root, "add", "other.txt")
+ git(self.root, "commit", "-qm", "other")
+ self.commit_code()
+ r = self.wf("done", "t-three", "-m", "ok", project=False, cwd=self.wt, env=IDENT)
+ self.assertIn("detached HEAD has 1 commit not in master: wf merge merges it", r[1])
+ code, out, err = self.merge("--no-push")
+ self.assertEqual((code, err), (0, ""), out)
+ self.assertEqual(out, "rebased onto master\nfast-forwarded master\n"
+ "committed TASKS.md tasks/archive.md\nmerged detached HEAD into master\n")
+ self.assertEqual(self.out("log", "--format=%s", "master"), "bookkeeping\ncode\nother\ninit\n")
+ self.assertEqual((self.root / "code.txt").read_text(), "x\n")
+ self.assertEqual(self.out("rev-parse", "HEAD", cwd=self.wt), self.out("rev-parse", "master"))
+ self.assertEqual(self.out("status", "--porcelain"), "")
+
+ def test_merge_detached_refuses_dirty_worktree(self):
+ git(self.wt, "switch", "-q", "--detach", "master")
+ (self.wt / "code.txt").write_text("x\n")
+ code, out, err = self.merge("--no-push")
+ self.assertEqual((code, err), (1, "wf: worktree has uncommitted changes: commit them first\n"))
+
+ def test_merge_detached_conflict_aborts_and_keeps_commit(self):
+ git(self.wt, "switch", "-q", "--detach", "master")
+ (self.root / "DESIGN.md").write_text("main side\n")
+ git(self.root, "commit", "-qam", "main edit")
+ (self.wt / "DESIGN.md").write_text("branch side\n")
+ git(self.wt, "commit", "-qam", "branch edit")
+ code, out, err = self.merge("--no-push")
+ self.assertEqual((code, err), (1, "wf: rebase onto master conflicts: git rebase master, resolve, verify, "
+ "then wf merge again\n"))
+ self.assertEqual(self.out("log", "-1", "--format=%s", cwd=self.wt), "branch edit\n")
+
+ def test_merge_pushes_to_home(self):
+ bare = self.root.parent / "home.git"
+ subprocess.run(["git", "init", "-q", "--bare", str(bare)], check=True)
+ git(self.root, "remote", "add", "home", str(bare))
+ self.commit_code()
+ code, out, err = self.merge()
+ self.assertEqual((code, err), (0, ""), out)
+ self.assertIn("pushed home\n", out)
+ self.assertEqual(self.out("log", "--format=%s", "master", cwd=bare), "code\ninit\n")
+
+ def test_merge_waits_for_lock(self):
+ self.commit_code()
+ (self.root / ".wf").mkdir(exist_ok=True)
+ with open(self.root / ".wf" / "lock", "w") as held:
+ fcntl.flock(held, fcntl.LOCK_EX)
+ import os
+ p = subprocess.Popen([sys.executable, str(WF), "merge", "--no-push"], cwd=self.wt, text=True,
+ stdout=subprocess.PIPE, stderr=subprocess.PIPE, env={**os.environ, **IDENT})
+ time.sleep(0.5)
+ self.assertIsNone(p.poll(), "merge ran while the lock was held")
+ out, err = p.communicate(timeout=20)
+ self.assertEqual((p.returncode, err), (0, ""))
+
+
+if __name__ == "__main__":
+ unittest.main()
diff --git a/tests/test_migrate.py b/tests/test_migrate.py
new file mode 100644
index 0000000..bd02f92
--- /dev/null
+++ b/tests/test_migrate.py
@@ -0,0 +1,160 @@
+import sys
+import unittest
+from pathlib import Path
+
+sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
+from wflib import migrate as M
+from wflib import tasks as T
+
+OLD = """\
+# Tasks — demo
+
+Conventions: CLAUDE.md.
+
+## Awaiting your decision
+
+- **Smoke screenshots (need you present)**: windows open hidden,
+ so screenshots are black.
+- Online mod updates (spec step 4) not built.
+
+## Pending
+
+1. **[P1] Terrain: cave seams** (Effort: <1h) — close the slit beside the lintel.
+ - Tests (red first): no edge left open.
+ - After: Terrain 6/6: cave rendering + sound
+ Reference: [DESIGN.md#terrain](DESIGN.md#terrain)
+
+2. N. **[P2] Reputation factions** (Effort: 5h, interactive) (in progress: master; slices 1–2/3 done (see plan)) — make factions friendly.
+ - Steps: use the table
+ - nested
+ - After: Terrain: cave seams
+ Reference: docs/notes/exe-map.md (ai.c rows), [spec](docs/spec.md#rep).
+
+3. **[P2] Spirit Powers 3/5: technique keys** (Effort: 1h) — keys 1/2/3.
+
+4. **[P3] Spirit Powers 4/5: passives. And more** (Effort: 1h)
+ - After: Spirit Powers 3/5: technique keys
+
+## Needs human
+
+Only what a script can't judge.
+
+1. **[P1] Play-test feel** (Effort: <1h) — last pass 2026-09-13.
+ - [ ] keyboard works
+ - [ ] motion smooth
+
+## Deferred (explicitly out of scope, tracked for later)
+
+**Final art**: later.
+"""
+
+NEW = """\
+# Tasks — demo
+
+Conventions: CLAUDE.md.
+
+## Awaiting your decision
+
+- **a-smoke-screenshots**: Smoke screenshots (need you present). windows open hidden,
+ so screenshots are black.
+
+- **a-online-mod-updates-not-built**: Online mod updates (spec step 4) not built.
+
+## Pending
+
+- **t-terrain-cave-seams** [P1] (<1h): Terrain: cave seams. close the slit beside the lintel.
+ - Tests (red first): no edge left open.
+ Ref: DESIGN.md#terrain
+
+- **t-reputation-factions** [P2] (5h, interactive) (in progress: master; slices 1–2/3 done [see plan]): Reputation factions. make factions friendly.
+ - Steps: use the table
+ - nested
+ - After: [[t-terrain-cave-seams]]
+ Ref: docs/notes/exe-map.md (ai.c rows), docs/spec.md#rep
+
+- **t-spirit-powers-3** [P2] (1h): Spirit Powers 3/5: technique keys. keys 1/2/3.
+
+- **t-spirit-powers-4** [P3] (1h): Spirit Powers 4/5: passives, And more.
+ - After: [[t-spirit-powers-3]]
+
+## Needs human
+
+Only what a script can't judge.
+
+- **t-play-test-feel** [P1] (<1h): Play-test feel. last pass 2026-09-13.
+ - [ ] keyboard works
+ - [ ] motion smooth
+
+## Deferred
+
+(explicitly out of scope, tracked for later)
+
+**Final art**: later.
+"""
+
+
+class MigrateTest(unittest.TestCase):
+ def test_converts_all_oddities(self):
+ new, report = M.migrate(OLD)
+ self.assertEqual(new, NEW)
+ self.assertEqual(report.ids, {
+ "Smoke screenshots (need you present)": "a-smoke-screenshots",
+ "Online mod updates (spec step 4) not built": "a-online-mod-updates-not-built",
+ "Terrain: cave seams": "t-terrain-cave-seams",
+ "Reputation factions": "t-reputation-factions",
+ "Spirit Powers 3/5: technique keys": "t-spirit-powers-3",
+ "Spirit Powers 4/5: passives. And more": "t-spirit-powers-4",
+ "Play-test feel": "t-play-test-feel",
+ })
+ self.assertEqual(report.notes, [
+ "t-terrain-cave-seams: dropped 'After: Terrain 6/6: cave rendering + sound' (no open task with that title: done)"])
+
+ def test_result_parses_clean(self):
+ doc = T.parse(M.migrate(OLD)[0])
+ self.assertEqual([i.error for i in doc.all_items()], [None] * 7)
+ self.assertEqual(T.render(doc), NEW)
+
+ def test_idempotent(self):
+ new, report = M.migrate(NEW)
+ self.assertEqual(new, NEW)
+ self.assertEqual((report.ids, report.notes), ({}, []))
+
+ def test_missing_sections_are_added(self):
+ new, _ = M.migrate("# T\n\n## Pending\n\n1. **[P1] A** (Effort: 1h) — a.\n\n## Notes\n\nprose\n")
+ self.assertEqual(new, "# T\n\n## Pending\n\n- **t-a** [P1] (1h): A. a.\n\n## Notes\n\nprose\n\n"
+ "## Awaiting your decision\n\n## Needs human\n\n## Deferred\n")
+
+ def test_same_title_twice_gets_distinct_ids(self):
+ new, _ = M.migrate("## Pending\n\n1. **[P1] A** (Effort: 1h) — x.\n\n2. **[P1] A** (Effort: 1h) — y.\n")
+ self.assertIn("- **t-a** [P1] (1h): A. x.", new)
+ self.assertIn("- **t-a-2** [P1] (1h): A. y.", new)
+
+ def test_ids_avoid_the_archive(self):
+ new, _ = M.migrate("## Pending\n\n1. **[P1] A** (Effort: 1h) — x.\n", taken={"t-a"})
+ self.assertIn("- **t-a-2** [P1] (1h): A. x.", new)
+
+ def test_one_column_deeper_under_a_colon_line_is_a_child(self):
+ new, _ = M.migrate("## Pending\n\n1. **[P1] A** (Effort: 1h) — x.\n - Seed Qs:\n - first?\n - second?\n"
+ " - [ ] box\n - [ ] sibling typo\n")
+ self.assertIn("- **t-a** [P1] (1h): A. x.\n - Seed Qs:\n - first?\n - second?\n"
+ " - [ ] box\n - [ ] sibling typo\n", new)
+
+ def test_reference_words_that_are_no_path_become_a_note(self):
+ new, _ = M.migrate("## Pending\n\n1. **[P1] A** (Effort: 1h) — x.\n"
+ " Reference: docs/a.md (fate rows), skill row notes, src/X.cs, DESIGN.md#terrain\n\n"
+ "2. **[P1] B** (Effort: 1h) — y.\n Reference: main spec §5\n")
+ self.assertIn(" Ref: docs/a.md (fate rows; skill row notes), src/X.cs, DESIGN.md#terrain\n", new)
+ self.assertIn("- **t-b** [P1] (1h): B. y.\n - Reference: main spec §5\n", new)
+
+ def test_reference_note_keeps_the_body_indent(self):
+ new, _ = M.migrate("## Pending\n\n1. **[P1] A** (Effort: 1h) — x.\n - Steps: s\n Reference: main spec §5\n")
+ self.assertIn("- **t-a** [P1] (1h): A. x.\n - Steps: s\n - Reference: main spec §5\n", new)
+
+ def test_unreadable_numbered_item_is_reported_and_kept(self):
+ new, report = M.migrate("## Pending\n\n1. something without the pattern\n body\n")
+ self.assertIn("1. something without the pattern\n body\n", new)
+ self.assertEqual(report.notes, ["line 3: numbered item not understood, left as it is: 1. something without the pattern"])
+
+
+if __name__ == "__main__":
+ unittest.main()
diff --git a/tests/test_model.py b/tests/test_model.py
new file mode 100644
index 0000000..b2e0878
--- /dev/null
+++ b/tests/test_model.py
@@ -0,0 +1,247 @@
+import json
+import os
+import subprocess
+import sys
+import unittest
+from pathlib import Path
+
+sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
+from wflib import tasks as T
+from wflib import lanes as L
+from test_cli import Cli
+from test_check import Base, CLEAN
+from wflib import check as K
+from wflib import config as C
+
+LANES = """\
+# Tasks — demo
+
+## Awaiting your decision
+
+## Pending
+
+- **t-one** [P1] (1h): One.
+ - Steps: a
+ Model: sonnet
+ - After: [[t-done]]
+
+- **t-two** [P2] (1h): Two.
+ - Model: haiku
+
+- **t-three** [P2] (<1h): Three.
+
+## Needs human
+
+## Deferred
+"""
+
+
+class ModelLineTest(unittest.TestCase):
+ def setUp(self):
+ self.doc = T.parse(LANES)
+
+ def test_parsed_from_body_default_opus(self):
+ self.assertEqual([i.model for i in self.doc.section("pending").items], ["sonnet", "haiku", "opus"])
+
+ def test_extra_words_ignored(self):
+ item = T.parse_block("- **t-x** [P1] (1h): X.\n - Model: sonnet ok\n")
+ self.assertEqual(item.model, "sonnet")
+
+ def test_set_replaces_in_place(self):
+ T.set_fields(self.doc, "t-one", model="opus")
+ self.assertEqual(self.doc.item("t-one").body, [" - Steps: a", " Model: opus", " - After: [[t-done]]"])
+
+ def test_set_adds_before_after_and_ref(self):
+ doc = T.parse(CLEAN)
+ T.set_fields(doc, "t-one", model="haiku")
+ self.assertEqual(doc.item("t-one").body,
+ [" Model: haiku", " - After: [[t-done]]", " Ref: DESIGN.md#terrain, docs/specs/terrain.md"])
+
+ def test_set_empty_removes(self):
+ T.set_fields(self.doc, "t-two", model="")
+ self.assertEqual(self.doc.item("t-two").body, [])
+ self.assertEqual(self.doc.item("t-two").model, "opus")
+
+ def test_set_bad_value(self):
+ with self.assertRaisesRegex(T.TaskError, r"model 'gpt' \(want haiku, sonnet, opus\)"):
+ T.set_fields(self.doc, "t-one", model="gpt")
+
+ def test_note_goes_before_model_line(self):
+ T.add_note(self.doc, "t-two", "why")
+ self.assertEqual(self.doc.item("t-two").body, [" - why", " - Model: haiku"])
+
+
+def doc(items):
+ return T.parse("## Awaiting your decision\n\n## Pending\n\n" + items + "\n## Needs human\n\n## Deferred\n")
+
+
+class PickOrderTest(unittest.TestCase):
+ def pick(self, items, model="opus", archived=()):
+ item, _ = L.pick(doc(items), set(archived), L.DEFAULT_LANES, "1h", None, model)
+ return item.id if item else None
+
+ def test_priority_inherited_from_waiting_task(self):
+ self.assertEqual(self.pick("- **t-c** [P1] (1h): C.\n\n- **t-b** [P3] (1h): B.\n\n"
+ "- **t-a** [P0] (1h): A.\n - After: [[t-b]]\n"), "t-b")
+
+ def test_inherited_transitively(self):
+ self.assertEqual(self.pick("- **t-c** [P1] (1h): C.\n\n- **t-x** [P3] (1h): X.\n\n"
+ "- **t-b** [P3] (1h): B.\n - After: [[t-x]]\n\n"
+ "- **t-a** [P0] (1h): A.\n - After: [[t-b]]\n"), "t-x")
+
+ def test_unblocking_only_low_work_does_not_jump(self):
+ self.assertEqual(self.pick("- **t-c** [P1] (1h): C.\n\n- **t-b** [P3] (1h): B.\n\n"
+ "- **t-a** [P3] (1h): A.\n - After: [[t-b]]\n"), "t-c")
+
+ def test_other_lane_waiting_first(self):
+ self.assertEqual(self.pick("- **t-f** [P2] (1h): F.\n\n- **t-k** [P2] (1h): K.\n\n"
+ "- **t-l** [P2] (1h): L.\n - After: [[t-k]]\n\n- **t-g** [P2] (1h): G.\n\n"
+ "- **t-h** [P2] (<1h): H.\n - After: [[t-g]]\n"), "t-g")
+
+ def test_more_waiting_first(self):
+ self.assertEqual(self.pick("- **t-f** [P2] (1h): F.\n - After: [[t-z]]\n\n- **t-g** [P2] (1h): G.\n\n"
+ "- **t-i** [P2] (1h): I.\n - After: [[t-g]]\n\n"
+ "- **t-j** [P2] (1h): J.\n - After: [[t-i]]\n\n"
+ "- **t-z** [P2] (1h): Z.\n", archived=()), "t-g")
+
+ def test_parent_waits_on_its_slices(self):
+ self.assertEqual(self.pick("- **t-q** [P1] (1h): Q.\n\n- **t-p** [P0] (5h): P.\n - Slices: [[t-p-1]]\n\n"
+ "- **t-p-1** [P2] (1h): P1.\n"), "t-p-1")
+
+ def test_lane_filter(self):
+ items = ("- **t-a** [P0] (1h): A.\n\n- **t-b** [P1] (1h): B.\n Model: sonnet\n\n"
+ "- **t-c** [P2] (1h): C.\n Model: haiku\n")
+ self.assertEqual([self.pick(items, m) for m in ("opus", "sonnet", "haiku", None)],
+ ["t-a", "t-b", "t-c", "t-a"])
+ self.assertIsNone(self.pick("- **t-a** [P0] (1h): A.\n", "haiku"))
+
+ def test_skipped_are_ranked_before_pick(self):
+ d = doc("- **t-a** [P0] (1h) (blocked: [[a-k]]): A.\n\n- **t-s** [P0] (1h): S.\n Model: sonnet\n\n"
+ "- **t-b** [P1] (1h): B.\n\n- **t-c** [P2] (1h): C.\n - After: [[t-b]]\n")
+ item, skipped = L.pick(d, set(), L.DEFAULT_LANES, "1h", "slow", "opus")
+ self.assertEqual((item.id, [(i.id, why) for i, why in skipped]), ("t-s", [("t-a", "blocked: a-k")]))
+ item, skipped = L.pick(d, set(), L.DEFAULT_LANES, "1h", "slow", "haiku")
+ self.assertEqual((item, skipped), (None, []))
+
+
+class ModelCheckTest(Base):
+ def test_values(self):
+ self.pending("- **t-a** [P1] (1h): A.\n Model: gpt\n\n"
+ "- **t-b** [P1] (1h): B.\n - Model: sonnet ok\n\n"
+ "- **t-c** [P1] (1h): C.\n Model: haiku\n Model: opus\n\n"
+ "- **t-d** [P1] (1h): D.\n")
+ errors, warnings = K.check(C.load(self.root))
+ self.assertEqual([(p.id, p.message) for p in errors],
+ [("t-a", "Model 'gpt' (want haiku, sonnet, opus)"), ("t-c", "two Model lines")])
+ self.assertEqual([(p.id, p.message) for p in warnings],
+ [("t-b", "Model line 'sonnet ok': write 'Model: sonnet' (wf set --model)")])
+
+
+class ModelCliTest(Cli):
+ tasks_text = LANES
+
+ def test_list_column_and_filter(self):
+ self.assertEqual(self.ok("list"),
+ "t-one P1 1h - sonnet slow One\n"
+ "t-two P2 1h - haiku slow Two\n"
+ "t-three P2 <1h - opus fast Three\n"
+ "pending 3 · human 0 · awaiting 0 · deferred 0\n")
+ self.assertEqual(self.ok("list", "--model", "opus"),
+ "t-three P2 <1h - opus fast Three\n"
+ "pending 3 · human 0 · awaiting 0 · deferred 0\n")
+
+ def test_add_and_set(self):
+ self.assertEqual(self.ok("add", "Four. Goal.", "-p", "3", "-e", "1h", "--model", "haiku", "--ref", "DESIGN.md"),
+ "- **t-four** [P3] (1h): Four. Goal.\n")
+ self.assertEqual(self.item("t-four"), "- **t-four** [P3] (1h): Four. Goal.\n Model: haiku\n Ref: DESIGN.md\n")
+ self.ok("set", "t-four", "--model", "sonnet")
+ self.assertEqual(self.item("t-four"), "- **t-four** [P3] (1h): Four. Goal.\n Model: sonnet\n Ref: DESIGN.md\n")
+ self.ok("set", "t-four", "--model", "")
+ self.assertEqual(self.item("t-four"), "- **t-four** [P3] (1h): Four. Goal.\n Ref: DESIGN.md\n")
+
+ def test_next_model_ceiling(self):
+ # --as M takes tasks with Model ≤ M; no --lane = all lanes, priority order
+ self.assertEqual(self.ok("next", "--as", "haiku", "--brief"), "- **t-two** [P2] (1h): Two.\n - Model: haiku\n")
+ self.assertEqual(self.ok("next", "--as", "opus", "--brief").splitlines()[0], "- **t-one** [P1] (1h): One.")
+ self.assertEqual(self.ok("next", "--as", "sonnet", "--brief").splitlines()[0], "- **t-one** [P1] (1h): One.")
+ self.assertEqual(self.ok("next", "--lane", "fast", "--as", "opus", "--brief"), "- **t-three** [P2] (<1h): Three.\n")
+ self.ok("set", "t-two", "--model", "opus")
+ self.assertEqual(self.fails("next", "--as", "haiku", "--brief").splitlines()[-1],
+ "wf: nothing pickable for all lanes (haiku) in Pending")
+
+ def test_next_without_as_is_haiku_with_hint(self):
+ out = self.ok("next", "--brief")
+ self.assertEqual(out, "no --as: treated as haiku; pass --as haiku|sonnet|opus\n"
+ "- **t-two** [P2] (1h): Two.\n - Model: haiku\n")
+
+ def session_env(self, sock):
+ sock.write_text("")
+ return {"CLAUDE_CODE_MESSAGING_SOCKET": str(sock), "CLAUDE_PID": str(os.getpid()),
+ "CLAUDE_CODE_SESSION_ID": "sess-1"}
+
+ def dead_pid(self):
+ p = subprocess.Popen(["true"])
+ p.wait()
+ return p.pid
+
+ def test_next_registers_session(self):
+ env = self.session_env(self.root / "me.sock")
+ self.ok("next", "--as", "opus", "--brief", env=env)
+ reg = json.loads((self.root / ".wf" / "sessions" / "all.json").read_text())
+ self.assertEqual({k: reg[k] for k in ("lane", "model", "socket", "pid", "session")},
+ {"lane": "all", "model": "opus", "socket": str(self.root / "me.sock"), "pid": os.getpid(), "session": "sess-1"})
+ self.assertEqual((self.root / ".wf" / ".gitignore").read_text(), "*\n")
+
+ def test_no_register_without_as_or_env(self):
+ none = {"CLAUDE_CODE_MESSAGING_SOCKET": "", "CLAUDE_PID": "", "CLAUDE_CODE_SESSION_ID": ""}
+ self.ok("next", "--as", "opus", env=none)
+ self.ok("next", env=self.session_env(self.root / "me.sock"))
+ self.assertFalse((self.root / ".wf").exists())
+
+ def register(self, lane, model, sock, pid):
+ d = self.root / ".wf" / "sessions"
+ d.mkdir(parents=True, exist_ok=True)
+ (d / f"{lane}.json").write_text(json.dumps({"lane": lane, "model": model, "socket": str(sock), "pid": pid,
+ "session": "s", "at": "2026-10-04T10:00"}))
+
+ def test_next_shows_lanes_and_waiting(self):
+ self.ok("set", "t-three", "--after", "t-one")
+ self.ok("set", "t-two", "--model", "opus")
+ sock = self.root / "son.sock"
+ sock.write_text("")
+ self.register("slow", "sonnet", sock, os.getpid())
+ self.register("fast", "haiku", self.root / "gone.sock", self.dead_pid())
+ none = {"CLAUDE_CODE_MESSAGING_SOCKET": ""}
+ code, out, err = self.wf("next", "--lane", "fast", "--as", "opus", env=none) # fast: no fallback lane
+ self.assertEqual((code, err), (1, "wf: nothing pickable for fast (opus) in Pending\n"))
+ self.assertIn("===== Lanes =====\n"
+ "fast (you): 0 pickable · 1 waiting\n"
+ f"slow: 2 pickable · 0 waiting · session uds:{sock} (alive, sonnet)\n\n"
+ "===== Waiting on other lanes =====\n"
+ f'- t-three waits on t-one (slow lane) → message uds:{sock}: "t-one blocks my t-three, please take it"\n', out)
+ self.assertEqual(self.ok("lanes", env=none),
+ "fast: 0 pickable · 1 waiting · no session\n"
+ f"slow: 0 pickable · 0 waiting · 2 not runner-ready (no Done) · session uds:{sock} (alive, sonnet)\n")
+
+ def test_next_single_lane_prints_no_lanes_block(self):
+ (self.root / "workflow.toml").write_text(self.toml + '[lanes.one]\nefforts = ["<1h", "1h"]\n')
+ self.assertNotIn("Lanes", self.ok("next", "--as", "opus", env={"CLAUDE_CODE_MESSAGING_SOCKET": ""}))
+
+ def test_done_notifies_other_lanes(self):
+ self.ok("add", "Four.", "-p", "3", "-e", "<1h", "--model", "haiku", "--after", "t-three")
+ self.ok("add", "Five.", "-p", "3", "-e", "1h", "--after", "t-three")
+ self.ok("add", "Six.", "-p", "3", "-e", "1h", "--model", "sonnet", "--after", "t-three")
+ sock = self.root / "h.sock"
+ sock.write_text("")
+ self.register("slow", "opus", sock, os.getpid())
+ out = self.ok("done", "t-three", "-m", "ok")
+ self.assertIn(f"notify slow uds:{sock}: now pickable t-five, t-six\n", out)
+ self.assertNotIn("t-four", out) # same lane as t-three (fast)
+
+ def test_bad_model_is_usage_error(self):
+ self.fails("add", "Four.", "-p", "3", "-e", "1h", "--model", "gpt", code=2)
+ self.fails("set", "t-one", "--model", "gpt", code=2)
+
+
+if __name__ == "__main__":
+ unittest.main()
diff --git a/tests/test_orch.py b/tests/test_orch.py
new file mode 100644
index 0000000..83216b2
--- /dev/null
+++ b/tests/test_orch.py
@@ -0,0 +1,227 @@
+import json
+import os
+import subprocess
+import unittest
+
+from test_cli import TOML, Cli
+from test_claims import git
+from test_merge import IDENT
+
+TASKS = """\
+# Tasks — demo
+
+## Awaiting your decision
+
+## Pending
+
+- **t-one** [P1] (<1h): One.
+ - Done: one works
+ - Model: sonnet
+
+- **t-two** [P2] (<1h): Two.
+ - Done: two works
+
+- **t-raw** [P2] (<1h): Raw, no Done line.
+
+- **t-big** [P2] (1h): Big.
+ - Done: big works
+
+## Needs human
+
+## Deferred
+"""
+
+
+class OrchTest(Cli):
+ tasks_text = TASKS
+ toml = TOML.replace('verify = ["make test"]\n', "")
+
+ def setUp(self):
+ super().setUp()
+ git(self.root, "init", "-q", "-b", "master")
+ (self.root / ".gitignore").write_text(".worktrees/\n.wf/\nout/\n")
+ git(self.root, "add", "-A")
+ git(self.root, "commit", "-qm", "init")
+
+ def orch(self, *args, code=0):
+ got, out, err = self.wf("orch", *args, env=IDENT)
+ self.assertEqual(got, code, out + err)
+ self.assertEqual(err, "")
+ return out
+
+ def run_wf(self, cwd, *args):
+ got, out, err = self.wf(*args, project=False, cwd=cwd, env=IDENT)
+ self.assertEqual(got, 0, out + err)
+ return out
+
+ def record(self, id):
+ return json.loads((self.root / ".wf" / "orch" / f"{id}.json").read_text())
+
+ def worker_done(self, id="t-one", lane="fast", merge=True):
+ """What a worker does: wf start, code, wf done/finish (finish merges, done alone leaves the branch)."""
+ wt = self.root / ".worktrees" / lane
+ self.run_wf(self.root, "start", id, "--worktree", str(wt), "--branch", f"{lane}/{id}")
+ (wt / "code.txt").write_text("x\n")
+ if merge:
+ self.run_wf(wt, "finish", id, "-m", "ok", "--commit", "impl", "code.txt", "--no-push")
+ else:
+ git(wt, "add", "code.txt")
+ git(wt, "commit", "-qm", "impl")
+ self.run_wf(wt, "done", id, "-m", "ok")
+ return wt
+
+ def log(self):
+ return (self.root / "out" / "wf-orch.log").read_text()
+
+ def test_pick_claims_and_prints_prompt(self):
+ out = self.orch("pick", "fast")
+ wt = self.root / ".worktrees" / "fast"
+ self.assertEqual(out, "pick: t-one (lane fast, model sonnet, effort <1h) · claimed (in progress: worker)\n"
+ "agent: subagent_type wf-worker, model sonnet, no isolation; prompt:\n"
+ "Task: t-one Lane: fast Model: sonnet\n"
+ f"Main tree: {self.root} Worktree: {wt} Branch: fast/t-one\n"
+ "Final message: the 4 report lines only.\n")
+ self.assertIn("- **t-one** [P1] (<1h) (in progress: worker): One.", self.tasks())
+ self.assertEqual(self.record("t-one")["worktree"], str(wt))
+
+ def test_second_pick_skips_in_progress_and_busy_worktree(self):
+ self.orch("pick", "fast")
+ out = self.orch("pick", "fast")
+ self.assertIn("pick: t-two (lane fast, model opus", out)
+ self.assertIn(f"Worktree: {self.root / '.worktrees' / 'fast-2'} Branch: fast/t-two\n", out)
+
+ def test_worktree_on_other_branch_not_reused(self):
+ git(self.root, "worktree", "add", "-q", str(self.root / ".worktrees" / "fast"), "-b", "fast/t-old")
+ self.assertIn(".worktrees/fast-2 Branch: fast/t-one", self.orch("pick", "fast"))
+
+ def test_clean_detached_worktree_reused(self):
+ git(self.root, "worktree", "add", "-q", "--detach", str(self.root / ".worktrees" / "fast"), "master")
+ self.assertIn(".worktrees/fast Branch: fast/t-one", self.orch("pick", "fast"))
+
+ def test_worktree_with_live_session_not_reused(self):
+ wt = self.root / ".worktrees" / "fast"
+ git(self.root, "worktree", "add", "-q", "--detach", str(wt), "master")
+ folder = self.root.parent / "no-claude" / "sessions"
+ folder.mkdir(parents=True)
+ (folder / "1.json").write_text(json.dumps({"pid": os.getpid(), "cwd": str(wt / "docs")}))
+ self.assertIn(".worktrees/fast-2 Branch: fast/t-one", self.orch("pick", "fast"))
+
+ def test_stop_file_spawns_nothing(self):
+ (self.root / "out").mkdir()
+ (self.root / "out" / "wf-batch.stop").write_text("")
+ self.assertEqual(self.orch("pick", "fast"), "stop: out/wf-batch.stop exists: spawn nothing (let running workers finish)\n")
+ self.assertNotIn("in progress", self.tasks())
+
+ def test_none_pickable(self):
+ self.orch("pick", "fast")
+ self.orch("pick", "fast")
+ self.assertTrue(self.orch("pick", "fast").startswith("none: lane fast has no runner-ready task"))
+
+ def test_explicit_id_and_recovery_line(self):
+ out = self.orch("pick", "slow", "--id", "t-big", "--recovery", "crashed")
+ self.assertIn("Branch: slow/t-big\nRecovery: crashed\nFinal message", out)
+
+ def test_post_done_logs_and_picks_next(self):
+ self.orch("pick", "fast")
+ self.worker_done()
+ out = self.orch("post", "t-one", "fast", "--result", "done", "--commit", "abc1234", "--duration", "75")
+ self.assertTrue(out.startswith("post: t-one done\n\npick: t-two"), out)
+ self.assertRegex(self.log(), r"^\S+ fast sonnet t-one done abc1234 1m15s\n$")
+ self.assertFalse((self.root / ".wf" / "orch" / "t-one.json").exists())
+
+ def test_post_no_next_claims_nothing(self):
+ self.orch("pick", "fast")
+ self.worker_done()
+ out = self.orch("post", "t-one", "fast", "--result", "done", "--no-next", "--duration", "5")
+ self.assertTrue(out.startswith("post: t-one done"), out)
+ self.assertNotIn("pick:", out)
+ self.assertEqual([f.name for f in (self.root / ".wf" / "orch").glob("*.json")], [])
+
+ def test_post_done_without_record_takes_model_from_history(self):
+ self.orch("pick", "fast")
+ self.worker_done()
+ (self.root / ".wf" / "orch" / "t-one.json").unlink()
+ self.orch("post", "t-one", "fast", "--result", "done", "--no-pick", "--duration", "5")
+ self.assertRegex(self.log(), r"^\S+ fast sonnet t-one done ")
+
+ def test_post_done_merges_unmerged_branch(self):
+ self.orch("pick", "fast")
+ wt = self.worker_done(merge=False)
+ out = self.orch("post", "t-one", "fast", "--result", "done", "--no-pick", "--no-push")
+ self.assertEqual(out, "post: t-one done\n merged worktree HEAD: merged fast/t-one into master\n")
+ self.assertEqual(subprocess.run(["git", "-C", str(self.root), "show", "master:code.txt"],
+ capture_output=True, text=True).stdout, "x\n")
+ self.assertEqual(subprocess.run(["git", "-C", str(wt), "branch", "--show-current"],
+ capture_output=True, text=True).stdout, "")
+
+ def test_post_done_without_archive_is_red(self):
+ self.orch("pick", "fast")
+ out = self.orch("post", "t-one", "fast", "--result", "done")
+ self.assertEqual(out, "post: t-one post-check-red\n no archive line for t-one\n"
+ "stop lane fast: post-check-red → tell the owner\n")
+ self.assertIn(" post-check-red ", self.log())
+
+ def backdate_pick(self, id="t-one", secs=125):
+ f = self.root / ".wf" / "orch" / f"{id}.json"
+ rec = json.loads(f.read_text())
+ rec["at"] -= secs
+ f.write_text(json.dumps(rec) + "\n")
+
+ def test_post_zero_duration_falls_back_to_pick_time(self):
+ self.orch("pick", "fast")
+ self.backdate_pick()
+ self.worker_done()
+ self.orch("post", "t-one", "fast", "--result", "done", "--no-pick", "--duration", "0")
+ self.assertRegex(self.log(), r" t-one done \S+ 2m0[5-9]s\n$")
+
+ def test_post_red_keeps_pick_time_for_the_repost(self):
+ self.orch("pick", "fast")
+ self.backdate_pick()
+ self.orch("post", "t-one", "fast", "--result", "done")
+ self.assertTrue((self.root / ".wf" / "orch" / "t-one.json").exists())
+ self.worker_done()
+ self.orch("post", "t-one", "fast", "--result", "done", "--no-pick")
+ self.assertRegex(self.log().splitlines()[-1], r" t-one done \S+ 2m0[5-9]s$")
+ self.assertFalse((self.root / ".wf" / "orch" / "t-one.json").exists())
+
+ def test_post_handback_commits_leftovers_and_stops(self):
+ self.orch("pick", "fast")
+ out = self.orch("post", "t-one", "fast", "--result", "handback unclear spec")
+ self.assertEqual(out, "post: t-one handback\n committed leftover TASKS.md tasks/archive.md\n"
+ "stop lane fast: handback → tell the owner\n")
+ self.assertEqual(subprocess.run(["git", "-C", str(self.root), "log", "-1", "--format=%s"],
+ capture_output=True, text=True).stdout, "t-one handback (orchestrator)\n")
+
+ def test_post_done_on_slice_job_counts_as_sliced(self):
+ self.orch("pick", "slow", "--id", "t-big")
+ self.run_wf(self.root, "add", "--parent", "t-big", "-e", "<1h", "--done", "x", "Slice one")
+ out = self.orch("post", "t-big", "slow", "--result", "done", "--no-next", "--duration", "5")
+ self.assertTrue(out.startswith("post: t-big sliced"), out)
+ self.assertNotIn("post-check-red", self.log())
+ self.assertIn(" t-big sliced ", self.log())
+
+ def test_post_model_raised_continues(self):
+ self.orch("pick", "fast")
+ self.wf("set", "t-one", "--model", "opus")
+ self.wf("status", "t-one", "clear")
+ out = self.orch("post", "t-one", "fast", "--result", "handback beyond sonnet")
+ self.assertTrue(out.startswith("post: t-one model-raised\n"), out)
+ self.assertIn("pick: t-one (lane fast, model opus", out)
+
+ def test_post_no_report_suggests_recovery(self):
+ self.orch("pick", "fast")
+ self.assertIn('wf orch pick fast --id t-one --recovery "<why>"', self.orch("post", "t-one", "fast"))
+
+ def test_post_agent_without_transcript_still_logs(self):
+ self.orch("pick", "fast")
+ out = self.orch("post", "t-one", "fast", "--result", "wip", "--agent", "abc")
+ self.assertIn(" cost: not logged (no transcript for agent abc)\n", out)
+
+ def test_refused_in_linked_worktree(self):
+ git(self.root, "worktree", "add", "-q", "--detach", str(self.root / ".worktrees" / "x"), "master")
+ got, _, err = self.wf("orch", "pick", "fast", project=False, cwd=self.root / ".worktrees" / "x")
+ self.assertEqual((got, err), (1, "wf: orch runs in the main tree (the orchestrator's)\n"))
+
+
+if __name__ == "__main__":
+ unittest.main()
diff --git a/tests/test_publish_snapshot.py b/tests/test_publish_snapshot.py
new file mode 100644
index 0000000..3243d4b
--- /dev/null
+++ b/tests/test_publish_snapshot.py
@@ -0,0 +1,122 @@
+import os
+import subprocess
+import sys
+import tempfile
+import unittest
+from pathlib import Path
+
+SCRIPT = Path(__file__).resolve().parent.parent / "scripts" / "publish_snapshot.py"
+ENV = {**os.environ, "GIT_AUTHOR_NAME": "t", "GIT_AUTHOR_EMAIL": "t@t", "GIT_COMMITTER_NAME": "t",
+ "GIT_COMMITTER_EMAIL": "t@t", "GIT_CONFIG_GLOBAL": "/dev/null"}
+
+
+def git(cwd, *args):
+ return subprocess.run(["git", *args], cwd=cwd, capture_output=True, text=True, env=ENV, check=True).stdout
+
+
+class PublishSnapshotTest(unittest.TestCase):
+ def setUp(self):
+ self.tmp = tempfile.TemporaryDirectory()
+ t = Path(self.tmp.name)
+ self.src, self.dest, self.deny = t / "tool", t / "pub", t / "deny.txt"
+ self.src.mkdir()
+ git(self.src, "init", "-q", "-b", "master")
+ self.put({"wf.py": "print('hi')\n", "CHANGES.md": "# Changes\n\n- 2026-01-01 first.\n",
+ "wflib/a.py": "x = 1\n", "docs/d.md": "doc\n",
+ "inbox.md": "tracked inbox\n", "out/log": "x\n", "sub/__pycache__/c.pyc": "bin\n"})
+ (self.src / "wf.py").chmod(0o755)
+ self.commit("one")
+ self.deny.write_text("# private\nSecretProj\n\n")
+
+ def tearDown(self):
+ self.tmp.cleanup()
+
+ def put(self, files):
+ for p, text in files.items():
+ f = self.src / p
+ f.parent.mkdir(parents=True, exist_ok=True)
+ f.write_text(text)
+
+ def commit(self, msg):
+ git(self.src, "add", "-A")
+ git(self.src, "commit", "-qm", msg)
+
+ def run_it(self, *extra):
+ return subprocess.run([sys.executable, str(SCRIPT), str(self.dest), "--src", str(self.src),
+ "--denylist", str(self.deny), *extra], capture_output=True, text=True, env=ENV)
+
+ def files(self):
+ return sorted(git(self.dest, "ls-files").split())
+
+ def test_first_run_one_commit_excludes(self):
+ r = self.run_it()
+ self.assertEqual(r.returncode, 0, r.stderr)
+ self.assertEqual(self.files(), ["CHANGES.md", "docs/d.md", "wf.py", "wflib/a.py"])
+ self.assertEqual(git(self.dest, "log", "--format=%s").splitlines(), ["initial public snapshot"])
+ self.assertTrue(os.access(self.dest / "wf.py", os.X_OK))
+ self.assertEqual(git(self.dest, "remote"), "")
+
+ def test_denylist_hit_content_and_path_writes_nothing(self):
+ self.put({"docs/d.md": "doc\nsee secretproj here\n", "SecretProj.md": "x\n"})
+ self.commit("leak")
+ r = self.run_it()
+ self.assertEqual(r.returncode, 1)
+ self.assertIn("docs/d.md:2: SecretProj", r.stderr)
+ self.assertIn("SecretProj.md: SecretProj (path)", r.stderr)
+ self.assertFalse(self.dest.exists())
+
+ def test_uncommitted_files_not_published(self):
+ (self.src / "wip.py").write_text("SecretProj\n")
+ r = self.run_it()
+ self.assertEqual(r.returncode, 0, r.stderr)
+ self.assertNotIn("wip.py", self.files())
+
+ def test_missing_or_empty_denylist_refuses(self):
+ self.deny.unlink()
+ r = self.run_it()
+ self.assertEqual(r.returncode, 1)
+ self.assertIn("no denylist", r.stderr)
+ self.deny.write_text("# only comments\n")
+ self.assertEqual(self.run_it().returncode, 1)
+ self.assertFalse(self.dest.exists())
+
+ def test_second_run_commit_message_new_changes_lines(self):
+ self.run_it()
+ self.put({"CHANGES.md": "# Changes\n\n- 2026-01-03 third.\n- 2026-01-02 second.\n- 2026-01-01 first.\n",
+ "wflib/b.py": "y = 2\n"})
+ (self.src / "docs/d.md").unlink()
+ self.commit("two")
+ r = self.run_it()
+ self.assertEqual(r.returncode, 0, r.stderr)
+ self.assertEqual(git(self.dest, "log", "-1", "--format=%B").strip(),
+ "public snapshot\n\n- 2026-01-03 third.\n- 2026-01-02 second.")
+ self.assertEqual(self.files(), ["CHANGES.md", "wf.py", "wflib/a.py", "wflib/b.py"])
+ self.assertEqual(len(git(self.dest, "log", "--format=%h").split()), 2)
+
+ def test_no_change_no_commit(self):
+ self.run_it()
+ r = self.run_it()
+ self.assertEqual(r.returncode, 0, r.stderr)
+ self.assertIn("nothing new", r.stdout)
+ self.assertEqual(len(git(self.dest, "log", "--format=%h").split()), 1)
+
+ def test_ref_option_and_never_touches_remote(self):
+ self.run_it()
+ git(self.dest, "remote", "add", "home", "/nonexistent")
+ self.put({"wflib/a.py": "x = 2\n"})
+ self.commit("two")
+ self.assertEqual(self.run_it("--ref", "HEAD~1").stdout.strip(), "nothing new: no commit")
+ r = self.run_it()
+ self.assertIn("committed", r.stdout)
+ self.assertEqual(git(self.dest, "remote").split(), ["home"])
+
+ def test_nonempty_non_git_dest_refused(self):
+ self.dest.mkdir()
+ (self.dest / "keep.txt").write_text("x")
+ r = self.run_it()
+ self.assertEqual(r.returncode, 1)
+ self.assertIn("not a git repo", r.stderr)
+
+
+if __name__ == "__main__":
+ unittest.main()
diff --git a/tests/test_refs.py b/tests/test_refs.py
new file mode 100644
index 0000000..4242a03
--- /dev/null
+++ b/tests/test_refs.py
@@ -0,0 +1,109 @@
+import sys
+import tempfile
+import unittest
+from pathlib import Path
+
+sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
+from wflib import refs as R
+
+DOC = """\
+# Design
+
+Intro.
+
+## Subsystems
+
+### Character Model
+
+Bones and poses.
+
+#### Detail
+
+Deep.
+
+<a id="terrain"></a>
+### Terrain & caves!
+
+Heightmap.
+
+## Other
+
+<a id="loose"></a>
+Loose paragraph anchor.
+"""
+
+
+class LinksTest(unittest.TestCase):
+ def test_links(self):
+ self.assertEqual(R.links("see [[t-a]] and [[a-b|label]] and [[doc#part]]"), ["t-a", "a-b", "doc"])
+
+ def test_code_is_skipped(self):
+ self.assertEqual(R.links("`[[t-x]]` then [[t-y]]\n```\n[[t-z]]\n```\n"), ["t-y"])
+
+
+class SectionsTest(unittest.TestCase):
+ def test_slugify(self):
+ self.assertEqual(R.slugify("Character Model"), "character-model")
+ self.assertEqual(R.slugify("Terrain & caves!"), "terrain-caves")
+
+ def test_sections(self):
+ got = [(s.level, s.heading, sorted(s.anchors), s.start, s.end) for s in R.sections(DOC)]
+ self.assertEqual(got, [
+ (1, "Design", ["design"], 0, 23),
+ (2, "Subsystems", ["subsystems"], 4, 19),
+ (3, "Character Model", ["character-model"], 6, 14),
+ (4, "Detail", ["detail"], 10, 14),
+ (3, "Terrain & caves!", ["terrain", "terrain-caves"], 15, 19),
+ (2, "Other", ["other"], 19, 23),
+ (0, "", ["loose"], 21, 23),
+ ])
+
+ def test_anchors(self):
+ self.assertEqual(R.anchors(DOC), {"design", "subsystems", "character-model", "detail",
+ "terrain", "terrain-caves", "other", "loose"})
+
+ def test_headings_in_code_fences_are_not_sections(self):
+ self.assertEqual(R.anchors("# A\n```\n# not a heading\n```\n"), {"a"})
+
+
+class ResolveTest(unittest.TestCase):
+ def setUp(self):
+ self.tmp = tempfile.TemporaryDirectory()
+ self.root = Path(self.tmp.name)
+ (self.root / "docs").mkdir()
+ (self.root / "DESIGN.md").write_text(DOC)
+ long = "# Long\n\n## Part\n" + "".join(f"line {n}\n" for n in range(1, 100))
+ (self.root / "docs" / "x.md").write_text(long)
+ (self.root / "docs" / "plan.md").write_text("# The Plan\n\n**Goal:** Ship it.\n\nMore.\n")
+
+ def tearDown(self):
+ self.tmp.cleanup()
+
+ def test_anchor_section(self):
+ self.assertEqual(R.resolve(self.root, "DESIGN.md", "terrain"),
+ "===== DESIGN.md#terrain (line 16) =====\n### Terrain & caves!\n\nHeightmap.")
+
+ def test_long_section_is_cut(self):
+ out = R.resolve(self.root, "docs/x.md", "part").split("\n")
+ self.assertEqual(out[0], "===== docs/x.md#part (line 3) =====")
+ self.assertEqual(out[1], "## Part")
+ self.assertEqual(out[80], "line 79")
+ self.assertEqual(out[81], "… (20 more lines: docs/x.md:83)")
+ self.assertEqual(len(out), 82)
+
+ def test_missing_anchor(self):
+ self.assertEqual(R.resolve(self.root, "DESIGN.md", "nope"), "(no anchor 'nope' in DESIGN.md)")
+
+ def test_missing_file(self):
+ self.assertEqual(R.resolve(self.root, "docs/none.md", None), "(missing: docs/none.md)")
+
+ def test_path_only_gives_heading_and_goal(self):
+ self.assertEqual(R.resolve(self.root, "docs/plan.md", None), "docs/plan.md: The Plan — Ship it.")
+ self.assertEqual(R.resolve(self.root, "DESIGN.md", None), "DESIGN.md: Design")
+
+ def test_directory(self):
+ self.assertEqual(R.resolve(self.root, "docs", None), "docs/ (directory)")
+
+
+if __name__ == "__main__":
+ unittest.main()
diff --git a/tests/test_res.py b/tests/test_res.py
new file mode 100644
index 0000000..6c92549
--- /dev/null
+++ b/tests/test_res.py
@@ -0,0 +1,751 @@
+import datetime as dt
+import json
+import sys
+import unittest
+from pathlib import Path
+
+sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
+from wflib import res as R
+
+UTC = dt.timezone.utc
+
+
+def T(h, m=0, day=1):
+ return dt.datetime(2026, 10, day, h, m, tzinfo=UTC)
+
+
+NOW = T(14, 10)
+GIB = 1024 ** 3
+
+
+def running(id="r-4", project="proj-a", title="dotnet e2e", mem=10.0, cpus=2, est=40, started=T(14, 0)):
+ return R.Entry(id=id, project=project, owner=1, title=title, mem_gb=mem, cpus=cpus, est_min=est,
+ state="running", unit=f"wf-{id}.service", started=started,
+ expires=started + dt.timedelta(minutes=2 * est))
+
+
+def note(id="r-8", owner=77, mem=3.0, cpus=1, expires=T(14, 30)):
+ return R.Entry(id=id, project="proj-b", owner=owner, title="dotnet test", mem_gb=mem, cpus=cpus,
+ est_min=20, state="note", started=T(14, 0), expires=expires)
+
+
+def facts(available=20.0, units=None, live=(), results=None, slice_gb=0.0, now=NOW):
+ return R.Facts(now=now, total_gb=30.0, available_gb=available, nproc=16, units=units or {},
+ live=set(live), results=results or {}, slice_gb=slice_gb)
+
+
+class Parse(unittest.TestCase):
+ def test_size(self):
+ self.assertEqual(R.parse_size("10G"), 10.0)
+ self.assertEqual(R.parse_size("512M"), 0.5)
+ self.assertEqual(R.parse_size("1.5g"), 1.5)
+ with self.assertRaisesRegex(R.ResError, "size '10'"):
+ R.parse_size("10")
+
+ def test_duration(self):
+ self.assertEqual(R.parse_duration("40m"), 40)
+ self.assertEqual(R.parse_duration("4h"), 240)
+ self.assertEqual(R.parse_duration("1h30m"), 90)
+ self.assertEqual(R.parse_duration("90"), 90)
+ for bad in ("0m", "x", "", "h"):
+ with self.assertRaises(R.ResError):
+ R.parse_duration(bad)
+
+ def test_meminfo(self):
+ text = "MemTotal: 31457280 kB\nMemFree: 1 kB\nMemAvailable: 12582912 kB\n"
+ self.assertEqual(R.meminfo(text), (30.0, 12.0))
+ with self.assertRaises(R.ResError):
+ R.meminfo("MemFree: 1 kB\n")
+
+ def test_show_units(self):
+ text = ("Id=agents.slice\nActiveState=active\nMemoryCurrent=0\n\n"
+ "Id=wf-r-1.service\nActiveState=inactive\nMemoryCurrent=[not set]\n")
+ self.assertEqual(R.show_units(text), {
+ "agents.slice": {"Id": "agents.slice", "ActiveState": "active", "MemoryCurrent": "0"},
+ "wf-r-1.service": {"Id": "wf-r-1.service", "ActiveState": "inactive", "MemoryCurrent": "[not set]"}})
+
+ def test_gb_or_none(self):
+ self.assertEqual(R.gb_or_none("1073741824"), 1.0)
+ self.assertIsNone(R.gb_or_none("[not set]"))
+ self.assertIsNone(R.gb_or_none("infinity"))
+
+ def test_formats(self):
+ self.assertEqual(R.fmt_gb(2.25), "2.2 GB")
+ self.assertEqual(R.fmt_bytes(1288490188), "1.2 GB")
+ self.assertEqual(R.fmt_bytes(300 * 1024 ** 2), "300 MB")
+ self.assertEqual(R.fmt_dur(240), "4h")
+ self.assertEqual(R.fmt_dur(90), "1h30m")
+ self.assertEqual(R.fmt_dur(40), "40m")
+ self.assertEqual(R.hhmm(T(9, 5)), "09:05")
+
+
+class ConfigTest(unittest.TestCase):
+ def test_defaults_and_override(self):
+ cfg = R.load_config("user_reserve_gb = 8\n")
+ self.assertEqual((cfg.user_reserve_gb, cfg.user_reserve_cpus, cfg.game_reserve_gb, cfg.game_reserve_cpus,
+ cfg.game_hours, cfg.small_headroom_gb, cfg.scratch_hours), (8.0, 4, 12.0, 8, 4.0, 2.0, 2.0))
+
+ def test_unknown_key(self):
+ self.assertEqual(R.load_config("session_mem_gb = 12").session_mem_gb, 12.0)
+ self.assertEqual(R.Config().session_mem_gb, 6.0)
+ with self.assertRaisesRegex(R.ResError, "unknown key\\(s\\) foo"):
+ R.load_config("foo = 1\n")
+
+ def test_negative(self):
+ with self.assertRaisesRegex(R.ResError, "game_hours must be a number ≥ 0"):
+ R.load_config("game_hours = -1\n")
+
+
+class LedgerJson(unittest.TestCase):
+ def test_roundtrip(self):
+ led = R.Ledger(next=9, game_until=T(18), entries=[running(), note()])
+ text = R.dumps(led)
+ self.assertIn('"started": "2026-10-01T14:00:00+00:00"', text)
+ self.assertEqual(R.loads(text), led)
+
+ def test_empty(self):
+ self.assertEqual(R.loads(""), R.Ledger())
+
+ def test_corrupt(self):
+ for bad in ("{", '{"entries": [{"id": "r-1"}]}', "[]"):
+ with self.assertRaisesRegex(R.ResError, "corrupt ledger"):
+ R.loads(bad)
+
+ def test_unknown_entry_keys_ignored(self):
+ d = json.loads(R.dumps(R.Ledger(next=9, entries=[running()])))
+ d["entries"][0]["from_a_newer_version"] = 1
+ self.assertEqual(R.loads(json.dumps(d)), R.Ledger(next=9, entries=[running()]))
+
+ def test_unknown_keys_survive_rewrite(self):
+ d = json.loads(R.dumps(R.Ledger(next=9, entries=[running()])))
+ d["entries"][0]["field_x"] = {"a": 1}
+ d["top_x"] = 5
+ out = json.loads(R.dumps(R.loads(json.dumps(d))))
+ self.assertEqual(out["entries"][0]["field_x"], {"a": 1})
+ self.assertEqual(out["top_x"], 5)
+
+ def test_new_id_and_get(self):
+ led = R.Ledger(next=3)
+ self.assertEqual(led.new_id(), "r-3")
+ self.assertEqual(led.next, 4)
+ with self.assertRaisesRegex(R.ResError, "no entry 'r-9'"):
+ led.get("r-9")
+
+
+class Prune(unittest.TestCase):
+ def test_prune(self):
+ r1, r2 = running("r-1"), running("r-2")
+ n3, n5 = note("r-3", owner=77), note("r-5", owner=78, expires=T(14, 5))
+ d6 = R.Entry(id="r-6", project="p", owner=1, title="old", mem_gb=1, cpus=1, est_min=1, state="done",
+ ended=NOW - dt.timedelta(hours=25))
+ d7 = R.Entry(id="r-7", project="p", owner=1, title="new", mem_gb=1, cpus=1, est_min=1, state="done",
+ ended=NOW - dt.timedelta(hours=23))
+ led = R.Ledger(game_until=T(14, 9), entries=[r1, r2, n3, n5, d6, d7])
+ f = facts(units={"wf-r-1.service": R.Unit(False)}, live={78}, results={"r-1": (0, 9.4, "")})
+ lines = R.prune(led, f)
+ self.assertEqual(lines, ["r-1 exited rc=0", "r-2 exited rc=?", "r-3 freed (owner gone)",
+ "r-5 freed (expired)", "game off (expired)"])
+ self.assertEqual([e.id for e in led.entries], ["r-1", "r-2", "r-3", "r-5", "r-7"])
+ self.assertEqual((r1.state, r1.rc, r1.peak_gb, r1.why, r1.ended), ("done", 0, 9.4, "exited", NOW))
+ self.assertEqual((n3.why, n5.why), ("owner gone", "expired"))
+ self.assertIsNone(led.game_until)
+
+ def test_prune_missing_results(self):
+ r = running("r-1")
+ R.prune(R.Ledger(entries=[r]), facts(units={"wf-r-1.service": R.Unit(False)}))
+ self.assertEqual((r.state, r.rc, r.peak_gb), ("done", None, None))
+
+ def test_overdue_running_is_kept(self):
+ r = running(started=T(10))
+ R.prune(R.Ledger(entries=[r]), facts(units={"wf-r-4.service": R.Unit(True, 4.0)}))
+ self.assertEqual(r.state, "running")
+
+
+class Capacity(unittest.TestCase):
+ def setUp(self):
+ self.cfg = R.Config()
+
+ def test_budget(self):
+ led = R.Ledger(entries=[running(), note()])
+ f = facts(units={"wf-r-4.service": R.Unit(True, 4.0)}, live={77})
+ # 20 − 6 reserve − 2 headroom − (10−4) unused − 3 note = 3; 16 − 4 − (2+1) = 9
+ self.assertEqual(R.budget(self.cfg, led, f), (3.0, 9))
+
+ def test_budget_gaming(self):
+ led = R.Ledger(game_until=T(15), entries=[running(), note()])
+ f = facts(units={"wf-r-4.service": R.Unit(True, 4.0)}, live={77})
+ # 20 − 12 − 2 − 6 − 3 = −3; 16 − 8 − 3 = 5
+ self.assertEqual(R.budget(self.cfg, led, f), (-3.0, 5))
+
+ def test_held_over_reservation_is_zero(self):
+ self.assertEqual(R.held_gb(running(), facts(units={"wf-r-4.service": R.Unit(True, 12.0)})), 0.0)
+ self.assertEqual(R.frees_gb(running(), facts(units={"wf-r-4.service": R.Unit(True, 12.0)})), 12.0)
+
+ def test_fits(self):
+ self.assertTrue(R.fits(3.0, 9, (3.0, 9)))
+ self.assertFalse(R.fits(3.1, 1, (3.0, 9)))
+ self.assertFalse(R.fits(1.0, 10, (3.0, 9)))
+
+ def test_never_fits(self):
+ f = facts()
+ self.assertEqual(R.never_fits(self.cfg, f, 25.0, 1), "25.0 GB can never fit (max 22.0 GB for agents)")
+ self.assertEqual(R.never_fits(self.cfg, f, 1.0, 13), "13 cpus can never fit (max 12 for agents)")
+ self.assertIsNone(R.never_fits(self.cfg, f, 22.0, 12))
+
+ def test_queued_total(self):
+ q = running("r-9")
+ q.state = "queued"
+ self.assertEqual(R.queued_total(R.Ledger(entries=[q, running()])), (10.0, 2))
+
+
+class BusyLine(unittest.TestCase):
+ def setUp(self):
+ self.cfg = R.Config()
+ self.led = R.Ledger(entries=[running()])
+ self.f = facts(units={"wf-r-4.service": R.Unit(True, 4.0)}) # budget 6.0 GB, 10 cpus
+
+ def test_memory(self):
+ self.assertEqual(R.busy_line(self.cfg, self.led, self.f, 8.0, 1),
+ 'busy: 10.0 GB held by proj-a "dotnet e2e" (r-4) until ~14:40; 6.0 GB free for agents; '
+ 'retry after ~14:40 or work on something else')
+
+ def test_cpus(self):
+ self.assertEqual(R.busy_line(self.cfg, self.led, self.f, 1.0, 12),
+ 'busy: 2 cpus held by proj-a "dotnet e2e" (r-4) until ~14:40; 10 cpus free for agents; '
+ 'retry after ~14:40 or work on something else')
+
+ def test_no_holder(self):
+ self.assertEqual(R.busy_line(self.cfg, R.Ledger(), facts(available=7.0), 1.0, 1),
+ "busy: only 0.0 GB free for agents and no agent job holds any; other programs use the rest; "
+ "retry later or work on something else")
+
+ def test_busy_overdue_holder(self):
+ led = R.Ledger(entries=[running(started=T(10))])
+ self.assertEqual(R.busy_line(self.cfg, led, self.f, 8.0, 1),
+ 'busy: 10.0 GB held by proj-a "dotnet e2e" (r-4) until overdue; 6.0 GB free for agents; '
+ 'retry later or work on something else')
+
+ def test_two_holders_earliest_first(self):
+ a = running("r-1", project="a", title="A", mem=4.0, cpus=1, est=60) # ends 15:00
+ b = running("r-2", project="b", title="B", mem=4.0, cpus=1, est=30) # ends 14:30
+ f = facts(available=14.0, units={}) # 14 − 6 − 2 − 4 − 4 = −2.0 budget
+ self.assertEqual(R.busy_line(self.cfg, R.Ledger(entries=[a, b]), f, 5.0, 1),
+ 'busy: 4.0 GB held by b "B" (r-2) until ~14:30, 4.0 GB held by a "A" (r-1) until ~15:00; '
+ '0.0 GB free for agents; retry after ~15:00 or work on something else')
+
+
+class Lines(unittest.TestCase):
+ def test_run_argv(self):
+ e = running("r-3", mem=10.0)
+ e.cwd, e.log, e.cmd = "/projects/x", "/s/logs/r-3.log", ["make", "it big", "--fast"]
+ self.assertEqual(R.run_argv(e, "/s/logs"), [
+ "systemd-run", "--user", "--quiet", "--collect", "--slice=agents-jobs.slice", "--unit=wf-r-3.service",
+ "--working-directory=/projects/x", "--setenv=WF_RES_ID=r-3",
+ "-p", "MemoryMax=10737418240", "-p", "MemoryHigh=9663676416", "-p", "MemorySwapMax=0", "-p", "Nice=10",
+ "-p", "StandardOutput=append:/s/logs/r-3.log", "-p", "StandardError=append:/s/logs/r-3.log",
+ "/bin/sh", "-c",
+ '"$@"; rc=$?; cat /sys/fs/cgroup$(cut -d: -f3 /proc/self/cgroup)/memory.peak > /s/logs/r-3.peak '
+ '2>/dev/null; echo $rc > /s/logs/r-3.rc',
+ "sh", "make", "it big", "--fast"])
+
+ def test_run_argv_env(self):
+ e = running("r-3", mem=1.0)
+ e.cwd, e.log, e.cmd, e.env = "/p", "/s/r-3.log", ["x"], {"B": "2", "A": "1 2"}
+ argv = R.run_argv(e, "/s")
+ self.assertEqual(argv[argv.index("--working-directory=/p") + 1:][:2], ["--setenv=A=1 2", "--setenv=B=2"])
+
+ def test_env_diff(self):
+ self.assertEqual(R.env_diff({"A": "1", "B": "2", "C": "3", "PWD": "/", "OLDPWD": "/", "SHLVL": "1", "_": "/x",
+ "MY_TOKEN": "t", "API_SECRET": "s", "PGPASSWORD": "p", "SSH_AUTH_SOCK": "/s"},
+ {"A": "1", "B": "9", "SSH_AUTH_SOCK": "/s"}),
+ {"B": "2", "C": "3"})
+
+ def test_manager_env_parse(self):
+ self.assertEqual(R.parse_show_environment("A=1\nB=x=y\n\nbad\n"), {"A": "1", "B": "x=y"})
+
+ def test_entry_lines(self):
+ f = facts(units={"wf-r-4.service": R.Unit(True, 4.0)}, live={77})
+ q = running("r-9", project="p", title="Q", mem=1.0, cpus=1)
+ q.state = "queued"
+ d = R.Entry(id="r-2", project="p", owner=1, title="D", mem_gb=1, cpus=1, est_min=1, state="done",
+ ended=T(13, 50), rc=None, peak_gb=None, why="exited")
+ led = R.Ledger(entries=[running(), note(), q, d])
+ self.assertEqual([R.entry_line(e, led, f) for e in led.entries], [
+ 'r-4 proj-a "dotnet e2e" running 10.0 GB used 4.0 GB 2 cpu since 14:00 ETA ~14:40',
+ 'r-8 proj-b "dotnet test" note 3.0 GB 1 cpu until ~14:30',
+ 'r-9 p "Q" queued #1 1.0 GB 1 cpu',
+ 'r-2 p "D" done (exited) rc=? peak ? at 13:50'])
+
+ def test_status_lines(self):
+ f = facts(units={"wf-r-4.service": R.Unit(True, 4.0)}, slice_gb=5.5)
+ self.assertEqual(R.status_lines(R.Config(), R.Ledger(entries=[running()]), f), [
+ 'r-4 proj-a "dotnet e2e" running 10.0 GB used 4.0 GB 2 cpu since 14:00 ETA ~14:40',
+ "agents may use 6.0 GB, 10 cpus now; reserve 6.0 GB/4 cpus",
+ "unreserved agent memory 1.5 GB"])
+ f.slice_cache_gb = 0.6
+ self.assertEqual(R.status_lines(R.Config(), R.Ledger(entries=[running()]), f)[-1],
+ "unreserved agent memory 1.5 GB (0.6 GB of it file cache, reclaimable)")
+ f.slice_cache_gb = 3.0 # cache partly inside the jobs' 4.0 GB → capped at the unreserved figure
+ self.assertEqual(R.status_lines(R.Config(), R.Ledger(entries=[running()]), f)[-1],
+ "unreserved agent memory 1.5 GB (1.5 GB of it file cache, reclaimable)")
+
+ def test_cache_gb(self):
+ # agents.slice memory.stat excerpt from this machine: file 5602209792, shmem (tmpfs, not reclaimable) 3861041152
+ self.assertEqual(round(R.cache_gb("anon 11335516160\nfile 5602209792\nkernel 1\nshmem 3861041152\n"), 3), 1.622)
+ self.assertEqual(R.cache_gb(""), 0.0)
+
+ def test_status_lines_empty_overdue(self):
+ self.assertEqual(R.status_lines(R.Config(), R.Ledger(), facts(available=7.0)), [
+ "no reservations", "agents may use 0.0 GB, 12 cpus now; reserve 6.0 GB/4 cpus"])
+ r = running(started=T(10))
+ self.assertIn("ETA overdue (still running, not killed)", R.entry_line(r, R.Ledger(entries=[r]), facts()))
+
+ def test_status_json(self):
+ import json
+ d = json.loads(R.status_json(R.Config(), R.Ledger(entries=[running()]),
+ facts(units={"wf-r-4.service": R.Unit(True, 4.0)})))
+ self.assertEqual((d["budget_gb"], d["budget_cpus"], d["gaming"], d["entries"][0]["id"]),
+ (6.0, 10, False, "r-4"))
+
+ def test_done_line(self):
+ e = running("r-3", started=T(14, 0))
+ e.state, e.ended, e.rc, e.peak_gb = "done", T(14, 37), 0, 9.4
+ self.assertEqual(R.done_line(e), "r-3 done rc=0 peak 9.4 GB in 37 min")
+ e.rc, e.peak_gb = None, None
+ self.assertEqual(R.done_line(e), "r-3 done rc=? peak ? in 37 min")
+ e.why = "killed: timeout"
+ self.assertEqual(R.done_line(e), "r-3 done rc=? peak ? in 37 min (killed: timeout)")
+
+ def test_prune_killed(self):
+ r = running("r-1")
+ f = facts(units={"wf-r-1.service": R.Unit(False)}, results={"r-1": (None, 3.9, "killed: timeout")})
+ self.assertEqual(R.prune(R.Ledger(entries=[r]), f), ["r-1 killed: timeout"])
+ self.assertEqual((r.state, r.rc, r.peak_gb, r.why), ("done", None, 3.9, "killed: timeout"))
+
+
+# journalctl --user -u wf-r-12.service -o cat, verbatim from a real systemd-oomd kill (systemd 258)
+OOMD = """Started wf-r-12.service - [systemd-run] /bin/sh -c "x" sh dotnet test.
+wf-r-12.service: systemd-oomd killed some process(es) in this unit.
+wf-r-12.service: Main process exited, code=killed, status=9/KILL
+wf-r-12.service: Failed with result 'oom-kill'.
+wf-r-12.service: Consumed 11min 2.837s CPU time, 3.9G memory peak.
+"""
+
+
+class JournalReason(unittest.TestCase):
+ def test_oomd(self):
+ self.assertEqual(R.journal_reason(OOMD, 4.0),
+ ("killed: oom-kill by systemd-oomd, limit 4.0 GB; raise --mem", 3.9))
+
+ def test_kernel_oom(self):
+ text = ("u: A process of this unit has been killed by the OOM killer.\n"
+ "u: Main process exited, code=killed, status=9/KILL\n"
+ "u: Failed with result 'oom-kill'.\nu: Consumed 2s CPU time, 512M memory peak.\n")
+ self.assertEqual(R.journal_reason(text, 0.5),
+ ("killed: oom-kill at MemoryMax, limit 0.5 GB; raise --mem", 0.5))
+
+ def test_signal(self):
+ text = "u: Main process exited, code=killed, status=15/TERM\nu: Failed with result 'signal'.\n"
+ self.assertEqual(R.journal_reason(text, 1.0), ("killed: signal 15/TERM", None))
+
+ def test_timeout_and_exit(self):
+ self.assertEqual(R.journal_reason("u: Failed with result 'timeout'.\n", 1.0), ("killed: timeout", None))
+ text = "u: Main process exited, code=exited, status=3/NOTIMPLEMENTED\nu: Failed with result 'exit-code'.\n"
+ self.assertEqual(R.journal_reason(text, 1.0), ("failed: exit 3/NOTIMPLEMENTED", None))
+
+ def test_nothing(self):
+ self.assertEqual(R.journal_reason("", 1.0), ("", None))
+ self.assertEqual(R.journal_reason("u: Deactivated successfully.\nu: Consumed 1s CPU time, 1.5G memory peak.\n", 1.0),
+ ("", 1.5))
+
+def queued(id, mem, cpus=1, at=T(13, 0)):
+ e = R.Entry(id=id, project="p", owner=1, title=id, mem_gb=mem, cpus=cpus, est_min=10, state="queued",
+ cmd=["x"], queued=at)
+ return e
+
+
+class Queue(unittest.TestCase):
+ def test_fifo_blocked_head(self):
+ # budget 20 − 6 − 2 = 12 GB
+ led = R.Ledger(entries=[queued("r-1", 5, at=T(13, 0)), queued("r-2", 9, at=T(13, 1)),
+ queued("r-3", 1, at=T(13, 2))])
+ self.assertEqual([e.id for e in R.to_start(R.Config(), led, facts())], ["r-1"])
+
+ def test_all_fit(self):
+ led = R.Ledger(entries=[queued("r-2", 4, at=T(13, 1)), queued("r-1", 4, at=T(13, 0))])
+ self.assertEqual([e.id for e in R.to_start(R.Config(), led, facts())], ["r-1", "r-2"])
+
+ def test_estimate(self):
+ # running r-4 holds 10 (used 4), ends 14:40; budget 6; queue r-1 (5) ahead of r-2 (4): need 9 − 6 = 3
+ led = R.Ledger(entries=[running(), queued("r-1", 5), queued("r-2", 4, at=T(13, 5))])
+ f = facts(units={"wf-r-4.service": R.Unit(True, 4.0)})
+ self.assertEqual(R.queue_estimate(R.Config(), led, f, led.get("r-2")), T(14, 40))
+
+ def test_estimate_unknown(self):
+ led = R.Ledger(entries=[queued("r-1", 20)])
+ self.assertIsNone(R.queue_estimate(R.Config(), led, facts(), led.get("r-1")))
+
+class Game(unittest.TestCase):
+ def setUp(self):
+ self.cfg = R.Config()
+
+ def test_slice_props(self):
+ self.assertEqual(R.slice_props(self.cfg, R.Ledger(), facts()),
+ {"CPUWeight": "20", "IOWeight": "20", "MemoryHigh": str(24 * GIB)})
+ on = R.Ledger(game_until=T(18))
+ self.assertEqual(R.slice_props(self.cfg, on, facts(slice_gb=10.0))["MemoryHigh"], str(18 * GIB))
+ self.assertEqual(R.slice_props(self.cfg, on, facts(slice_gb=20.0)),
+ {"CPUWeight": "5", "IOWeight": "5", "MemoryHigh": str(20 * GIB)})
+
+ def test_shortfall(self):
+ a = running("r-4", project="proj-a", title="dotnet e2e", mem=2.5, est=55, started=T(14, 10)) # ~15:05
+ b = running("r-7", project="proj-b", title="r5 rebuild", mem=9.0, est=80, started=T(14, 10)) # ~15:30
+ f = facts(available=11.5, units={"wf-r-4.service": R.Unit(True, 2.5), "wf-r-7.service": R.Unit(True, 9.0)})
+ led = R.Ledger(game_until=T(18), entries=[a, b])
+ # free for you 11.5 − 0 unused = 11.5 → short 0.5: r-4 alone frees 2.5
+ self.assertEqual(R.shortfall_lines(self.cfg, led, f), [
+ 'short 0.5 GB of 12.0 GB: r-4 proj-a "dotnet e2e" 2.5 GB ~15:05',
+ "full reserve free ~15:05 (est.); free now: wf res release r-4 --stop"])
+ f.available_gb = 8.9 # short 3.1: needs both
+ self.assertEqual(R.shortfall_lines(self.cfg, led, f), [
+ 'short 3.1 GB of 12.0 GB: r-4 proj-a "dotnet e2e" 2.5 GB ~15:05, r-7 proj-b "r5 rebuild" 9.0 GB ~15:30',
+ "full reserve free ~15:30 (est.); free now: wf res release r-7 --stop"])
+ f.available_gb = 13.0
+ self.assertEqual(R.shortfall_lines(self.cfg, led, f), [])
+
+ def test_shortfall_not_agents(self):
+ self.assertEqual(R.shortfall_lines(self.cfg, R.Ledger(game_until=T(18)), facts(available=5.0)), [
+ "short 7.0 GB of 12.0 GB: other programs, not agent jobs",
+ "agent jobs alone cannot free it; close other programs"])
+
+ def test_game_on_lines(self):
+ led = R.Ledger(game_until=T(18, 10))
+ self.assertEqual(R.game_on_lines(self.cfg, led, facts(available=20.0), 240),
+ ["game on until 18:10 (4h); CPU/IO now yours"])
+
+ def test_status_shows_game(self):
+ lines = R.status_lines(self.cfg, R.Ledger(game_until=T(18, 10)), facts(available=5.0))
+ self.assertEqual(lines[1:], ["agents may use 0.0 GB, 8 cpus now; reserve 12.0 GB/8 cpus",
+ "game on until 18:10 (4h left)",
+ "short 7.0 GB of 12.0 GB: other programs, not agent jobs",
+ "agent jobs alone cannot free it; close other programs"])
+
+class Clean(unittest.TestCase):
+ def test_scratch_victims(self):
+ now = 10_000_000.0
+ h = 3600
+ tree = {
+ "/s/old-file.png": (now - 3 * h, {}),
+ "/s/-projects-a": (now - 3 * h, {"/s/-projects-a/s1": now - 3 * h}),
+ "/s/-projects-b": (now - 60, {"/s/-projects-b/s1": now - 5 * h, "/s/-projects-b/s2": now - 60}),
+ "/s/new.txt": (now - 60, {}),
+ }
+ self.assertEqual(R.scratch_victims(tree, now, 2.0),
+ ["/s/-projects-a", "/s/-projects-b/s1", "/s/old-file.png"])
+ # live sessions keep their dir however idle; a stale top holding one is pruned per child
+ self.assertEqual(R.scratch_victims(tree, now, 2.0, keep={"s1"}), ["/s/old-file.png"])
+ tree["/s/-projects-a"] = (now - 3 * h, {"/s/-projects-a/s1": now - 3 * h, "/s/-projects-a/s9": now - 3 * h})
+ self.assertEqual(R.scratch_victims(tree, now, 2.0, keep={"s1"}), ["/s/-projects-a/s9", "/s/old-file.png"])
+
+ def test_live_session_id(self):
+ text = '{"pid": 817151, "sessionId": "4ad140ab-69ca", "procStart": "31379851", "kind": "interactive"}'
+ self.assertEqual(R.live_session_id(text, "31379851"), "4ad140ab-69ca")
+ self.assertIsNone(R.live_session_id(text, "999")) # pid reused by another process
+ self.assertEqual(R.live_session_id('{"sessionId": "x"}', "5"), "x") # no procStart recorded → trust pid
+ self.assertIsNone(R.live_session_id("{oops", "5"))
+ self.assertIsNone(R.live_session_id('{"pid": 1}', "5"))
+
+ def test_cleanup_rule(self):
+ self.assertEqual(R.cleanup_rule("out/prof"), ("out/prof", None))
+ self.assertEqual(R.cleanup_rule("out/history-logs/*.log:30d"), ("out/history-logs/*.log", 30))
+ for bad in ("/etc", "../x", "out/../../x"):
+ with self.assertRaisesRegex(R.ResError, "cleanup pattern"):
+ R.cleanup_rule(bad)
+
+ def test_proc_name(self):
+ # background sessions run the versioned binary: comm is the version, argv0 the path or `claude`
+ self.assertEqual(R.proc_name("2.1.283", "claude"), "claude")
+ self.assertEqual(R.proc_name("2.1.283", "claude bg-pty-host --bg-pty-host /tmp/x.sock"), "claude") # rewritten title
+ self.assertEqual(R.proc_name("2.1.283", "/h/u/.local/share/claude/versions/2.1.283"), "claude")
+ self.assertEqual(R.proc_name("claude", "/h/u/.local/bin/claude"), "claude")
+ self.assertEqual(R.proc_name("2.1.284", "ugrep"), "2.1.284") # tool re-exec of the binary: not a session
+ self.assertEqual(R.proc_name("bash", "/bin/bash"), "bash")
+ self.assertEqual(R.proc_name("x", ""), "x") # kernel thread / unreadable cmdline
+
+ def test_adopt_groups(self):
+ procs = [
+ (100, 1, "bash", "/u/app.slice/tab1.scope"),
+ (101, 100, "claude", "/u/app.slice/tab1.scope"), # root, outside
+ (102, 101, "bash", "/u/app.slice/tab1.scope"), # its shell
+ (103, 102, "make", "/u/app.slice/tab1.scope"),
+ (104, 101, "claude", "/u/app.slice/tab1.scope"), # child claude: not a root, but a descendant
+ (105, 101, "job", "/u/agents.slice/agents-jobs.slice/wf-r-1.service"), # already inside: skipped
+ (200, 1, "claude", "/u/agents.slice/run-9.scope"), # root already inside
+ (300, 1, "vim", "/u/app.slice/tab2.scope"),
+ ]
+ self.assertEqual(R.adopt_groups(procs), {101: [101, 102, 103, 104]})
+
+ def test_warning_lines(self):
+ self.assertEqual(R.warning_lines(30.0, 7.6, 2), [
+ "warning: /tmp (RAM) holds 7.6 GB; wf res clean",
+ "warning: 2 claude sessions outside agents.slice (wf res adopt)"])
+ self.assertEqual(R.warning_lines(30.0, 7.4, 0), [])
+
+class Units(unittest.TestCase):
+ def test_adopt_argv(self):
+ self.assertEqual(R.adopt_argv(101, [101, 102]), [
+ "busctl", "--user", "call", "org.freedesktop.systemd1", "/org/freedesktop/systemd1",
+ "org.freedesktop.systemd1.Manager", "StartTransientUnit", "ssa(sv)a(sa(sv))",
+ "wf-claude-101.scope", "fail", "2", "PIDs", "au", "2", "101", "102", "Slice", "s", "agents.slice", "0"])
+
+ def test_texts(self):
+ self.assertEqual(R.service_text("/usr/bin/python3", "/projects/public/workflow/wf.py"),
+ "[Unit]\nDescription=wf res tick\n\n[Service]\nType=oneshot\n"
+ "ExecStart=/usr/bin/python3 /projects/public/workflow/wf.py res tick\n")
+ self.assertEqual(R.timer_text(),
+ "[Unit]\nDescription=wf res tick every minute\n\n[Timer]\nOnBootSec=1min\n"
+ "OnUnitActiveSec=1min\n\n[Install]\nWantedBy=timers.target\n")
+ self.assertEqual(R.shell_init_line(),
+ "alias claude='systemd-run --user --scope --quiet --slice=agents.slice claude'")
+
+
+if __name__ == "__main__":
+ unittest.main()
+
+
+class Throttle(unittest.TestCase):
+ def test_psi(self):
+ text = "some avg10=41.50 avg60=33.20 avg300=12.00 total=99\nfull avg10=30.00 avg60=25.00 avg300=9.00 total=88\n"
+ self.assertEqual(R.psi_some_avg60(text), 33.2)
+ self.assertIsNone(R.psi_some_avg60(""))
+
+ def test_lines(self):
+ hot = running("r-1", mem=4.0) # 3.5 of 4.0 GB (≥ 0.85×), stall 33% → warn
+ cache = running("r-2", mem=4.0) # full of page cache, no stall → quiet
+ small = running("r-3", mem=10.0) # stalled but far below its limit (machine pressure) → quiet
+ f = facts(units={"wf-r-1.service": R.Unit(True, 3.5, 33.2), "wf-r-2.service": R.Unit(True, 3.9, 0.0),
+ "wf-r-3.service": R.Unit(True, 2.0, 50.0)})
+ self.assertEqual(R.throttle_lines(R.Ledger(entries=[hot, cache, small]), f), [
+ "r-1 throttled at its memory limit (3.5 of 4.0 GB, stalled 33% of the last minute): "
+ "likely too small; `wf res release r-1 --stop` and re-run with a bigger --mem"])
+
+
+class Owner(unittest.TestCase):
+ REC = [{"lane": "fast", "pid": 11, "socket": "/s/a"}, {"lane": "slow", "pid": 22, "socket": "/s/b"}]
+
+ def test_from_env_branch_and_session_record(self):
+ env = {"CLAUDE_PID": "22", "CLAUDE_CODE_MESSAGING_SOCKET": "/s/b", "WF_RES_ID": "r-7"}
+ self.assertEqual(R.owner_by(env, "slow/t-res-owner", self.REC),
+ {"name": "slow session", "task": "t-res-owner", "batch": "r-7", "address": "uds:/s/b"})
+
+ def test_explicit_name_wins_then_env(self):
+ env = {"CLAUDE_PID": "22", "WF_SESSION_NAME": "pilot", "WF_TASK": "t-x"}
+ self.assertEqual(R.owner_by(env, "slow/t-y", self.REC, "worker-3"), {"name": "worker-3", "task": "t-x"})
+ self.assertEqual(R.owner_by(env, "", self.REC), {"name": "pilot", "task": "t-x"})
+
+ def test_nothing_known(self):
+ self.assertEqual(R.owner_by({"CLAUDE_PID": "99"}, "master", self.REC), {})
+ self.assertEqual(R.owner_by({}, "feature-x", [{"pid": 1}]), {})
+
+ def test_status_and_throttle_name_owner(self):
+ e = running("r-1", mem=4.0)
+ e.by = {"name": "slow session", "task": "t-a", "batch": "r-7", "address": "uds:/s/b"}
+ f = facts(units={"wf-r-1.service": R.Unit(True, 3.5, 33.2)})
+ led = R.Ledger(entries=[e])
+ self.assertTrue(R.status_lines(R.Config(), led, f)[0].endswith(
+ " [by slow session t-a batch r-7, message uds:/s/b]"))
+ self.assertTrue(R.throttle_lines(led, f)[0].endswith(
+ "bigger --mem; started by slow session t-a batch r-7, message uds:/s/b"))
+
+ def test_ledger_roundtrip_and_old_entries(self):
+ e = running("r-1")
+ e.by = {"task": "t-a"}
+ self.assertEqual(R.loads(R.dumps(R.Ledger(entries=[e]))).get("r-1").by, {"task": "t-a"})
+ d = json.loads(R.dumps(R.Ledger(entries=[running("r-2")])))
+ for k in ("by", "lock"): # entries written before these fields
+ del d["entries"][0][k]
+ old = R.loads(json.dumps(d)).get("r-2")
+ self.assertEqual((old.by, old.lock), ({}, ""))
+
+
+class BareUnits(unittest.TestCase):
+ def test_bare_units(self):
+ blocks = R.show_units(
+ "Id=wh-t27.service\nTransient=yes\nSlice=app.slice\nMemoryCurrent=2147483648\n\n"
+ "Id=run-u42.service\nTransient=yes\nSlice=app.slice\nMemoryCurrent=[not set]\n\n"
+ "Id=app-org.kde.konsole@65c0.service\nTransient=yes\nSlice=app.slice\nMemoryCurrent=1\n\n"
+ "Id=dbus-:1.1-org.kde.kwalletd6@0.service\nTransient=yes\nSlice=app.slice\nMemoryCurrent=1\n\n"
+ "Id=wf-r-3.service\nTransient=yes\nSlice=agents-jobs.slice\nMemoryCurrent=1\n\n"
+ "Id=pipewire.service\nTransient=no\nSlice=session.slice\nMemoryCurrent=1\n")
+ self.assertEqual(R.bare_units(blocks), [("run-u42.service", None), ("wh-t27.service", 2.0)])
+
+ def test_warning(self):
+ self.assertEqual(R.warning_lines(30.0, 0.0, 0, [("wh-t27.service", 2.0), ("run-u42.service", None)]), [
+ "warning: 2 jobs outside wf res (bare systemd-run): wh-t27.service 2.0 GB, run-u42.service ? GB; "
+ "start jobs with wf res run, also from project scripts"])
+
+
+class Hook(unittest.TestCase):
+ BASH = {"tool_name": "Bash", "tool_input": {"command": "dotnet run --project Desktop", "timeout": 5000}}
+
+ def test_gaming_hides_display(self):
+ out = R.hook_output(self.BASH, T(16, 0), NOW)
+ self.assertEqual(out, {"hookSpecificOutput": {
+ "hookEventName": "PreToolUse",
+ "updatedInput": {"command": "unset DISPLAY WAYLAND_DISPLAY; dotnet run --project Desktop", "timeout": 5000},
+ "additionalContext": "wf res game mode until 16:00: this command has no display, so GUI windows fail. "
+ "Run headless/offscreen, or do other work until game off."}})
+
+ def test_quiet_when_not_gaming_or_not_bash(self):
+ self.assertIsNone(R.hook_output(self.BASH, None, NOW))
+ self.assertIsNone(R.hook_output(self.BASH, T(14, 0), NOW)) # expired
+ self.assertIsNone(R.hook_output({"tool_name": "Read", "tool_input": {"file_path": "x"}}, T(16, 0), NOW))
+ self.assertIsNone(R.hook_output({"tool_name": "Bash", "tool_input": {}}, T(16, 0), NOW))
+
+
+class Litter(unittest.TestCase):
+ def test_victims(self):
+ now, h = 10_000_000.0, 3600
+ entries = [("/t/aB3xYz", "emptydir", now - 3 * h), # old empty dir → go
+ ("/t/MSBuildTemp1000", "emptydir", now - 5 * h), # → go
+ ("/t/fresh", "emptydir", now - 60), # too new
+ ("/t/full", "dir", now - 9 * h), # has content → never
+ ("/t/clr-debug-pipe-111-222-in", "other", now - 60), # pid 111 dead → go
+ ("/t/dotnet-diagnostic-333-444-socket", "other", now), # pid 333 alive
+ ("/t/notes.txt", "file", now - 9 * h)] # file → never
+ alive = {333}.__contains__
+ self.assertEqual(R.litter_victims(entries, now, 2.0, alive),
+ ["/t/MSBuildTemp1000", "/t/aB3xYz", "/t/clr-debug-pipe-111-222-in"])
+class SessionCap(unittest.TestCase):
+ BLOCKS = {
+ "wf-claude-1.scope": {"Slice": "agents.slice", "MemoryHigh": "infinity", "MemoryCurrent": str(GIB)},
+ "wf-claude-2.scope": {"Slice": "agents.slice", "MemoryHigh": str(10 * GIB), "MemoryCurrent": str(GIB)},
+ "run-u7.scope": {"Slice": "agents.slice", "MemoryHigh": str(4 * GIB), "MemoryCurrent": str(GIB)},
+ "podman-pause.scope": {"Slice": "user.slice", "MemoryHigh": "infinity", "MemoryCurrent": str(GIB)},
+ }
+
+ def test_caps_to_set(self):
+ self.assertEqual(R.session_caps(self.BLOCKS, 10.0),
+ [("run-u7.scope", str(10 * GIB)), ("wf-claude-1.scope", str(10 * GIB))])
+ self.assertEqual(R.session_caps(self.BLOCKS, 0.0), # 0 = off → lift caps
+ [("run-u7.scope", "infinity"), ("wf-claude-2.scope", "infinity")])
+
+ def test_warn_at_cap(self):
+ blocks = {"wf-claude-5.scope": {"Slice": "agents.slice", "MemoryCurrent": str(int(9.6 * GIB))},
+ "wf-claude-6.scope": {"Slice": "agents.slice", "MemoryCurrent": str(int(9.8 * GIB))},
+ "wf-claude-7.scope": {"Slice": "agents.slice", "MemoryCurrent": str(int(2 * GIB))}}
+ stall = {"wf-claude-5.scope": 45.0, "wf-claude-6.scope": 1.0, "wf-claude-7.scope": 90.0}
+ self.assertEqual(R.session_cap_lines(blocks, stall, 10.0), [
+ "warning: claude session 5 at its memory cap (9.6 of 10.0 GB, stalled 45% of the last minute): "
+ "run big work with wf res run"])
+ self.assertEqual(R.session_cap_lines(blocks, stall, 0.0), [])
+
+
+def run(title="gate 30735e0", peak=1.0, minutes=10.0, project="proj", mem=12.0, est=60, rc=0, id="r-1"):
+ return {"id": id, "project": project, "title": title, "mem_gb": mem, "est_min": est, "rc": rc,
+ "peak_gb": peak, "min": minutes}
+
+
+TEN = [run(id=f"r-{i}", title=f"gate {i:07x}a", peak=float(i), minutes=6.0 * i) for i in range(1, 11)]
+
+
+class History(unittest.TestCase):
+ def test_title_kind(self):
+ for title, kind in (("gate 30735e0", "gate"), ("gate 2c91cac7 (fix optimizer test)", "gate"),
+ ("mem-profile 16x5M (t-optimizer-x)", "mem-profile 16x5M"), ("wf-batch", "wf-batch"),
+ ("gate 97e6044b direct", "gate 97e6044b direct"), ("build deadbeef", "build deadbeef"),
+ ("gate a3cbde1 0cf9a79", "gate"), ("abc1234", "abc1234"), (" x ", "x")):
+ self.assertEqual(R.title_kind(title), kind, title)
+
+ def test_pct_nearest_rank(self):
+ xs = [5.0, 1.0, 3.0, 2.0, 4.0]
+ self.assertEqual([R.pct(xs, p) for p in (0.5, 0.9, 0.95, 0.2)], [3.0, 5.0, 5.0, 1.0])
+
+ def test_history_record(self):
+ e = running(project="p", title="gate x", mem=10.0, est=40)
+ e.state, e.ended, e.rc, e.peak_gb = "done", T(14, 13, ), 0, 9.1
+ e.ended = e.started + dt.timedelta(minutes=13, seconds=30)
+ self.assertEqual(R.history_record(e), {"id": "r-4", "project": "p", "title": "gate x", "mem_gb": 10.0,
+ "est_min": 40, "rc": 0, "peak_gb": 9.1, "min": 13.5})
+ for change in ({"rc": 1}, {"peak_gb": None}, {"state": "running"}, {"ended": None}):
+ bad = R.Entry(**{**{f.name: getattr(e, f.name) for f in R.fields(R.Entry)}, **change})
+ self.assertIsNone(R.history_record(bad), change)
+
+ def test_suggest_p95_p90(self):
+ # peaks 1..10 → p95 = 10 → ×1.15 = 11.5; durations 6..60 → p90 = 54 → ×1.5 = 81
+ self.assertEqual(R.suggest(TEN, "proj", "gate"), (11.5, 81, 10))
+
+ def test_suggest_floors_and_rounding(self):
+ tiny = [run(id=f"r-{i}", peak=0.01, minutes=1.0) for i in range(3)]
+ self.assertEqual(R.suggest(tiny, "proj", "gate"), (0.2, 5, 3))
+ odd = [run(id=f"r-{i}", peak=1.01, minutes=7.1) for i in range(3)]
+ self.assertEqual(R.suggest(odd, "proj", "gate"), (1.2, 11, 3)) # 1.1615 → 1.2 up; 10.65 → 11 up
+
+ def test_suggest_needs_three_matching(self):
+ two = TEN[:2] + [run(id="r-a", project="other"), run(id="r-b", title="build"),
+ run(id="r-c", rc=1), run(id="r-d", peak=None)]
+ self.assertIsNone(R.suggest(two, "proj", "gate"))
+ self.assertIsNone(R.suggest([], "proj", "gate"))
+
+ def test_hint_line(self):
+ sug = (11.5, 81, 10)
+ line = "hint: history says ~11.5 GB / 81 min (10 runs)"
+ self.assertEqual(R.hint_line(23.1, 60, sug), line)
+ self.assertEqual(R.hint_line(10.0, 163, sug), line)
+ self.assertIsNone(R.hint_line(23.0, 162, sug))
+ self.assertIsNone(R.hint_line(99.0, 999, None))
+
+ def test_merge_runs(self):
+ led = R.Ledger(entries=[running(id="r-4", project="proj", title="gate 1234567"), running(id="r-5")])
+ led.entries[0].state, led.entries[0].rc, led.entries[0].peak_gb = "done", 0, 2.0
+ led.entries[0].ended = led.entries[0].started + dt.timedelta(minutes=3)
+ merged = R.all_runs([run(id="r-1"), run(id="r-4", peak=7.0)], led)
+ self.assertEqual([(r["id"], r["peak_gb"]) for r in merged], [("r-1", 1.0), ("r-4", 7.0)])
+ merged = R.all_runs([run(id="r-1")], led)
+ self.assertEqual([(r["id"], r["peak_gb"], r["min"]) for r in merged], [("r-1", 1.0, 10.0), ("r-4", 2.0, 3.0)])
+
+ def test_hist_lines(self):
+ runs = TEN + [run(id="r-x", title="wf-batch", peak=3.0, minutes=20.0, mem=6.0, est=400),
+ run(id="r-y", project="other", title="build")]
+ self.assertEqual(R.hist_lines(runs, "proj"), [
+ "proj · gate · n 10 · req 12.0 GB · peak 5.0/10.0 GB · est 1h · dur 30m/54m · suggest 11.5 GB 1h21m",
+ "proj · wf-batch · n 1 · req 6.0 GB · peak 3.0/3.0 GB · est 6h40m · dur 20m/20m · suggest - (< 3 runs)"])
+ self.assertEqual(len(R.hist_lines(runs, None)), 3)
+ self.assertEqual(R.hist_lines([], None), ["no finished runs with a peak yet"])
+
+
+ORCH_LOG = """\
+2026-10-06T10:00:00 slow opus t-a done abc1234 10m00s
+2026-10-06T10:10:00 fast haiku t-b handback - 20m30s
+2026-10-06T10:20:00 slow opus t-c done+gate-red abc1234 1h05m
+2026-10-06T10:30:00 fast haiku t-d post-check-red abc1234 50m00s (branch x still there)
+2026-10-06T10:40:00 fast haiku t-e done abc1234 0m00s
+2026-10-06T10:50:00 slow opus t-f done abc1234 -
+2026-10-06T11:00:00 cloud opus t-g done abc1234 3h00m
+2026-10-06T11:10:00 slow sonnet t-h awaiting - 40m00s
+21:30 pilot batch 1 rc=0 done=t-a stopped= added=
+ALERT wf-pilot proj: awaiting a-1
+2026-10-06T11:20:00 slow opus t-i done abc1234 30m00s
+"""
+
+
+class TaskFit(unittest.TestCase):
+ def test_durations_from_log(self):
+ # done/handback rows of local lanes with a real duration only (minutes)
+ self.assertEqual(R.orch_durations(ORCH_LOG), [10.0, 20.5, 65.0, 30.0])
+
+ def test_p90(self):
+ # 4 rows: nearest rank ceil(0.9*4)=4 → 65 min
+ self.assertEqual(R.task_p90(ORCH_LOG), (65, 4))
+ self.assertEqual(R.task_p90(""), (30, 0))
+ two = "2026-10-06T10:00:00 slow opus t-a done abc 5m00s\n" * 2
+ self.assertEqual(R.task_p90(two), (30, 2)) # < 3 runs → default 30 min
+ odd = "2026-10-06T10:00:00 slow opus t-a done abc 4m10s\n" * 3
+ self.assertEqual(R.task_p90(odd), (5, 3)) # rounded up to whole minutes
+
+ def test_batch_fit(self):
+ self.assertEqual(R.batch_fit(4, 150, 30), 4)
+ self.assertEqual(R.batch_fit(4, 90, 30), 3)
+ self.assertEqual(R.batch_fit(4, 59, 30), 1)
+ self.assertEqual(R.batch_fit(4, 29, 30), 0)
+ self.assertEqual(R.batch_fit(4, 0, 30), 0)
diff --git a/tests/test_res_io.py b/tests/test_res_io.py
new file mode 100644
index 0000000..d91046a
--- /dev/null
+++ b/tests/test_res_io.py
@@ -0,0 +1,815 @@
+import datetime as dt
+import io
+import json
+import os
+import subprocess
+import sys
+import tempfile
+import unittest
+from contextlib import redirect_stderr, redirect_stdout
+from pathlib import Path
+
+HERE = Path(__file__).resolve().parent.parent
+sys.path.insert(0, str(HERE))
+import wf_res # noqa: E402
+from wflib import res as R # noqa: E402
+
+UTC = dt.timezone.utc
+GIB_KB = 1024 * 1024
+
+
+class Fake:
+ """Records argv; answers systemctl show from self.units {name: (ActiveState, MemoryCurrent bytes|None)}."""
+
+ def __init__(self):
+ self.calls, self.units, self.fail, self.journal, self.cgroups, self.bare = [], {}, {}, {}, {}, {}
+ self.scopes = {} # name → {prop: value} for agent session scopes
+ self.manager_env = "PATH=/usr/bin\nHOME=/h/u\n"
+ self.git = {} # git subcommand (branch | rev-parse) → stdout
+
+ def __call__(self, argv):
+ self.calls.append(list(argv))
+ key = argv[0] if argv[0] != "systemctl" else argv[2]
+ if key in self.fail:
+ return subprocess.CompletedProcess(argv, 1, "", self.fail[key] + "\n")
+ out = ""
+ if argv[:3] == ["systemctl", "--user", "show"]:
+ blocks = []
+ for name in argv[5:]:
+ if name in self.scopes:
+ blocks.append("\n".join([f"Id={name}"] + [f"{k}={v}" for k, v in self.scopes[name].items()]))
+ continue
+ if name in self.bare:
+ blocks.append(f"Id={name}\nTransient=yes\nSlice=app.slice\nMemoryCurrent={self.bare[name]}")
+ continue
+ state, cur = self.units.get(name, ("inactive", None))
+ blocks.append(f"Id={name}\nActiveState={state}\nMemoryCurrent={'[not set]' if cur is None else cur}"
+ f"\nControlGroup={self.cgroups.get(name, '')}")
+ out = "\n\n".join(blocks) + "\n"
+ elif argv[:3] == ["systemctl", "--user", "list-units"]:
+ names = self.scopes if "--type=scope" in argv else self.bare
+ out = "".join(f"{n} loaded active running x\n" for n in names)
+ elif argv[:3] == ["systemctl", "--user", "set-property"]:
+ if argv[4] in self.scopes:
+ self.scopes[argv[4]].update(a.split("=", 1) for a in argv[5:])
+ elif argv[:3] == ["systemctl", "--user", "show-environment"]:
+ out = self.manager_env
+ elif argv[0] == "git":
+ out = self.git.get(argv[3], "")
+ elif argv[0] == "journalctl":
+ out = self.journal.get(argv[argv.index("-u") + 1], "")
+ elif argv[0] == "systemd-run":
+ unit = next(a for a in argv if a.startswith("--unit="))[len("--unit="):]
+ self.units[unit] = ("active", 0)
+ elif argv[:3] == ["systemctl", "--user", "stop"]:
+ self.units[argv[3]] = ("inactive", None)
+ return subprocess.CompletedProcess(argv, 0, out, "")
+
+ def ran(self, word):
+ return [c for c in self.calls if word in c[:7]]
+
+
+class Box:
+ """Temp dirs + fake machine; .env is a wf_res.Env."""
+
+ def __init__(self, tmp: Path, available_gb=20, total_gb=30, now=dt.datetime(2026, 10, 1, 14, 0, tzinfo=UTC)):
+ self.tmp, self.fake, self.clock, self.alive = tmp, Fake(), [now], {4242}
+ self.available_gb, self.total_gb = available_gb, total_gb
+ (tmp / "proj").mkdir(exist_ok=True)
+ self.env = wf_res.Env(
+ state=tmp / "state", config=tmp / "cfg" / "resources.toml", units=tmp / "units",
+ scratch=tmp / "scratch", run=self.fake,
+ meminfo=lambda: f"MemTotal: {int(self.total_gb * GIB_KB)} kB\nMemAvailable: {int(self.available_gb * GIB_KB)} kB\n",
+ nproc=16, now=lambda: self.clock[0], pid_alive=lambda p: p in self.alive, owner=lambda: 4242,
+ root=lambda: tmp / "proj", cwd=lambda: str(tmp / "proj"), procs=lambda: [],
+ sleep=self.tick, tmp_used_gb=lambda: 0.0, cgread=lambda cg, name: self.psi.get(cg, "") if name == "memory.pressure"
+ else self.stat.get(cg, ""))
+ self.psi, self.stat = {}, {}
+ self.env.live_sessions = lambda: set()
+ self.caller = {"PATH": "/usr/bin", "HOME": "/h/u"}
+ self.env.environ = lambda: dict(self.caller)
+ self.env.tmp_dirs = []
+
+ def tick(self, seconds):
+ self.clock[0] += dt.timedelta(seconds=seconds)
+
+ def wf(self, *argv):
+ out, err = io.StringIO(), io.StringIO()
+ with redirect_stdout(out), redirect_stderr(err):
+ code = wf_res.main(list(argv), self.env)
+ return code, out.getvalue(), err.getvalue()
+
+ def ledger(self):
+ return R.loads((self.env.state / "resources.json").read_text())
+
+
+class IOBase(unittest.TestCase):
+ def setUp(self):
+ self._tmp = tempfile.TemporaryDirectory()
+ self.box = Box(Path(self._tmp.name))
+
+ def tearDown(self):
+ self._tmp.cleanup()
+
+
+class Run(IOBase):
+ def test_run_starts_unit(self):
+ code, out, err = self.box.wf("run", "--mem", "4G", "--for", "40m", "--title", "build", "--", "make", "-j8")
+ self.assertEqual((code, err), (0, ""))
+ log = self.box.env.state / "logs" / "r-1.log"
+ self.assertEqual(out, f"r-1 started; log {log}; ETA ~14:40 (estimate: not killed when over)\n")
+ argv = self.box.fake.ran("systemd-run")[0]
+ self.assertEqual(argv[-3:], ["sh", "make", "-j8"])
+ self.assertIn("--unit=wf-r-1.service", argv)
+ e = self.box.ledger().get("r-1")
+ self.assertEqual((e.state, e.project, e.owner, e.cwd), ("running", "proj", 4242, str(self.box.tmp / "proj")))
+ self.assertTrue((self.box.env.units / "agents.slice").exists())
+
+ def test_run_keeps_caller_env(self):
+ self.box.caller.update({"APP_DIR": "/data/arc", "PATH": "/venv/bin:/usr/bin", "PWD": "/x", "SHLVL": "2",
+ "CLAUDE_CODE_MESSAGING_TOKEN": "t", "GH_TOKEN": "t", "DB_PASSWORD": "p"})
+ self.box.wf("run", "--mem", "1G", "--for", "1m", "--title", "x", "--", "x")
+ argv = self.box.fake.ran("systemd-run")[0]
+ self.assertEqual([a for a in argv if a.startswith("--setenv=")],
+ ["--setenv=APP_DIR=/data/arc", "--setenv=PATH=/venv/bin:/usr/bin", "--setenv=WF_RES_ID=r-1"])
+ self.assertEqual(self.box.ledger().get("r-1").env, {"APP_DIR": "/data/arc", "PATH": "/venv/bin:/usr/bin"})
+
+ def test_run_records_owner(self):
+ main = self.box.tmp / "main"
+ (main / ".wf" / "sessions").mkdir(parents=True)
+ (main / ".wf" / "sessions" / "slow.json").write_text('{"lane": "slow", "pid": 77, "socket": "/s/o"}\n')
+ self.box.fake.git = {"branch": "slow/t-big\n", "rev-parse": f"{main}/.git\n"}
+ self.box.caller.update({"CLAUDE_PID": "77", "CLAUDE_CODE_MESSAGING_SOCKET": "/s/o", "WF_RES_ID": "r-9"})
+ self.box.wf("run", "--mem", "1G", "--for", "1m", "--title", "x", "--", "x")
+ self.assertEqual(self.box.ledger().get("r-1").by,
+ {"name": "slow session", "task": "t-big", "batch": "r-9", "address": "uds:/s/o"})
+ argv = self.box.fake.ran("systemd-run")[0]
+ self.assertEqual([a for a in argv if a.startswith("--setenv=WF_RES_ID")], ["--setenv=WF_RES_ID=r-1"])
+ self.assertIn('"x" running', self.box.wf("status")[1])
+ self.assertIn("[by slow session t-big batch r-9, message uds:/s/o]", self.box.wf("status")[1])
+ self.box.wf("note", "--mem", "1G", "--for", "5m", "--by", "pilot", "edit")
+ self.assertEqual(self.box.ledger().get("r-2").by["name"], "pilot")
+
+ def test_queued_run_starts_later_with_its_env(self):
+ self.box.wf("run", "--mem", "10G", "--for", "40m", "--title", "big", "--", "x")
+ self.box.caller["FOO"] = "1"
+ self.box.wf("run", "--mem", "8G", "--for", "10m", "--title", "q", "--queue", "--", "y")
+ self.box.caller.pop("FOO")
+ self.box.wf("release", "r-1", "--stop")
+ self.box.wf("status")
+ starts = self.box.fake.ran("systemd-run")
+ self.assertEqual([[a for a in s if a.startswith("--setenv=")] for s in starts], [["--setenv=WF_RES_ID=r-1"], ["--setenv=FOO=1", "--setenv=WF_RES_ID=r-2"]])
+
+ def test_busy_exit_3(self):
+ self.box.wf("run", "--mem", "10G", "--for", "40m", "--title", "big", "--", "x")
+ code, out, _ = self.box.wf("run", "--mem", "8G", "--for", "10m", "--title", "two", "--", "y")
+ # 20 − 6 − 2 − 10 unused = 2.0 free
+ self.assertEqual(code, 3)
+ self.assertEqual(out, 'busy: 10.0 GB held by proj "big" (r-1) until ~14:40; 2.0 GB free for agents; '
+ 'retry after ~14:40 or work on something else; 12.0 GB really free beyond the reserve: '
+ '--force starts it past the ledger (only if the holders will not use what they reserved)\n')
+ self.assertEqual([e.id for e in self.box.ledger().entries], ["r-1"])
+
+ def test_force_runs_past_unused_claim(self):
+ self.box.wf("run", "--mem", "10G", "--for", "40m", "--title", "big", "--", "x")
+ code, out, _ = self.box.wf("run", "--mem", "8G", "--for", "10m", "--title", "two", "--force", "--", "y")
+ # really free: 20 − 6 reserve − 2 headroom = 12 ≥ 8 (r-1's unused 10 GB ignored)
+ self.assertEqual(code, 0)
+ self.assertTrue(out.startswith("r-2 started"), out)
+ self.assertEqual([(e.id, e.state) for e in self.box.ledger().entries], [("r-1", "running"), ("r-2", "running")])
+
+ def test_force_refused_when_memory_really_used(self):
+ self.box.available_gb = 12 # really free beyond reserve: 12 − 6 − 2 = 4
+ self.box.wf("run", "--mem", "3G", "--for", "40m", "--title", "a", "--", "x")
+ code, out, _ = self.box.wf("run", "--mem", "8G", "--for", "10m", "--title", "two", "--", "y")
+ self.assertEqual(code, 3)
+ self.assertNotIn("--force", out) # hint only when --force would fit
+ code, out, _ = self.box.wf("run", "--mem", "8G", "--for", "10m", "--title", "two", "--force", "--", "y")
+ self.assertEqual((code, out), (3, "busy even with --force: only 4.0 GB really free beyond the reserve; "
+ "retry later or work on something else\n"))
+ self.assertEqual([e.id for e in self.box.ledger().entries], ["r-1"])
+
+ def test_never_fits(self):
+ code, _, err = self.box.wf("run", "--mem", "25G", "--for", "1h", "--title", "x", "--", "x")
+ self.assertEqual((code, err), (1, "wf: 25.0 GB can never fit (max 22.0 GB for agents)\n"))
+
+ def test_systemd_run_failure(self):
+ self.box.fake.fail["systemd-run"] = "Failed to start transient service unit: boom"
+ code, _, err = self.box.wf("run", "--mem", "1G", "--for", "1m", "--title", "x", "--", "x")
+ self.assertEqual((code, err), (1, "wf: systemd-run: Failed to start transient service unit: boom\n"))
+ self.assertEqual(self.box.ledger().entries, [])
+
+ def test_no_command(self):
+ code, _, err = self.box.wf("run", "--mem", "1G", "--for", "1m", "--title", "x")
+ self.assertEqual((code, err), (1, "wf: no command after --\n"))
+
+ def test_usage_error_exit_2(self):
+ with redirect_stderr(io.StringIO()):
+ self.assertEqual(self.box.wf("run", "--for", "1m")[0], 2)
+
+
+class Status(IOBase):
+ def test_status_and_exit_pruned(self):
+ self.box.wf("run", "--mem", "4G", "--for", "40m", "--title", "build", "--", "make")
+ self.box.fake.units["wf-r-1.service"] = ("active", 2 * 1024 ** 3)
+ code, out, _ = self.box.wf("status")
+ self.assertEqual(out.splitlines()[0], 'r-1 proj "build" running 4.0 GB used 2.0 GB 1 cpu since 14:00 ETA ~14:40')
+ logs = self.box.env.state / "logs"
+ (logs / "r-1.rc").write_text("0\n")
+ (logs / "r-1.peak").write_text(str(3 * 1024 ** 3) + "\n")
+ self.box.fake.units["wf-r-1.service"] = ("inactive", None)
+ code, out, _ = self.box.wf("status")
+ self.assertEqual(out.splitlines()[0], 'r-1 proj "build" done (exited) rc=0 peak 3.0 GB at 14:00')
+
+ def test_status_warns_throttled(self):
+ self.box.wf("run", "--mem", "4G", "--for", "40m", "--title", "build", "--", "make")
+ self.box.fake.units["wf-r-1.service"] = ("active", int(3.8 * 1024 ** 3))
+ self.box.fake.cgroups["wf-r-1.service"] = "/x/wf-r-1.service"
+ self.box.psi["/x/wf-r-1.service"] = "some avg10=50.00 avg60=40.00 avg300=10.00 total=1\n"
+ out = self.box.wf("status")[1]
+ self.assertIn("r-1 throttled at its memory limit (3.8 of 4.0 GB, stalled 40% of the last minute): "
+ "likely too small; `wf res release r-1 --stop` and re-run with a bigger --mem\n", out)
+
+ def test_status_warns_bare_units(self):
+ self.box.fake.bare["wh-t27.service"] = 2 * 1024 ** 3
+ out = self.box.wf("status")[1]
+ self.assertIn("warning: 1 jobs outside wf res (bare systemd-run): wh-t27.service 2.0 GB; "
+ "start jobs with wf res run, also from project scripts\n", out)
+ self.assertIn(["systemctl", "--user", "list-units", "--type=service", "--state=running", "--no-legend",
+ "--plain"], self.box.fake.calls)
+
+ def test_status_splits_slice_cache(self):
+ self.box.fake.units["agents.slice"] = ("active", 5 * 1024 ** 3)
+ self.box.fake.cgroups["agents.slice"] = "/a.slice"
+ self.box.stat["/a.slice"] = f"anon {3 * 1024 ** 3}\nfile {2 * 1024 ** 3}\nshmem {1024 ** 3 // 2}\n"
+ self.assertIn("unreserved agent memory 5.0 GB (1.5 GB of it file cache, reclaimable)\n", self.box.wf("status")[1])
+
+ def test_status_id_and_json(self):
+ self.box.wf("run", "--mem", "4G", "--for", "40m", "--title", "build", "--", "make", "a b")
+ out = self.box.wf("status", "r-1")[1]
+ self.assertIn("cmd: make 'a b'", out)
+ self.assertEqual(json.loads(self.box.wf("status", "--json")[1])["entries"][0]["id"], "r-1")
+
+ def test_show_failure_keeps_ledger(self):
+ self.box.wf("run", "--mem", "4G", "--for", "40m", "--title", "build", "--", "make")
+ self.box.fake.fail["show"] = "Failed to connect to bus"
+ code, _, err = self.box.wf("status")
+ self.assertEqual((code, err), (1, "wf: systemctl show: Failed to connect to bus\n"))
+ self.assertEqual(self.box.ledger().get("r-1").state, "running")
+
+ def test_corrupt_ledger_moved_aside(self):
+ self.box.env.state.mkdir(parents=True)
+ (self.box.env.state / "resources.json").write_text("{oops")
+ code, out, err = self.box.wf("status")
+ self.assertEqual(code, 0)
+ self.assertRegex(err, r"^wf: corrupt ledger: .*; moved to resources\.json\.bad-20261001-140000, starting empty\n$")
+ self.assertTrue((self.box.env.state / "resources.json.bad-20261001-140000").exists())
+ self.assertEqual(out.splitlines()[0], "no reservations")
+
+
+class LedgerCompat(IOBase):
+ """Readers of another code version (a long `wf res wait` started before a release) must keep entries."""
+
+ def _write(self, extra: dict, drop=()):
+ self.box.wf("run", "--mem", "1G", "--for", "5m", "--title", "t", "--", "x")
+ path = self.box.env.state / "resources.json"
+ d = json.loads(path.read_text())
+ for e in d["entries"]:
+ for k in drop:
+ e.pop(k, None)
+ e.update(extra)
+ path.write_text(json.dumps(d))
+ return path
+
+ def _kept(self, path):
+ for argv in (("status",), ("tick",), ("wait", "r-1", "--timeout", "1m")):
+ self.box.wf(*argv)
+ self.assertEqual(self.box.ledger().get("r-1").state, "running", argv)
+ self.assertEqual(sorted(p.name for p in path.parent.glob("resources.json*")), ["resources.json"])
+ self.assertIn("r-1", self.box.wf("status")[1])
+
+ def test_pre_owner_ledger_kept(self): # written before e84d7ca/9533211: no by field
+ self._kept(self._write({}, drop=("by", "env")))
+
+ def test_future_fields_kept(self): # written by newer code: unknown keys
+ self._kept(self._write({"by": {"name": "x"}, "added_later": 1}))
+
+ def test_corrupt_ledger_keeps_id_counter(self):
+ self.box.env.state.mkdir(parents=True)
+ (self.box.env.state / "resources.json").write_text('{"next": 652, "entries": [{"id": "r-651"}]}')
+ self.box.wf("run", "--mem", "1G", "--for", "5m", "--title", "t", "--", "x")
+ self.assertEqual([e.id for e in self.box.ledger().entries], ["r-652"])
+
+ def test_start_clears_stale_results(self):
+ logs = self.box.env.state / "logs"
+ logs.mkdir(parents=True)
+ (logs / "r-1.rc").write_text("1\n")
+ (logs / "r-1.peak").write_text("5\n")
+ self.box.wf("run", "--mem", "1G", "--for", "5m", "--title", "t", "--", "x")
+ self.assertFalse((logs / "r-1.rc").exists() or (logs / "r-1.peak").exists())
+
+
+class HookIO(IOBase):
+ def hook(self, text):
+ self.box.env.stdin = lambda: text
+ return self.box.wf("hook")
+
+ def test_game_on_rewrites_bash(self):
+ self.box.wf("game", "on", "--for", "1h")
+ code, out, err = self.hook('{"tool_name": "Bash", "tool_input": {"command": "./app"}}')
+ self.assertEqual((code, err), (0, ""))
+ self.assertEqual(json.loads(out)["hookSpecificOutput"]["updatedInput"],
+ {"command": "unset DISPLAY WAYLAND_DISPLAY; ./app"})
+ calls = len(self.box.fake.calls)
+ self.hook('{"tool_name": "Bash", "tool_input": {"command": "./app"}}')
+ self.assertEqual(len(self.box.fake.calls), calls) # read-only: no systemctl, no prune
+
+ def test_silent_otherwise(self):
+ self.assertEqual(self.hook('{"tool_name": "Bash", "tool_input": {"command": "./app"}}'), (0, "", ""))
+ self.box.wf("game", "on")
+ self.assertEqual(self.hook("not json"), (0, "", ""))
+ (self.box.env.state / "resources.json").write_text("{oops")
+ self.assertEqual(self.hook('{"tool_name": "Bash", "tool_input": {"command": "x"}}'), (0, "", ""))
+ self.assertTrue((self.box.env.state / "resources.json").exists()) # hook never moves a bad ledger
+
+
+class Release(IOBase):
+ def test_release_running_needs_stop(self):
+ self.box.wf("run", "--mem", "4G", "--for", "40m", "--title", "build", "--", "make")
+ code, _, err = self.box.wf("release", "r-1")
+ self.assertEqual((code, err), (1, "wf: r-1 is running; --stop to kill it\n"))
+ code, out, _ = self.box.wf("release", "r-1", "--stop")
+ self.assertEqual((code, out), (0, "r-1 stopped\n"))
+ self.assertEqual(self.box.fake.ran("stop")[0], ["systemctl", "--user", "stop", "wf-r-1.service"])
+ e = self.box.ledger().get("r-1")
+ self.assertEqual((e.state, e.why), ("done", "released"))
+
+ def test_release_unknown(self):
+ self.assertEqual(self.box.wf("release", "r-7")[2], "wf: no entry 'r-7'\n")
+
+
+class Dispatch(unittest.TestCase):
+ def test_wf_forwards_res(self):
+ r = subprocess.run([sys.executable, str(HERE / "wf.py"), "res", "-h"], capture_output=True, text=True)
+ self.assertEqual(r.returncode, 0)
+ self.assertIn("usage: wf res", r.stdout)
+
+class Wait(IOBase):
+ def test_wait_until_done(self):
+ self.box.wf("run", "--mem", "1G", "--for", "5m", "--title", "t", "--", "x")
+ logs = self.box.env.state / "logs"
+ polls = []
+
+ def sleep(seconds):
+ polls.append(seconds)
+ self.box.tick(seconds)
+ if len(polls) == 2:
+ (logs / "r-1.rc").write_text("0\n")
+ (logs / "r-1.peak").write_text(str(1024 ** 3 // 2) + "\n")
+ self.box.fake.units["wf-r-1.service"] = ("inactive", None)
+ self.box.env.sleep = sleep
+ code, out, _ = self.box.wf("wait", "r-1")
+ self.assertEqual((code, out, polls), (0, "r-1 done rc=0 peak 0.5 GB in 0 min\n", [15, 15]))
+
+ def test_wait_warns_throttled_once(self):
+ self.box.wf("run", "--mem", "4G", "--for", "5m", "--title", "t", "--", "x")
+ self.box.fake.units["wf-r-1.service"] = ("active", int(3.8 * 1024 ** 3))
+ self.box.fake.cgroups["wf-r-1.service"] = "/x/wf-r-1.service"
+ self.box.psi["/x/wf-r-1.service"] = "some avg10=50.00 avg60=40.00 avg300=10.00 total=1\n"
+ code, _, err = self.box.wf("wait", "r-1", "--timeout", "1m")
+ self.assertEqual(err, "wf: r-1 throttled at its memory limit (3.8 of 4.0 GB, stalled 40% of the last minute): "
+ "likely too small; `wf res release r-1 --stop` and re-run with a bigger --mem\n"
+ "wf: r-1 not done after 1m (running)\n")
+
+ def test_wait_timeout(self):
+ self.box.wf("run", "--mem", "1G", "--for", "5m", "--title", "t", "--", "x")
+ code, _, err = self.box.wf("wait", "r-1", "--timeout", "1m")
+ self.assertEqual((code, err), (1, "wf: r-1 not done after 1m (running)\n"))
+
+ def test_wait_killed_job(self):
+ self.box.wf("run", "--mem", "1G", "--for", "5m", "--title", "t", "--", "x")
+ self.box.fake.units["wf-r-1.service"] = ("failed", None)
+ self.assertEqual(self.box.wf("wait", "r-1")[1],
+ "r-1 done rc=? peak ? in 0 min (no exit code; see journalctl --user -u wf-r-1.service)\n")
+
+ def test_wait_oom_killed_job(self):
+ self.box.wf("run", "--mem", "4G", "--for", "5m", "--title", "t", "--", "x")
+ self.box.fake.units["wf-r-1.service"] = ("failed", None)
+ self.box.fake.journal["wf-r-1.service"] = (
+ "wf-r-1.service: systemd-oomd killed some process(es) in this unit.\n"
+ "wf-r-1.service: Failed with result 'oom-kill'.\n"
+ "wf-r-1.service: Consumed 11min 2.837s CPU time, 3.9G memory peak.\n")
+ self.assertEqual(self.box.wf("wait", "r-1")[1], "r-1 done rc=? peak 3.9 GB in 0 min "
+ "(killed: oom-kill by systemd-oomd, limit 4.0 GB; raise --mem)\n")
+ j = self.box.fake.ran("journalctl")[0]
+ self.assertEqual(j, ["journalctl", "--user", "-u", "wf-r-1.service", "-o", "cat", "--no-pager",
+ "--since", "@" + str(int(dt.datetime(2026, 10, 1, 14, 0, tzinfo=UTC).timestamp()))])
+ out = self.box.wf("status")[1]
+ self.assertEqual(out.splitlines()[0], 'r-1 proj "t" done (killed: oom-kill by systemd-oomd, limit 4.0 GB; '
+ 'raise --mem) rc=? peak 3.9 GB at 14:00')
+
+ def test_rc_file_skips_journal(self):
+ self.box.wf("run", "--mem", "1G", "--for", "5m", "--title", "t", "--", "x")
+ (self.box.env.state / "logs" / "r-1.rc").write_text("2\n")
+ self.box.fake.units["wf-r-1.service"] = ("inactive", None)
+ self.assertEqual(self.box.wf("wait", "r-1")[1], "r-1 done rc=2 peak ? in 0 min\n")
+ self.assertEqual(self.box.fake.ran("journalctl"), [])
+
+class QueueIO(IOBase):
+ def test_queue_then_start_when_free(self):
+ self.box.wf("run", "--mem", "10G", "--for", "40m", "--title", "big", "--", "x")
+ code, out, _ = self.box.wf("run", "--mem", "8G", "--for", "10m", "--title", "two", "--queue", "--", "y")
+ self.assertEqual((code, out), (0, "r-2 queued, position 1; est. start ~14:40; cancel: wf res release r-2\n"))
+ self.box.fake.units["wf-r-1.service"] = ("inactive", None)
+ self.box.wf("status")
+ e = self.box.ledger().get("r-2")
+ self.assertEqual((e.state, e.unit), ("running", "wf-r-2.service"))
+ self.assertEqual(self.box.fake.ran("systemd-run")[-1][-2:], ["sh", "y"])
+
+ def test_direct_run_does_not_jump_queue(self):
+ self.box.wf("run", "--mem", "10G", "--for", "40m", "--title", "big", "--", "x")
+ self.box.wf("run", "--mem", "8G", "--for", "10m", "--title", "two", "--queue", "--", "y")
+ code, out, _ = self.box.wf("run", "--mem", "1G", "--for", "10m", "--title", "small", "--", "z")
+ self.assertEqual(code, 3)
+
+ def test_queued_start_failure_moves_on(self):
+ self.box.wf("run", "--mem", "10G", "--for", "40m", "--title", "big", "--", "x")
+ self.box.wf("run", "--mem", "8G", "--for", "10m", "--title", "two", "--queue", "--", "y")
+ self.box.fake.units["wf-r-1.service"] = ("inactive", None)
+ self.box.fake.fail["systemd-run"] = "boom"
+ code, out, err = self.box.wf("status")
+ self.assertEqual(code, 0)
+ e = self.box.ledger().get("r-2")
+ self.assertEqual((e.state, e.why), ("done", "start failed: systemd-run: boom"))
+ self.box.fake.fail.clear()
+ self.assertEqual(self.box.wf("status")[0], 0)
+
+class Note(IOBase):
+ def test_note_and_owner_exit(self):
+ code, out, _ = self.box.wf("note", "--mem", "3G", "--for", "20m", "dotnet test")
+ self.assertEqual((code, out), (0, "r-1 noted 3.0 GB until ~14:20 (then freed; nothing is killed)\n"))
+ self.assertEqual(self.box.ledger().get("r-1").owner, 4242)
+ self.box.alive.clear()
+ self.box.wf("status")
+ self.assertEqual(self.box.ledger().get("r-1").why, "owner gone")
+
+ def test_note_busy(self):
+ self.box.wf("note", "--mem", "10G", "--for", "20m", "a")
+ code, out, _ = self.box.wf("note", "--mem", "3G", "--for", "20m", "b")
+ self.assertEqual((code, out), (3, 'busy: 10.0 GB held by proj "a" (r-1) until ~14:20; 2.0 GB free for agents; '
+ 'retry after ~14:20 or work on something else; 12.0 GB really free '
+ 'beyond the reserve: --force starts it past the ledger (only if the '
+ 'holders will not use what they reserved)\n'))
+ code, out, _ = self.box.wf("note", "--mem", "3G", "--for", "20m", "--force", "b")
+ self.assertEqual(code, 0)
+ self.assertTrue(out.startswith("r-2 noted 3.0 GB"), out)
+
+
+def _race_child(tmp, q):
+ import time as _t
+ box = Box(Path(tmp), available_gb=14) # budget 14 − 6 − 2 = 6 GB: one 5G note fits, not two
+ slow = box.env.meminfo
+ box.env.meminfo = lambda: (_t.sleep(0.3), slow())[1]
+ with redirect_stdout(io.StringIO()), redirect_stderr(io.StringIO()):
+ q.put(wf_res.main(["note", "--mem", "5G", "--for", "10m", "x"], box.env))
+
+
+class LockRace(unittest.TestCase):
+ def test_two_processes_never_overbook(self):
+ import multiprocessing as mp
+ ctx = mp.get_context("fork")
+ with tempfile.TemporaryDirectory() as tmp:
+ q = ctx.Queue()
+ ps = [ctx.Process(target=_race_child, args=(tmp, q)) for _ in range(2)]
+ for p in ps:
+ p.start()
+ for p in ps:
+ p.join(10)
+ self.assertEqual([p.exitcode for p in ps], [0, 0]) # a crashed child must fail, not hang q.get()
+ self.assertEqual(sorted([q.get(timeout=5), q.get(timeout=5)]), [0, 3])
+
+class GameIO(IOBase):
+ def props(self):
+ return [c[5:] for c in self.box.fake.ran("set-property")]
+
+ def test_on_off(self):
+ code, out, _ = self.box.wf("game", "on")
+ self.assertEqual((code, out), (0, "game on until 18:00 (4h); CPU/IO now yours\n"))
+ self.assertEqual(self.props()[-1], ["CPUWeight=5", "IOWeight=5", f"MemoryHigh={18 * 1024 ** 3}"])
+ self.assertEqual(self.box.fake.ran("set-property")[-1][:5],
+ ["systemctl", "--user", "set-property", "--runtime", "agents.slice"])
+ self.assertEqual(self.box.wf("game", "off")[1], "game off\n")
+ self.assertEqual(self.props()[-1], ["CPUWeight=20", "IOWeight=20", f"MemoryHigh={24 * 1024 ** 3}"])
+
+ def test_expiry_restores(self):
+ self.box.wf("game", "on", "--for", "1h")
+ self.box.tick(3601)
+ self.box.wf("status")
+ self.assertIsNone(self.box.ledger().game_until)
+ self.assertEqual(self.props()[-1][0], "CPUWeight=20")
+
+ def test_game_refuses_new_job(self):
+ self.box.wf("game", "on")
+ # 20 − 12 − 2 = 6 GB budget
+ self.assertEqual(self.box.wf("run", "--mem", "7G", "--for", "5m", "--title", "x", "--", "x")[0], 3)
+
+ def test_on_again_extends(self):
+ self.box.wf("game", "on", "--for", "2h")
+ self.box.wf("game", "on", "--for", "1h")
+ self.assertEqual(self.box.ledger().game_until.hour, 16)
+
+
+class CleanIO(IOBase):
+ def make(self, rel, age_h, size=10):
+ p = self.box.env.scratch / rel
+ p.parent.mkdir(parents=True, exist_ok=True)
+ p.write_bytes(b"x" * size)
+ t = self.box.clock[0].timestamp() - age_h * 3600
+ os.utime(p, (t, t))
+ for d in [p.parent, *p.parent.parents]:
+ if d == self.box.env.scratch.parent:
+ break
+ os.utime(d, (t, t))
+ return p
+
+ def test_auto_clean_sweeps_tmp_litter(self):
+ t = self.box.tmp / "tmpdir"
+ (t / "old-empty").mkdir(parents=True)
+ (t / "full").mkdir()
+ (t / "full" / "x").write_text("x")
+ (t / "clr-debug-pipe-999999-1-in").write_text("")
+ (t / "keep.txt").write_text("x")
+ old = self.box.clock[0].timestamp() - 3 * 3600
+ for n in ("old-empty", "full", "keep.txt"):
+ os.utime(t / n, (old, old))
+ self.box.env.tmp_dirs = [t]
+ self.box.env.pid_alive = lambda p: p != 999999 and p in self.box.alive
+ self.box.wf("status")
+ self.assertEqual(sorted(os.listdir(t)), ["full", "keep.txt"])
+
+ def test_auto_clean_keeps_live_session(self):
+ live = self.make("-projects-a/live-1/scratchpad/notes.txt", 9)
+ dead = self.make("-projects-a/dead-2/scratchpad/x.txt", 9)
+ self.box.env.live_sessions = lambda: {"live-1"}
+ self.box.wf("status")
+ self.assertEqual((live.exists(), dead.exists()), (True, False))
+
+ def test_auto_clean_on_status_and_throttle(self):
+ old = self.make("-projects-a/s1/scratchpad/big.bin", 3)
+ new = self.make("-projects-b/s2/scratchpad/x.txt", 0.5)
+ self.box.wf("status")
+ self.assertFalse(old.exists())
+ self.assertTrue(new.exists())
+ self.assertTrue(any(c[2] == "reset-failed" for c in self.box.fake.calls if c[0] == "systemctl"))
+ again = self.make("-projects-c/s3/a.txt", 3)
+ self.box.wf("status")
+ self.assertTrue(again.exists()) # < 10 min since last clean
+ self.box.tick(601)
+ self.box.wf("status")
+ self.assertFalse(again.exists())
+
+ def test_clean_prints_and_lists_project_patterns(self):
+ self.make("-projects-a/s1/f.bin", 3, size=2048)
+ proj = self.box.tmp / "proj"
+ (proj / "workflow.toml").write_text('cleanup = ["out/prof", "out/logs/*.log:30d"]\n')
+ (proj / "out" / "prof").mkdir(parents=True)
+ (proj / "out" / "prof" / "p.dat").write_bytes(b"x" * 1024)
+ logs = proj / "out" / "logs"
+ logs.mkdir()
+ (logs / "new.log").write_text("n")
+ old = logs / "old.log"
+ old.write_text("o")
+ t = self.box.clock[0].timestamp() - 31 * 86400
+ os.utime(old, (t, t))
+ (logs / "link.log").symlink_to("/etc/hostname")
+ code, out, _ = self.box.wf("clean")
+ self.assertEqual(out.splitlines(), [
+ f"freed 0 MB: {self.box.env.scratch / '-projects-a'}",
+ "would delete 0 MB: out/prof (wf res clean --yes)",
+ "would delete 0 MB: out/logs/old.log (wf res clean --yes)"])
+ self.assertTrue((proj / "out" / "prof").exists())
+ self.box.wf("clean", "--yes")
+ self.assertFalse((proj / "out" / "prof").exists())
+ self.assertFalse(old.exists())
+ self.assertTrue((logs / "new.log").exists())
+ self.assertTrue((logs / "link.log").is_symlink())
+
+ def test_status_warnings(self):
+ self.box.env.tmp_used_gb = lambda: 8.0
+ self.box.env.procs = lambda: [(10, 1, "claude", "/u/app.slice/tab.scope"),
+ (20, 1, "claude", "/u/agents.slice/wf-claude-20.scope")]
+ out = self.box.wf("status")[1].splitlines()
+ self.assertEqual(out[-2:], ["warning: /tmp (RAM) holds 8.0 GB; wf res clean",
+ "warning: 1 claude sessions outside agents.slice (wf res adopt)"])
+
+class TimerIO(IOBase):
+ def test_timer_on_off(self):
+ code, out, _ = self.box.wf("timer", "on")
+ self.assertEqual((code, out), (0, "timer on: wf-res.timer every 1 min\n"))
+ units = self.box.env.units
+ self.assertIn("res tick", (units / "wf-res.service").read_text())
+ self.assertTrue((units / "wf-res.timer").exists())
+ self.assertIn(["systemctl", "--user", "enable", "--now", "wf-res.timer"], self.box.fake.calls)
+ self.assertEqual(self.box.wf("timer", "off")[1], "timer off\n")
+ self.assertFalse((units / "wf-res.timer").exists())
+ self.assertIn(["systemctl", "--user", "disable", "--now", "wf-res.timer"], self.box.fake.calls)
+
+ def test_adopt(self):
+ self.box.env.procs = lambda: [(101, 1, "claude", "/u/app.slice/t.scope"), (102, 101, "bash", "/u/app.slice/t.scope"),
+ (201, 1, "claude", "/u/app.slice/u.scope")]
+ code, out, _ = self.box.wf("adopt")
+ self.assertEqual(out, "adopted claude 101 (2 processes)\nadopted claude 201 (1 processes)\n")
+ self.assertEqual(self.box.fake.ran("StartTransientUnit")[0][8], "wf-claude-101.scope")
+
+ def test_adopt_failure_reported(self):
+ self.box.env.procs = lambda: [(101, 1, "claude", "/u/app.slice/t.scope")]
+ self.box.fake.fail["busctl"] = "Call failed: No such process"
+ self.assertEqual(self.box.wf("adopt")[1], "claude 101: Call failed: No such process\n")
+
+ def test_adopt_none(self):
+ self.assertEqual(self.box.wf("adopt")[1], "all claude sessions already in agents.slice\n")
+
+ def test_tick_adopts_silently(self):
+ self.box.env.procs = lambda: [(101, 1, "claude", "/u/app.slice/t.scope")]
+ self.assertEqual(self.box.wf("tick"), (0, "", ""))
+ self.assertEqual(len(self.box.fake.ran("StartTransientUnit")), 1)
+
+ def test_tick_caps_sessions(self):
+ self.box.fake.scopes = {"wf-claude-5.scope": {"Slice": "agents.slice", "MemoryHigh": "infinity",
+ "MemoryCurrent": str(1024 ** 3), "ControlGroup": "/a/5"},
+ "init.scope": {"Slice": "-.slice", "MemoryHigh": "infinity", "MemoryCurrent": "1"}}
+ self.assertEqual(self.box.wf("tick"), (0, "", ""))
+ self.assertEqual(self.box.fake.ran("set-property"),
+ [["systemctl", "--user", "set-property", "--runtime", "wf-claude-5.scope",
+ f"MemoryHigh={6 * 1024 ** 3}"]])
+ self.box.wf("tick")
+ self.assertEqual(len(self.box.fake.ran("set-property")), 1) # already capped → no call
+
+ def test_status_warns_session_at_cap(self):
+ self.box.fake.scopes = {"wf-claude-5.scope": {"Slice": "agents.slice", "MemoryHigh": str(6 * 1024 ** 3),
+ "MemoryCurrent": str(int(5.7 * 1024 ** 3)), "ControlGroup": "/a/5"}}
+ self.box.psi["/a/5"] = "some avg10=50.00 avg60=40.00 avg300=10.00 total=1\n"
+ self.assertIn("warning: claude session 5 at its memory cap (5.7 of 6.0 GB, stalled 40% of the last minute): "
+ "run big work with wf res run\n", self.box.wf("status")[1])
+
+ def test_tick_silent_and_starts_queue(self):
+ self.box.wf("run", "--mem", "10G", "--for", "40m", "--title", "big", "--", "x")
+ self.box.wf("run", "--mem", "8G", "--for", "10m", "--title", "two", "--queue", "--", "y")
+ self.box.fake.units["wf-r-1.service"] = ("inactive", None)
+ self.assertEqual(self.box.wf("tick"), (0, "", ""))
+ self.assertEqual(self.box.ledger().get("r-2").state, "running")
+
+ def test_shell_init(self):
+ self.assertEqual(self.box.wf("shell-init")[1],
+ "alias claude='systemd-run --user --scope --quiet --slice=agents.slice claude'\n")
+
+
+class Robust(IOBase):
+ def test_vanished_paths_are_skipped(self):
+ gone = self.box.tmp / "gone"
+ self.assertEqual((wf_res._newest(gone), wf_res._size(gone)), (0.0, 0))
+
+ def test_os_error_is_one_line(self):
+ def boom():
+ raise PermissionError(13, "Permission denied", "/proc/meminfo")
+ self.box.env.meminfo = boom
+ code, _, err = self.box.wf("status")
+ self.assertEqual((code, err), (1, "wf: [Errno 13] Permission denied: '/proc/meminfo'\n"))
+
+
+if __name__ == "__main__":
+ unittest.main()
+
+
+class LockIO(IOBase):
+ """--lock / 'gate …' titles: one job per lock key and main tree at a time (shared gate checkout)."""
+
+ def gate(self, title, *extra):
+ return self.box.wf("run", "--mem", "2G", "--for", "30m", "--title", title, *extra, "--", "x")
+
+ def test_second_gate_refused_even_with_force(self):
+ self.assertEqual(self.gate("gate c61907f0")[0], 0)
+ for extra in ((), ("--force",)):
+ code, out, err = self.gate("gate 2c91cac7 (fix)", *extra)
+ self.assertEqual(code, 3, err)
+ self.assertIn("lock 'gate' held by r-1", out + err)
+ self.assertIn("--queue", out + err)
+ self.assertEqual(len(self.box.fake.ran("systemd-run")), 1)
+
+ def test_second_gate_queues_and_starts_after_first(self):
+ self.gate("gate c61907f0")
+ code, out, _ = self.gate("gate 2c91cac7", "--queue")
+ self.assertEqual(code, 0)
+ self.assertTrue(out.startswith("r-2 queued, position 1 (lock held by r-1); est. start ~14:30"), out)
+ self.box.wf("status") # memory is free, lock is not
+ self.assertEqual(self.box.ledger().get("r-2").state, "queued")
+ other = self.box.wf("run", "--mem", "1G", "--for", "5m", "--title", "build", "--", "y")
+ self.assertEqual(other[0], 0) # unlocked work is not held by the locked queue
+ self.box.fake.units["wf-r-1.service"] = ("inactive", None)
+ self.box.wf("status")
+ self.assertEqual(self.box.ledger().get("r-2").state, "running")
+
+ def test_two_queued_gates_start_one_at_a_time(self):
+ self.gate("gate a")
+ self.gate("gate b", "--queue")
+ self.gate("gate c", "--queue")
+ self.box.fake.units["wf-r-1.service"] = ("inactive", None)
+ self.box.wf("status")
+ led = self.box.ledger()
+ self.assertEqual([led.get(i).state for i in ("r-2", "r-3")], ["running", "queued"])
+
+ def test_explicit_lock_and_unlocked_titles(self):
+ self.box.wf("run", "--mem", "2G", "--for", "30m", "--title", "e2e", "--lock", "e2e", "--", "x")
+ self.assertEqual(self.box.wf("run", "--mem", "2G", "--for", "30m", "--title", "e2e 2", "--lock", "e2e",
+ "--", "x")[0], 3)
+ self.assertEqual(self.gate("gate a")[0], 0) # other key
+ self.assertEqual(self.box.wf("run", "--mem", "2G", "--for", "30m", "--title", "gatekeeper", "--", "x")[0], 0)
+ self.assertEqual(self.box.ledger().get("r-1").lock, f"e2e@{self.box.tmp / 'proj'}")
+
+ def test_lock_scoped_to_main_tree(self):
+ self.gate("gate a")
+ self.box.fake.git["rev-parse"] = str(self.box.tmp / "other" / ".git") # another project's tree
+ self.assertEqual(self.gate("gate b")[0], 0)
+ self.box.fake.git["rev-parse"] = str(self.box.tmp / "proj" / ".git") # lane worktree of proj
+ self.assertEqual(self.gate("gate c")[0], 3)
+
+ def test_project_is_main_tree_name_from_lane_worktree(self):
+ self.box.fake.git["rev-parse"] = str(self.box.tmp / "home" / ".git") # cwd = lane worktree 'proj' of home
+ self.box.wf("run", "--mem", "1G", "--for", "1m", "--title", "build 1", "--", "x")
+ self.box.fake.git["rev-parse"] = "" # no git → root name
+ self.box.wf("run", "--mem", "1G", "--for", "1m", "--title", "build 2", "--", "x")
+ self.box.fake.git["rev-parse"] = str(self.box.tmp / "home" / ".git" / "modules" / "m") # submodule → root
+ self.box.wf("run", "--mem", "1G", "--for", "1m", "--title", "build 3", "--", "x")
+ self.assertEqual([self.box.ledger().get(f"r-{i}").project for i in (1, 2, 3)], ["home", "proj", "proj"])
+
+ def test_percent_args_reach_systemd_verbatim(self):
+ # r-671 'fatal: ambiguous argument %s"': caller quoting, not wf; argv passes through untouched
+ self.box.wf("run", "--mem", "1G", "--for", "5m", "--title", "log", "--", "git", "log", "--format=%h %s", "HEAD")
+ self.assertEqual(self.box.fake.ran("systemd-run")[0][-4:], ["git", "log", "--format=%h %s", "HEAD"])
+
+
+def hist_rec(id, title="build 1234567", peak=1.0, minutes=10.0, project="proj", mem=4.0, est=40):
+ return {"id": id, "project": project, "title": title, "mem_gb": mem, "est_min": est, "rc": 0,
+ "peak_gb": peak, "min": minutes}
+
+
+def seed_history(box, recs):
+ box.env.state.mkdir(parents=True, exist_ok=True)
+ (box.env.state / "resources-history.jsonl").write_text("".join(json.dumps(r) + "\n" for r in recs))
+
+
+class HistoryIO(IOBase):
+ def finish(self, id, peak_gb):
+ logs = self.box.env.state / "logs"
+ (logs / f"{id}.rc").write_text("0\n")
+ (logs / f"{id}.peak").write_text(str(int(peak_gb * 1024 ** 3)) + "\n")
+ self.box.fake.units[f"wf-{id}.service"] = ("inactive", None)
+
+ def test_pruned_run_kept_in_history(self):
+ self.box.wf("run", "--mem", "4G", "--for", "40m", "--title", "build abc1234", "--", "make")
+ self.box.tick(12 * 60)
+ self.finish("r-1", 3.0)
+ self.box.wf("status")
+ hist = self.box.env.state / "resources-history.jsonl"
+ self.assertFalse(hist.exists()) # still in the ledger: not copied yet
+ self.box.tick(25 * 3600)
+ self.box.wf("status")
+ self.assertEqual(self.box.ledger().entries, [])
+ self.assertEqual([json.loads(x) for x in hist.read_text().splitlines()], [
+ {"id": "r-1", "project": "proj", "title": "build abc1234", "mem_gb": 4.0, "est_min": 40, "rc": 0,
+ "peak_gb": 3.0, "min": 12.0}])
+
+ def test_history_trimmed(self):
+ seed_history(self.box, [hist_rec(f"r-{i}") for i in range(100, 100 + R.HIST_KEEP)])
+ self.box.wf("run", "--mem", "4G", "--for", "40m", "--title", "build", "--", "make")
+ self.finish("r-1", 1.0)
+ self.box.wf("status")
+ self.box.tick(25 * 3600)
+ self.box.wf("status")
+ lines = (self.box.env.state / "resources-history.jsonl").read_text().splitlines()
+ self.assertEqual(len(lines), R.HIST_KEEP)
+ self.assertEqual((json.loads(lines[0])["id"], json.loads(lines[-1])["id"]), ("r-101", "r-1"))
+
+ def test_run_hint_over_history(self):
+ # peaks 1,2,3 → p95 3 ×1.15 = 3.45 → 3.5 GB; durations 10,20,30 → p90 30 ×1.5 = 45 min
+ seed_history(self.box, [hist_rec("r-90", peak=1.0, minutes=10.0), hist_rec("r-91", peak=2.0, minutes=20.0),
+ hist_rec("r-92", title="build 89abcde (retry)", peak=3.0, minutes=30.0)])
+ code, out, err = self.box.wf("run", "--mem", "8G", "--for", "40m", "--title", "build 7654321", "--", "make")
+ self.assertEqual((code, err), (0, "hint: history says ~3.5 GB / 45 min (3 runs)\n"))
+ code, out, err = self.box.wf("run", "--mem", "7G", "--for", "90m", "--title", "build", "--", "make")
+ self.assertEqual(err, "")
+ code, out, err = self.box.wf("run", "--mem", "1G", "--for", "91m", "--title", "build", "--", "make")
+ self.assertEqual(err, "hint: history says ~3.5 GB / 45 min (3 runs)\n")
+ code, out, err = self.box.wf("note", "--mem", "8G", "--for", "10m", "build")
+ self.assertEqual(err, ("hint: history says ~3.5 GB / 45 min (3 runs)\n"))
+ code, out, err = self.box.wf("run", "--mem", "8G", "--for", "40m", "--title", "other", "--", "make")
+ self.assertEqual(err, "")
+
+ def test_hist_command(self):
+ seed_history(self.box, [hist_rec("r-90", peak=1.0, minutes=10.0), hist_rec("r-91", peak=2.0, minutes=20.0),
+ hist_rec("r-92", peak=3.0, minutes=30.0), hist_rec("r-93", project="zzz")])
+ code, out, err = self.box.wf("hist")
+ self.assertEqual((code, err), (0, ""))
+ self.assertEqual(out, "proj · build · n 3 · req 4.0 GB · peak 2.0/3.0 GB · est 40m · dur 20m/30m · suggest 3.5 GB 45m\n"
+ "zzz · build · n 1 · req 4.0 GB · peak 1.0/1.0 GB · est 40m · dur 10m/10m · suggest - (< 3 runs)\n")
+ self.assertEqual(self.box.wf("hist", "--project", "zzz")[1].count("\n"), 1)
diff --git a/tests/test_runner.py b/tests/test_runner.py
new file mode 100644
index 0000000..a095f2b
--- /dev/null
+++ b/tests/test_runner.py
@@ -0,0 +1,81 @@
+import sys
+import unittest
+from pathlib import Path
+
+HERE = Path(__file__).resolve().parent.parent
+sys.path.insert(0, str(HERE))
+sys.path.insert(0, str(HERE / "tests"))
+
+import test_cli # noqa: E402
+
+TASKS = """\
+# Tasks — demo
+
+## Awaiting your decision
+
+## Pending
+
+- **t-ok** [P1] (1h): Good. Headless.
+ Done: tests green.
+
+- **t-nodone** [P1] (1h): No done line.
+
+- **t-owner** [P1] (1h): Owner bound.
+ Done: tests green.
+ Sessions: owner
+
+- **t-confirm** [P1] (1h): Confirm.
+ Done: owner confirms it works.
+
+- **t-report** [P1] (1h): Report.
+ - Done: report to the owner.
+
+- **t-parent** [P1] (5h): Parent.
+ Done: all slices.
+ Slices: [[t-parent-1]]
+
+- **t-parent-1** [P1] (1h): Slice.
+ Done: ok.
+ Model: sonnet
+
+- **t-doneparent** [P1] (10h): Slices all archived.
+ Done: all slices.
+ - Slices: [[t-gone]]
+
+- **t-blk** [P1] (1h) (blocked: [[a-x]]): Blocked.
+ Done: ok.
+"""
+
+
+class Runner(test_cli.Cli):
+ tasks_text = TASKS
+
+ def col(self, *args):
+ return [l.split()[0] for l in self.ok("list", *args).splitlines()[:-1]]
+
+ def test_runner_filter(self):
+ self.assertEqual(self.col("--runner"), ["t-ok", "t-parent-1"])
+ self.assertEqual(self.col("--runner", "--model", "sonnet"), ["t-parent-1"])
+
+ def test_add_hint_without_done(self):
+ code, out, err = self.wf("add", "-p", "2", "-e", "1h", "Plain thing", hints=True)
+ self.assertEqual((code, err), (0, "hint: no Done line; add one (--done) so runners can pick it\n"))
+
+ def test_add_no_hint_with_done_or_awaiting(self):
+ code, out, err = self.wf("add", "-p", "2", "-e", "1h", "--body", "Body thing", stdin="Done: x\n", hints=True)
+ self.assertEqual((code, err), (0, ""))
+ code, out, err = self.wf("add", "-s", "awaiting", "Which one?", hints=True)
+ self.assertEqual((code, err), (0, ""))
+
+ def test_add_done_flag_writes_done_line(self):
+ code, out, err = self.wf("add", "-p", "2", "-e", "1h", "--done", "gate ALL GREEN", "Flagged thing", hints=True)
+ self.assertEqual((code, err), (0, ""))
+ self.assertIn("\n Done: gate ALL GREEN\n", self.ok("show", "t-flagged-thing") + "\n")
+
+ def test_add_p0_without_done_warns(self):
+ code, out, err = self.wf("add", "-p", "0", "-e", "1h", "Urgent thing", hints=True)
+ self.assertEqual((code, err), (0, 'wf: warning: P0 without Done line is not runner-pickable; use --done "<text>"\n'))
+
+
+if __name__ == "__main__":
+ unittest.main()
diff --git a/tests/test_search.py b/tests/test_search.py
new file mode 100644
index 0000000..4c02520
--- /dev/null
+++ b/tests/test_search.py
@@ -0,0 +1,78 @@
+import sys
+import unittest
+from pathlib import Path
+
+sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
+from wflib import search as S
+from wflib import tasks as T
+
+TASKS = """\
+## Awaiting your decision
+
+## Pending
+
+- **t-guards** [P2] (1h): Guards notice lock picking. Port the notify function.
+ - Steps: reaction drop
+ Ref: docs/ai.md
+
+- **t-throw** [P2] (5h): Thrown weapons and grenades. NPCs throw items.
+ - Steps: guard against throwing into allies
+
+- **t-upgrade** [P3] (1h): Shop upgrenade typo. Fix spelling.
+
+## Needs human
+
+## Deferred
+"""
+ARCHIVE = "# Archive\n\n- 2026-09-01 **t-wear** Weapon wear — guards bust weapons\n- old entry about grenades\n"
+DOCS = {
+ "docs/ai.md": "# AI notes\n\nIntro.\n\n## Guards and thieves\n\nText about catching.\n\n## Combat\n\nA guard attacks. Grenade use.\n",
+}
+
+
+def run(*words, **kw):
+ return S.search(list(words), T.parse(TASKS), ARCHIVE, DOCS, **kw)
+
+
+class SearchTest(unittest.TestCase):
+ def test_title_outranks_body(self):
+ hits = run("guard", kinds={"task"})
+ self.assertEqual([h.where for h in hits], ["t-guards", "t-throw"])
+ self.assertEqual(hits[0].line, "Guards notice lock picking. Port the notify function.")
+ self.assertEqual(hits[1].line, "- Steps: guard against throwing into allies")
+
+ def test_prefix_matches_word_start_only(self):
+ self.assertEqual([h.where for h in run("grenad", kinds={"task"})], ["t-throw"])
+
+ def test_all_words_outrank_one_strong_word(self):
+ hits = run("throwing", "allies", "guards", kinds={"task"})
+ self.assertEqual([h.where for h in hits], ["t-throw", "t-guards"])
+
+ def test_kinds_and_tie_order(self):
+ hits = run("guard")
+ self.assertEqual([(h.kind, h.where) for h in hits],
+ [("task", "t-guards"), ("doc", "docs/ai.md:5"), ("archive", "archive:3"),
+ ("task", "t-throw"), ("doc", "docs/ai.md:9")])
+ self.assertEqual(hits[1].label, "Guards and thieves")
+ self.assertEqual(hits[4].line, "A guard attacks. Grenade use.")
+
+ def test_archive_filter(self):
+ hits = run("grenades", kinds={"archive"})
+ self.assertEqual([(h.where, h.line) for h in hits], [("archive:4", "old entry about grenades")])
+
+ def test_limit(self):
+ self.assertEqual(len(run("guard", limit=2)), 2)
+
+ def test_case_and_regex_chars(self):
+ self.assertEqual([h.where for h in run("GUARD", kinds={"task"})], ["t-guards", "t-throw"])
+ self.assertEqual(run("a.i", "(x"), [])
+
+ def test_id_matches(self):
+ self.assertEqual([h.where for h in run("t-throw", kinds={"task"})], ["t-throw"])
+
+ def test_no_words(self):
+ self.assertEqual(run(), [])
+
+
+if __name__ == "__main__":
+ unittest.main()
diff --git a/tests/test_sessions_field.py b/tests/test_sessions_field.py
new file mode 100644
index 0000000..af1cb62
--- /dev/null
+++ b/tests/test_sessions_field.py
@@ -0,0 +1,197 @@
+import json
+import os
+import sys
+import unittest
+from pathlib import Path
+
+sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
+from wflib import tasks as T
+from wflib import lanes as L
+from test_cli import Cli
+from test_check import Base, CLEAN
+
+SESS = """\
+# Tasks — demo
+
+## Awaiting your decision
+
+## Pending
+
+- **t-solo** [P1] (1h): Solo.
+ - Sessions: solo — lib bump
+ Model: sonnet
+
+- **t-own** [P1] (1h, interactive): Own.
+
+- **t-owner** [P2] (1h): Owner.
+ Sessions: owner
+
+- **t-par** [P2] (<1h): Par.
+ Sessions: parallel
+
+- **t-plain** [P3] (<1h): Plain.
+
+## Needs human
+
+## Deferred
+"""
+
+
+class SessionsLineTest(unittest.TestCase):
+ def setUp(self):
+ self.doc = T.parse(SESS)
+
+ def test_values(self):
+ self.assertEqual([(i.id, i.sessions) for i in self.doc.section("pending").items],
+ [("t-solo", "solo"), ("t-own", "owner"), ("t-owner", "owner"),
+ ("t-par", "parallel"), ("t-plain", "parallel")])
+
+ def test_set_replaces_in_place_and_keeps_tail_order(self):
+ T.set_fields(self.doc, "t-solo", sessions="owner")
+ self.assertEqual(self.doc.item("t-solo").body, [" Sessions: owner", " Model: sonnet"])
+
+ def test_set_adds_before_model(self):
+ T.set_fields(self.doc, "t-plain", sessions="solo")
+ self.assertEqual(self.doc.item("t-plain").body, [" Sessions: solo"])
+
+ def test_set_empty_removes(self):
+ T.set_fields(self.doc, "t-owner", sessions="")
+ self.assertEqual(self.doc.item("t-owner").body, [])
+
+ def test_set_drops_interactive_flag(self):
+ T.set_fields(self.doc, "t-own", sessions="owner")
+ item = self.doc.item("t-own")
+ self.assertEqual(item.lines(), ["- **t-own** [P1] (1h): Own.", " Sessions: owner"])
+
+ def test_set_bad_value(self):
+ with self.assertRaises(T.TaskError):
+ T.set_fields(self.doc, "t-plain", sessions="many")
+
+ def test_note_goes_before_sessions_line(self):
+ T.add_note(self.doc, "t-owner", "x")
+ self.assertEqual(self.doc.item("t-owner").body, [" - x", " Sessions: owner"])
+
+
+class SessionsPickTest(unittest.TestCase):
+ def setUp(self):
+ self.doc = T.parse(SESS)
+
+ def pick(self, lane, model, **kw):
+ return L.pick(self.doc, set(), L.DEFAULT_LANES, "1h", lane, model, **kw)
+
+ def test_alone_solo_picked_owner_skipped(self):
+ item, skipped = self.pick(None, "sonnet")
+ self.assertEqual(item.id, "t-solo")
+ item, skipped = self.pick("slow", "opus", others=1)
+ self.assertEqual((item.id, [(i.id, why) for i, why in skipped]),
+ ("t-par", [("t-solo", "solo: 1 other live session"),
+ ("t-own", "owner: needs the owner (wf next --owner)"),
+ ("t-owner", "owner: needs the owner (wf next --owner)")]))
+
+ def test_owner_present(self):
+ item, _ = self.pick("slow", "opus", others=1, owner=True)
+ self.assertEqual(item.id, "t-own")
+
+ def test_solo_skipped_with_other_live_sessions(self):
+ item, skipped = self.pick(None, "sonnet", others=1)
+ self.assertEqual((item, [(i.id, why) for i, why in skipped]),
+ (None, [("t-solo", "solo: 1 other live session")]))
+ _, skipped = self.pick(None, "sonnet", others=2)
+ self.assertEqual(skipped[0][1], "solo: 2 other live sessions")
+
+ def test_solo_running(self):
+ self.assertEqual(T.solo_running(self.doc, {"t-solo": "sonnet session uds:/a"}),
+ ("t-solo", "sonnet session uds:/a"))
+ self.assertIsNone(T.solo_running(self.doc, {"t-par": "x"}))
+ self.assertIsNone(T.solo_running(self.doc, {}))
+
+ def test_solo_done_block(self):
+ sessions = {"opus": {"socket": "/o", "alive": True, "pid": 1},
+ "haiku": {"socket": "/h", "alive": False, "pid": 2},
+ "sonnet": {"socket": "/s", "alive": True, "pid": 3}}
+ self.assertEqual(L.solo_done_block(["t-solo"], sessions, "3"),
+ ["notify opus uds:/o: solo t-solo done, run wf next"])
+ self.assertEqual(L.solo_done_block([], sessions, "3"), [])
+
+
+class SessionsCheckTest(Base):
+ def test_bad_word_error_interactive_warning(self):
+ text = CLEAN.replace("- **t-one** [P1] (1h)", "- **t-one** [P1] (1h, interactive)").replace(
+ "## Needs human", "- **t-x** [P3] (1h): X.\n Sessions: lots\n\n## Needs human")
+ self.tasks(text)
+ errors, warnings = self.run_check()
+ self.assertTrue(any(e.endswith("t-x: Sessions 'lots' (want parallel, solo, owner)") for e in errors), errors)
+ self.assertTrue(any(w.endswith("t-one: 'interactive' flag: write 'Sessions: owner' "
+ "(wf set t-one --sessions owner)") for w in warnings), warnings)
+
+
+class SessionsCliTest(Cli):
+ tasks_text = SESS
+
+ def env(self, name, pid):
+ sock = self.root / f"{name}.sock"
+ sock.write_text("")
+ return {"CLAUDE_CODE_MESSAGING_SOCKET": str(sock), "CLAUDE_PID": str(pid), "CLAUDE_CODE_SESSION_ID": name}
+
+ def register(self, lane, model, sock, pid):
+ d = self.root / ".wf" / "sessions"
+ d.mkdir(parents=True, exist_ok=True)
+ (d / f"{lane}.json").write_text(json.dumps({"lane": lane, "model": model, "socket": str(sock), "pid": pid,
+ "session": "s", "at": "2026-10-04T10:00"}))
+
+ def test_list_marks(self):
+ self.assertEqual(self.ok("list"),
+ "t-solo P1 1h - sonnet slow [solo] Solo\n"
+ "t-own P1 1h - opus slow [owner] Own\n"
+ "t-owner P2 1h - opus slow [owner] Owner\n"
+ "t-par P2 <1h - opus fast Par\n"
+ "t-plain P3 <1h - opus fast Plain\n"
+ "pending 5 · human 0 · awaiting 0 · deferred 0\n")
+
+ def test_add_and_set(self):
+ self.ok("add", "Four.", "-p", "3", "-e", "1h", "--sessions", "solo", "--model", "haiku")
+ self.assertEqual(self.item("t-four"), "- **t-four** [P3] (1h): Four.\n Sessions: solo\n Model: haiku\n")
+ self.ok("set", "t-four", "--sessions", "")
+ self.assertEqual(self.item("t-four"), "- **t-four** [P3] (1h): Four.\n Model: haiku\n")
+ self.fails("set", "t-four", "--sessions", "many", code=2)
+ self.fails("add", "Five.", "-p", "3", "-e", "1h", "--sessions", "many", code=2)
+
+ def test_interactive_option_is_old_spelling(self):
+ code, out, err = self.wf("add", "Four.", "-p", "3", "-e", "1h", "--interactive")
+ self.assertEqual((code, err), (0, "wf: --interactive is now --sessions owner\n"))
+ self.assertEqual(self.item("t-four"), "- **t-four** [P3] (1h): Four.\n Sessions: owner\n")
+ code, out, err = self.wf("set", "t-plain", "--interactive", "yes")
+ self.assertEqual((code, err), (0, "wf: --interactive is now --sessions owner\n"))
+ self.assertEqual(self.item("t-plain"), "- **t-plain** [P3] (<1h): Plain.\n Sessions: owner\n")
+
+ def test_next_owner(self):
+ self.ok("done", "t-solo", "-m", "ok")
+ out = self.ok("next", "--lane", "slow", "--as", "opus", env={"CLAUDE_PID": "", "CLAUDE_CODE_MESSAGING_SOCKET": ""})
+ self.assertIn("- t-own: owner: needs the owner (wf next --owner)\n", out)
+ self.assertIn("===== Next task =====\n- **t-par**", out) # slow empty → fallback fast
+ self.assertTrue(self.ok("next", "--lane", "slow", "--as", "opus", "--owner", "--brief").startswith("- **t-own**"))
+
+ def test_next_solo_with_other_live_session(self):
+ sock = self.root / "o.sock"
+ sock.write_text("")
+ self.register("fast", "opus", sock, os.getppid())
+ err = self.fails("next", "--as", "sonnet", env=self.env("s", os.getpid()))
+ self.assertEqual(err, "wf: nothing pickable for all lanes (sonnet) in Pending\n")
+ out = self.wf("next", "--as", "sonnet", env=self.env("s", os.getpid()))[1]
+ self.assertIn("- t-solo: solo: 1 other live session\n", out)
+
+ def test_solo_in_progress_blocks_others_and_done_notifies(self):
+ s, o = self.env("s", os.getpid()), self.env("o", os.getppid())
+ self.ok("next", "--lane", "slow", "--as", "sonnet", "--brief", env=s)
+ self.ok("status", "t-solo", "progress", "x", env=s)
+ code, out, err = self.wf("next", "--lane", "fast", "--as", "opus", env=o)
+ self.assertEqual((code, err), (1, f"wf: solo t-solo in progress by sonnet session uds:{self.root / 's.sock'}: "
+ "wait (its wf done notifies you)\n"))
+ self.assertNotIn("Next task", out)
+ out = self.ok("done", "t-solo", "-m", "ok", env=s)
+ self.assertIn(f"notify fast uds:{self.root / 'o.sock'}: solo t-solo done, run wf next\n", out)
+ self.assertTrue(self.ok("next", "--lane", "fast", "--as", "opus", "--brief", env=o).startswith("- **t-par**"))
+
+
+if __name__ == "__main__":
+ unittest.main()
diff --git a/tests/test_setup.py b/tests/test_setup.py
new file mode 100644
index 0000000..ea44c33
--- /dev/null
+++ b/tests/test_setup.py
@@ -0,0 +1,164 @@
+import os
+import sys
+import tempfile
+import unittest
+from pathlib import Path
+
+sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
+from wflib import config as C
+from test_claims import FOUR, git
+from test_cli import TOML, Cli
+
+SETUP = TOML + 'worktree_setup = ["mkdir -p out && ln -sfn \\"$WF_MAIN/out/data\\" out/data", "echo ran >> log.txt"]\n'
+
+
+class SetupConfigTest(unittest.TestCase):
+ def setUp(self):
+ self.tmp = tempfile.TemporaryDirectory()
+ self.root = Path(self.tmp.name).resolve()
+
+ def tearDown(self):
+ self.tmp.cleanup()
+
+ def load(self, extra):
+ (self.root / "workflow.toml").write_text('format = 1\ntasks = "T.md"\narchive = "a.md"\n' + extra)
+ return C.load(self.root)
+
+ def test_default_empty(self):
+ self.assertEqual(self.load("").worktree_setup, [])
+
+ def test_list_of_commands(self):
+ self.assertEqual(self.load('worktree_setup = ["a", "b c"]\n').worktree_setup, ["a", "b c"])
+
+ def test_wrong_type(self):
+ with self.assertRaisesRegex(C.ConfigError, "'worktree_setup' must be a list of strings"):
+ self.load('worktree_setup = "a"\n')
+
+ def test_unknown_key_still_errors(self):
+ with self.assertRaisesRegex(C.ConfigError, "unknown key 'worktree_setp'"):
+ self.load('worktree_setp = []\n')
+
+
+class GitCli(Cli):
+ tasks_text = FOUR
+ toml = SETUP
+
+ def setUp(self):
+ super().setUp()
+ git(self.root, "init", "-q", "-b", "master")
+ (self.root / ".gitignore").write_text(".worktrees/\n.wf/\nout/\n")
+ git(self.root, "add", "-A")
+ git(self.root, "commit", "-qm", "init")
+
+
+class WorktreeCli(GitCli):
+ def setUp(self):
+ super().setUp()
+ self.wt = self.root / ".worktrees" / "opus"
+ git(self.root, "worktree", "add", "-q", str(self.wt), "-b", "opus/t-three")
+ (self.root / "out" / "data").mkdir(parents=True)
+ (self.root / "out" / "data" / "x.bin").write_text("main data\n")
+
+ def setup_(self, *args, cwd=None):
+ return self.wf("setup", *args, project=False, cwd=cwd or self.wt)
+
+
+class SetupCliTest(WorktreeCli):
+ def test_runs_commands_in_worktree_with_wf_main(self):
+ code, out, err = self.setup_()
+ self.assertEqual((code, err), (0, ""), out)
+ self.assertEqual((self.wt / "out" / "data" / "x.bin").read_text(), "main data\n")
+ self.assertEqual((self.wt / "log.txt").read_text(), "ran\n")
+ self.assertFalse((self.root / "log.txt").exists())
+ self.assertIn("$ echo ran >> log.txt\n", out)
+ self.assertTrue(out.endswith("worktree_setup: 2 commands ok in .worktrees/opus\n"), out)
+
+ def test_idempotent_rerun(self):
+ self.assertEqual(self.setup_()[0], 0)
+ self.assertEqual(self.setup_()[0], 0)
+ self.assertEqual((self.wt / "out" / "data" / "x.bin").read_text(), "main data\n")
+
+ def test_from_subfolder_runs_at_worktree_project_root(self):
+ code, out, err = self.setup_(cwd=self.wt / "docs")
+ self.assertEqual((code, err), (0, ""), out)
+ self.assertTrue((self.wt / "log.txt").is_file())
+
+ def test_nonzero_exit_reported_and_rest_skipped(self):
+ (self.wt / "workflow.toml").write_text(TOML + 'worktree_setup = ["echo a > a.txt", "exit 3", "echo c > c.txt"]\n')
+ code, out, err = self.setup_()
+ self.assertEqual(code, 1, out + err)
+ self.assertEqual(err, "wf: worktree_setup 'exit 3' failed (exit 3): later commands skipped\n")
+ self.assertTrue((self.wt / "a.txt").is_file())
+ self.assertFalse((self.wt / "c.txt").exists())
+
+ def test_outside_worktree_fails(self):
+ code, out, err = self.wf("setup", project=False, cwd=self.root)
+ self.assertEqual((code, err), (1, "wf: setup runs inside a linked git worktree (lane worktree)\n"))
+
+ def test_none_configured(self):
+ (self.wt / "workflow.toml").write_text(TOML)
+ code, out, err = self.setup_()
+ self.assertEqual((code, out, err), (0, "no worktree_setup in workflow.toml: nothing to do\n", ""))
+
+
+class WorktreeConfigTest(WorktreeCli):
+ """A lane worktree's own workflow.toml / area file: setup, gate, areas read it; --mark writes it."""
+
+ def edit_wt_toml(self, extra):
+ (self.wt / "workflow.toml").write_text(TOML + extra)
+
+ def test_setup_and_gate_read_worktree_toml(self):
+ self.edit_wt_toml('worktree_setup = ["echo branch > b.txt"]\nquick_gate = ["echo g > g.txt"]\n')
+ code, out, err = self.setup_()
+ self.assertEqual((code, err), (0, ""), out)
+ self.assertEqual((self.wt / "b.txt").read_text(), "branch\n")
+ self.assertFalse((self.wt / "log.txt").exists())
+ code, out, err = self.wf("gate", project=False, cwd=self.wt)
+ self.assertEqual((code, err), (0, ""), out)
+ self.assertTrue((self.wt / "g.txt").is_file())
+
+ def test_books_stay_main_tree(self):
+ self.edit_wt_toml("")
+ (self.wt / "TASKS.md").write_text("# other\n")
+ out = self.ok("show", "t-three", project=False, cwd=self.wt)
+ self.assertIn("t-three", out)
+ cfg = C.load_at(self.wt)
+ self.assertEqual((cfg.root, cfg.local, cfg.tasks), (self.root, self.wt, self.root / "TASKS.md"))
+ self.assertEqual(cfg.areas_file, self.wt / "CLAUDE.md")
+
+ def test_areas_read_and_mark_worktree_copy(self):
+ notes = "## Areas\n### Core\n- Code map: `nothing_here`\n"
+ (self.root / "CLAUDE.md").write_text(notes.replace("Core", "Main"))
+ (self.wt / "CLAUDE.md").write_text(notes)
+ self.assertIn("Core: ", self.ok("areas", project=False, cwd=self.wt))
+ self.ok("areas", "--mark", "Core", project=False, cwd=self.wt)
+ self.assertIn("- Checked: ", (self.wt / "CLAUDE.md").read_text())
+ self.assertEqual((self.root / "CLAUDE.md").read_text(), notes.replace("Core", "Main"))
+ self.assertIn("Main: ", self.ok("areas", project=False, cwd=self.root))
+
+ def test_main_tree_ignores_worktree(self):
+ self.edit_wt_toml('quick_gate = ["exit 9"]\n')
+ self.assertIsNone(C.find_local(self.root))
+ code, out, err = self.wf("gate", project=False, cwd=self.root)
+ self.assertEqual((code, out), (0, "no quick_gate in workflow.toml: nothing to do\n"))
+
+
+class SetupHintTest(GitCli):
+ def env(self, name, pid):
+ sock = self.root / f"{name}.sock"
+ sock.write_text("")
+ return {"CLAUDE_CODE_MESSAGING_SOCKET": str(sock), "CLAUDE_PID": str(pid), "CLAUDE_CODE_SESSION_ID": name}
+
+ def test_multi_session_hint_adds_setup(self):
+ son, me = self.env("s", os.getppid()), self.env("o", os.getpid())
+ self.ok("next", "--lane", "slow", "--as", "sonnet", "--brief", env=son)
+ out = self.ok("next", "--lane", "fast", "--as", "opus", env=me)
+ self.assertIn(" git worktree add .worktrees/fast -b fast/<task> master && cd .worktrees/fast && wf setup"
+ " (wf there writes this TASKS.md)\n", out)
+ (self.root / ".worktrees" / "fast").mkdir(parents=True)
+ out = self.ok("next", "--lane", "fast", "--as", "opus", env=me)
+ self.assertIn(" cd .worktrees/fast && git switch -c fast/<task> master && wf setup"
+ " (wf there writes this TASKS.md)\n", out)
+
+if __name__ == "__main__":
+ unittest.main()
diff --git a/tests/test_start.py b/tests/test_start.py
new file mode 100644
index 0000000..e1a2d6e
--- /dev/null
+++ b/tests/test_start.py
@@ -0,0 +1,109 @@
+import sys
+import unittest
+from pathlib import Path
+
+sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
+from test_claims import git
+from test_cli import WF
+from test_setup import GitCli
+
+NONE = {"CLAUDE_CODE_MESSAGING_SOCKET": ""}
+
+
+class StartTest(GitCli):
+ def setUp(self):
+ super().setUp()
+ self.wt = self.root / ".worktrees" / "slow"
+
+ def start(self, *extra, code=0):
+ got, out, err = self.wf("start", "t-three", "--worktree", ".worktrees/slow", "--branch", "slow/t-three",
+ *extra, project=False, env=NONE)
+ self.assertEqual(got, code, out + err)
+ self.assertNotIn("Traceback", err)
+ return out, err
+
+ def branch(self):
+ return (self.wt / ".git").is_file() and \
+ subprocess_out(self.wt, "branch", "--show-current")
+
+ def test_new_worktree_setup_progress_ctx_finish_line(self):
+ out, _ = self.start()
+ self.assertEqual(self.branch(), "slow/t-three")
+ self.assertEqual((self.wt / "log.txt").read_text(), "ran\n") # worktree_setup ran there
+ self.assertIn("**t-three** [P2] (<1h) (in progress: slow/t-three): Three.", self.tasks())
+ self.assertIn("worktree: .worktrees/slow (new, branch slow/t-three from master)\n", out)
+ self.assertEqual(out.count("- **t-three**"), 1, out)
+ self.assertIn("worktree_setup: 2 commands ok in .worktrees/slow\n", out)
+ self.assertIn("- **t-three** [P2] (<1h) (in progress: slow/t-three): Three.\n\nSection: Pending\n", out)
+ self.assertTrue(out.endswith(
+ f"Finish (after the work; fill in the quoted parts and the paths):\n"
+ f" cd {self.wt} && (make test) && python3 {WF} finish t-three -m \"<entry>\" --commit \"<msg + footer>\" <paths>\n"),
+ out)
+
+ def test_existing_branch_new_worktree_reports_wip(self):
+ git(self.root, "branch", "slow/t-three")
+ out, _ = self.start()
+ self.assertEqual(self.branch(), "slow/t-three")
+ self.assertIn("worktree: .worktrees/slow (new, existing branch slow/t-three: earlier WIP, read the notes)\n",
+ out)
+
+ def test_existing_clean_worktree_switches(self):
+ git(self.root, "worktree", "add", "-q", "--detach", str(self.wt), "master")
+ out, _ = self.start()
+ self.assertEqual(self.branch(), "slow/t-three")
+ self.assertIn("worktree: .worktrees/slow (switched to new branch slow/t-three from master)\n", out)
+ (self.wt / "log.txt").unlink() # setup output, not ignored here
+ out, _ = self.start() # rerun: already there
+ self.assertIn("worktree: .worktrees/slow (on slow/t-three)\n", out)
+
+ def test_dirty_worktree_refused_without_recovery(self):
+ git(self.root, "worktree", "add", "-q", "--detach", str(self.wt), "master")
+ (self.wt / "DESIGN.md").write_text("changed\n")
+ _, err = self.start(code=1)
+ self.assertEqual(err, "wf: worktree .worktrees/slow has uncommitted changes: M DESIGN.md "
+ "(hand back, or --recovery when a dead worker left them)\n")
+ self.assertNotIn("in progress", self.tasks())
+ self.assertFalse((self.wt / "log.txt").exists())
+
+ def test_recovery_keeps_dirty_wip_and_shows_it(self):
+ git(self.root, "worktree", "add", "-q", str(self.wt), "-b", "slow/t-three")
+ (self.wt / "DESIGN.md").write_text("changed\n")
+ out, _ = self.start("--recovery")
+ self.assertEqual((self.wt / "DESIGN.md").read_text(), "changed\n")
+ self.assertIn("recovery: uncommitted:\n M DESIGN.md\n", out)
+ self.assertIn("recovery: commits master..HEAD: none\n", out)
+ self.assertIn("(in progress: slow/t-three)", self.tasks())
+
+ def test_recovery_dirty_on_other_branch_refused(self):
+ git(self.root, "worktree", "add", "-q", "--detach", str(self.wt), "master")
+ (self.wt / "DESIGN.md").write_text("changed\n")
+ _, err = self.start("--recovery", code=1)
+ self.assertIn("uncommitted changes on another branch", err)
+
+ def test_setup_failure_stops_before_progress(self):
+ (self.root / "workflow.toml").write_text(self.toml.replace('"echo ran >> log.txt"', '"false"'))
+ git(self.root, "commit", "-qam", "red setup")
+ _, err = self.start(code=1)
+ self.assertIn("worktree_setup 'false' failed", err)
+ self.assertNotIn("in progress", self.tasks())
+
+ def test_unknown_id_creates_nothing(self):
+ got, out, err = self.wf("start", "t-nope", "--worktree", ".worktrees/slow", "--branch", "b",
+ project=False, env=NONE)
+ self.assertEqual(got, 1, out + err)
+ self.assertFalse(self.wt.exists())
+
+ def test_inside_worktree_refused(self):
+ git(self.root, "worktree", "add", "-q", "--detach", str(self.wt), "master")
+ got, out, err = self.wf("start", "t-three", "--worktree", str(self.wt), "--branch", "b",
+ project=False, cwd=self.wt, env=NONE)
+ self.assertEqual((got, err), (1, "wf: start runs in the main tree (it creates the worktree)\n"))
+
+
+def subprocess_out(cwd, *args):
+ import subprocess
+ return subprocess.run(["git", "-C", str(cwd), *args], capture_output=True, text=True).stdout.strip()
+
+
+if __name__ == "__main__":
+ unittest.main()
diff --git a/tests/test_tasks.py b/tests/test_tasks.py
new file mode 100644
index 0000000..64f219c
--- /dev/null
+++ b/tests/test_tasks.py
@@ -0,0 +1,523 @@
+import sys
+import unittest
+from pathlib import Path
+
+sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
+from wflib import tasks as T
+from wflib import lanes as L
+
+SAMPLE = """\
+# Tasks — demo
+
+Commands: `make test`.
+
+## Awaiting your decision
+
+- **a-smoke**: Smoke screenshots. Need you present.
+
+## Pending
+
+Intro prose.
+
+- **t-first** [P1] (<1h): First thing. Do it well.
+ - Steps: one
+ - After: [[t-zero]], [[t-other]]
+ Ref: DESIGN.md#terrain, docs/x.md (skill rows)
+
+- **t-second** [P2] (5h, interactive) (in progress: master): Second: with colon. Goal here.
+
+- **t-third** [P3] (1h) (blocked: [[a-smoke]]): Third.
+
+Trailing prose after items.
+
+## Needs human
+
+- **t-play** [P1] (<1h): Play-test feel.
+ - [ ] keyboard works
+ - [ ] motion smooth
+
+## Deferred
+
+## Known caveats
+
+- plain bullet, mentions [[t-first]]
+"""
+
+
+class HeaderTest(unittest.TestCase):
+ def setUp(self):
+ self.doc = T.parse(SAMPLE)
+
+ def item(self, id):
+ section, i = self.doc.find(id)
+ return section.items[i]
+
+ def test_task_with_priority_and_effort(self):
+ it = self.item("t-first")
+ self.assertEqual((it.id, it.prio, it.effort, it.interactive, it.status, it.text),
+ ("t-first", 1, "<1h", False, None, "First thing. Do it well."))
+
+ def test_interactive_and_in_progress(self):
+ it = self.item("t-second")
+ self.assertEqual((it.prio, it.effort, it.interactive, it.status),
+ (2, "5h", True, "in progress: master"))
+
+ def test_blocked(self):
+ it = self.item("t-third")
+ self.assertEqual(it.status, "blocked: [[a-smoke]]")
+ self.assertEqual(it.blocked_on, "a-smoke")
+ self.assertIsNone(self.item("t-first").blocked_on)
+
+ def test_awaiting_item_has_no_priority(self):
+ it = self.item("a-smoke")
+ self.assertEqual((it.prio, it.effort, it.text), (None, None, "Smoke screenshots. Need you present."))
+
+ def test_header_rebuilds_the_line(self):
+ self.assertEqual(self.item("t-first").header(), "- **t-first** [P1] (<1h): First thing. Do it well.")
+ self.assertEqual(self.item("t-second").header(),
+ "- **t-second** [P2] (5h, interactive) (in progress: master): Second: with colon. Goal here.")
+ self.assertEqual(self.item("t-third").header(), "- **t-third** [P3] (1h) (blocked: [[a-smoke]]): Third.")
+ self.assertEqual(self.item("a-smoke").header(), "- **a-smoke**: Smoke screenshots. Need you present.")
+
+ def test_title_and_goal(self):
+ it = self.item("t-second")
+ self.assertEqual(it.title, "Second: with colon")
+ self.assertEqual(it.goal, "Goal here.")
+ self.assertEqual(self.item("t-third").title, "Third")
+ self.assertEqual(self.item("t-third").goal, "")
+
+ def test_after(self):
+ self.assertEqual(self.item("t-first").after, ["t-zero", "t-other"])
+ self.assertEqual(self.item("t-second").after, [])
+
+ def test_refs(self):
+ self.assertEqual(self.item("t-first").refs, [("DESIGN.md", "terrain"), ("docs/x.md", None)])
+ self.assertEqual(self.item("t-second").refs, [])
+
+ def test_body_lines(self):
+ self.assertEqual(self.item("t-play").body, [" - [ ] keyboard works", " - [ ] motion smooth"])
+
+
+class StructureTest(unittest.TestCase):
+ def test_round_trip(self):
+ self.assertEqual(T.render(T.parse(SAMPLE)), SAMPLE)
+
+ def test_sections_and_keys(self):
+ doc = T.parse(SAMPLE)
+ self.assertEqual([(s.heading, s.key) for s in doc.sections],
+ [("Awaiting your decision", "awaiting"), ("Pending", "pending"),
+ ("Needs human", "human"), ("Deferred", "deferred"), ("Known caveats", None)])
+ self.assertEqual([i.id for i in doc.section("pending").items], ["t-first", "t-second", "t-third"])
+ self.assertEqual(doc.section("pending").prefix, ["", "Intro prose.", ""])
+ self.assertEqual(doc.section("pending").suffix, ["Trailing prose after items.", ""])
+ self.assertEqual(doc.ids(), {"a-smoke", "t-first", "t-second", "t-third", "t-play"})
+
+ def test_prose_section_bullets_are_not_items(self):
+ doc = T.parse(SAMPLE)
+ self.assertEqual(doc.sections[-1].items, [])
+ self.assertEqual(doc.sections[-1].prefix, ["", "- plain bullet, mentions [[t-first]]"])
+
+ def test_missing_section_raises(self):
+ doc = T.parse("# x\n\n## Pending\n")
+ with self.assertRaisesRegex(T.TaskError, "no '## Deferred' section"):
+ doc.section("deferred")
+
+ def test_find_unknown_id_names_nearest(self):
+ with self.assertRaisesRegex(T.TaskError, r"unknown id 't-frist' \(nearest: t-first"):
+ T.parse(SAMPLE).find("t-frist")
+
+ def test_items_are_separated_by_one_blank_line(self):
+ text = "## Pending\n- **t-a** [P1] (1h): A.\n\n\n\n- **t-b** [P1] (1h): B.\n body\n- **t-c** [P1] (1h): C.\n"
+ self.assertEqual(T.render(T.parse(text)),
+ "## Pending\n- **t-a** [P1] (1h): A.\n\n- **t-b** [P1] (1h): B.\n body\n\n- **t-c** [P1] (1h): C.\n")
+
+ def test_malformed_header_is_kept_and_flagged(self):
+ text = "## Pending\n\n- **t-x** [P9] (1h) no colon\n body\n\n- **t-ok** [P1] (1h): Fine.\n"
+ doc = T.parse(text)
+ bad = doc.section("pending").items[0]
+ self.assertEqual(bad.raw, "- **t-x** [P9] (1h) no colon")
+ self.assertIn("header", bad.error)
+ self.assertEqual(bad.id, "t-x")
+ self.assertEqual(T.render(doc), text)
+
+ def test_flush_left_line_ends_the_items_and_survives(self):
+ text = ("## Pending\n\n- **t-a** [P1] (1h): A.\n body\nstray continuation\n"
+ "- **t-b** [P1] (1h): B.\n\n## Deferred\n")
+ doc = T.parse(text)
+ pending = doc.section("pending")
+ self.assertEqual([i.id for i in pending.items], ["t-a"])
+ self.assertEqual(pending.suffix, ["stray continuation", "- **t-b** [P1] (1h): B.", ""])
+ self.assertEqual(T.render(doc),
+ "## Pending\n\n- **t-a** [P1] (1h): A.\n body\n\nstray continuation\n"
+ "- **t-b** [P1] (1h): B.\n\n## Deferred\n")
+
+ def test_crlf_is_kept(self):
+ text = "## Pending\r\n\r\n- **t-a** [P1] (1h): A.\r\n body\r\n"
+ doc = T.parse(text)
+ self.assertEqual(doc.newline, "\r\n")
+ self.assertEqual(doc.section("pending").items[0].body, [" body"])
+ self.assertEqual(T.render(doc), text)
+
+ def test_missing_final_newline_is_added(self):
+ self.assertEqual(T.render(T.parse("## Pending\n\n- **t-a** [P1] (1h): A.")),
+ "## Pending\n\n- **t-a** [P1] (1h): A.\n")
+
+
+if __name__ == "__main__":
+ unittest.main()
+
+
+EDIT = """\
+## Awaiting your decision
+
+- **a-key**: Key needed. Which one?
+
+## Pending
+
+- **t-p0** [P0] (1h): Zero.
+
+- **t-p1** [P1] (1h): One.
+ - Steps: a
+ - After: [[t-p0]]
+ Ref: docs/a.md
+
+- **t-p2** [P2] (1h) (blocked: [[a-key]]): Two.
+
+- **t-p3** [P3] (1h): Three.
+ - After: [[t-gone]]
+
+## Needs human
+
+- **t-play** [P1] (<1h): Play.
+ - [ ] keyboard works
+ - [x] motion smooth
+ - [ ] gait feels right
+
+## Deferred
+"""
+
+
+def ids(doc, key):
+ return [i.id for i in doc.section(key).items]
+
+
+class MakeIdTest(unittest.TestCase):
+ def test_slug_of_title(self):
+ self.assertEqual(T.make_id("Guards notice lock picking and busting", set()),
+ "t-guards-notice-lock-picking-and-busting")
+
+ def test_long_title_is_cut_at_a_word(self):
+ self.assertEqual(T.make_id("Exe function map plus port re-verification of everything", set()),
+ "t-exe-function-map-plus-port-re")
+ self.assertEqual(T.make_id("Thrown weapons and grenades for everyone here", set()),
+ "t-thrown-weapons-and-grenades-for-everyone")
+
+ def test_taken_id_gets_a_number(self):
+ self.assertEqual(T.make_id("Cave seams", {"t-cave-seams"}), "t-cave-seams-2")
+ self.assertEqual(T.make_id("Cave seams", {"t-cave-seams", "t-cave-seams-2"}), "t-cave-seams-3")
+
+ def test_empty_slug_is_refused(self):
+ with self.assertRaisesRegex(T.TaskError, "--id"):
+ T.make_id("???", set())
+
+ def test_non_ascii_is_dropped(self):
+ self.assertEqual(T.make_id("Åäö test", set()), "t-test")
+
+ def test_awaiting_prefix(self):
+ self.assertEqual(T.make_id("Smoke screenshots", set(), "a-"), "a-smoke-screenshots")
+
+ def test_slice_id(self):
+ self.assertEqual(T.slice_id("t-map", {"t-map"}), "t-map-1")
+ self.assertEqual(T.slice_id("t-map", {"t-map", "t-map-1", "t-map-2"}), "t-map-3")
+
+
+class EditTest(unittest.TestCase):
+ def setUp(self):
+ self.doc = T.parse(EDIT)
+
+ def new(self, id, prio, body=()):
+ return T.Item(id=id, prio=prio, effort="1h", text="New.", body=list(body))
+
+ def test_insert_by_priority(self):
+ T.insert(self.doc, self.new("t-n", 1), "pending")
+ self.assertEqual(ids(self.doc, "pending"), ["t-p0", "t-p1", "t-n", "t-p2", "t-p3"])
+
+ def test_insert_into_empty_section(self):
+ T.insert(self.doc, self.new("t-n", 0), "deferred")
+ self.assertEqual(ids(self.doc, "deferred"), ["t-n"])
+
+ def test_insert_goes_after_its_dependency(self):
+ T.insert(self.doc, self.new("t-n", 1, [" - After: [[t-p3]]"]), "pending")
+ self.assertEqual(ids(self.doc, "pending"), ["t-p0", "t-p1", "t-p2", "t-p3", "t-n"])
+
+ def test_insert_refuses_taken_id(self):
+ with self.assertRaisesRegex(T.TaskError, "'t-p1' already exists"):
+ T.insert(self.doc, self.new("t-p1", 1), "pending")
+
+ def test_insert_refuses_wrong_kind_for_section(self):
+ with self.assertRaisesRegex(T.TaskError, "a- items belong in Awaiting"):
+ T.insert(self.doc, T.Item(id="a-x", text="X."), "pending")
+ with self.assertRaisesRegex(T.TaskError, "needs a priority"):
+ T.insert(self.doc, T.Item(id="t-x", effort="1h", text="X."), "pending")
+
+ def test_remove(self):
+ item = T.remove(self.doc, "t-p1")
+ self.assertEqual(item.body, [" - Steps: a", " - After: [[t-p0]]", " Ref: docs/a.md"])
+ self.assertEqual(ids(self.doc, "pending"), ["t-p0", "t-p2", "t-p3"])
+
+ def test_remove_unknown_names_nearest(self):
+ with self.assertRaisesRegex(T.TaskError, r"unknown id 't-p9' \(nearest: t-p"):
+ T.remove(self.doc, "t-p9")
+
+ def test_set_prio_moves_the_item_and_keeps_its_body(self):
+ T.set_prio(self.doc, "t-p1", 3)
+ self.assertEqual(ids(self.doc, "pending"), ["t-p0", "t-p2", "t-p3", "t-p1"])
+ item = self.doc.item("t-p1")
+ self.assertEqual((item.prio, item.body[0]), (3, " - Steps: a"))
+
+ def test_set_prio_range(self):
+ with self.assertRaisesRegex(T.TaskError, "priority 0-3"):
+ T.set_prio(self.doc, "t-p1", 4)
+
+ def test_move_to_section(self):
+ T.move_to(self.doc, "t-p3", "deferred")
+ self.assertEqual(ids(self.doc, "pending"), ["t-p0", "t-p1", "t-p2"])
+ self.assertEqual(ids(self.doc, "deferred"), ["t-p3"])
+
+ def test_move_to_refuses_kind_change(self):
+ with self.assertRaisesRegex(T.TaskError, "a- items belong in Awaiting"):
+ T.move_to(self.doc, "a-key", "pending")
+
+ def test_move_rel_refuses_priority_break(self):
+ with self.assertRaisesRegex(T.TaskError, "breaks priority order.*--force"):
+ T.move_rel(self.doc, "t-p3", "t-p0", before=True)
+ self.assertEqual(ids(self.doc, "pending"), ["t-p0", "t-p1", "t-p2", "t-p3"])
+
+ def test_move_rel_with_force(self):
+ T.move_rel(self.doc, "t-p3", "t-p0", before=True, force=True)
+ self.assertEqual(ids(self.doc, "pending"), ["t-p3", "t-p0", "t-p1", "t-p2"])
+
+ def test_move_rel_never_puts_an_item_before_its_dependency(self):
+ with self.assertRaisesRegex(T.TaskError, "'t-p1' is After: \\[\\[t-p0\\]\\]"):
+ T.move_rel(self.doc, "t-p1", "t-p0", before=True, force=True)
+
+ def test_move_rel_after_into_other_section(self):
+ T.move_rel(self.doc, "t-p1", "t-play", before=False)
+ self.assertEqual(ids(self.doc, "human"), ["t-play", "t-p1"])
+
+ def test_status(self):
+ T.set_status(self.doc, "t-p1", "in progress: feature/x")
+ self.assertEqual(self.doc.item("t-p1").header(), "- **t-p1** [P1] (1h) (in progress: feature/x): One.")
+ T.set_status(self.doc, "t-p1", "blocked: [[a-key]]")
+ self.assertEqual(self.doc.item("t-p1").blocked_on, "a-key")
+ T.set_status(self.doc, "t-p1", None)
+ self.assertEqual(self.doc.item("t-p1").header(), "- **t-p1** [P1] (1h): One.")
+
+ def test_status_blocked_on_unknown_awaiting_item(self):
+ with self.assertRaisesRegex(T.TaskError, "'a-none' is not an open Awaiting item"):
+ T.set_status(self.doc, "t-p1", "blocked: [[a-none]]")
+
+ def test_set_fields_header(self):
+ T.set_fields(self.doc, "t-p1", title="Uno", effort="5h", interactive=True)
+ self.assertEqual(self.doc.item("t-p1").header(), "- **t-p1** [P1] (5h, interactive): Uno.")
+ T.set_fields(self.doc, "t-p0", title="Nil")
+ self.assertEqual(self.doc.item("t-p0").text, "Nil.")
+
+ def test_set_fields_title_keeps_goal(self):
+ doc = T.parse("## Pending\n\n- **t-a** [P1] (1h): Old name. The goal. More.\n")
+ T.set_fields(doc, "t-a", title="New name")
+ self.assertEqual(doc.item("t-a").text, "New name. The goal. More.")
+
+ def test_set_fields_full_text_replaces_goal(self):
+ doc = T.parse("## Awaiting your decision\n\n- **a-q** : Colour. OK?\n\n## Pending\n\n"
+ "- **t-a** [P1] (1h): Old name. Old goal.\n".replace(" :", ":"))
+ T.set_fields(doc, "a-q", title="Colour red. OK?")
+ self.assertEqual(doc.item("a-q").text, "Colour red. OK?")
+ T.set_fields(doc, "t-a", title="New name. New goal")
+ self.assertEqual(doc.item("t-a").text, "New name. New goal.")
+ T.set_fields(doc, "t-a", title="Why not?")
+ self.assertEqual(doc.item("t-a").text, "Why not?")
+
+ def test_set_fields_bad_effort(self):
+ with self.assertRaisesRegex(T.TaskError, "effort '2h'"):
+ T.set_fields(self.doc, "t-p1", effort="2h")
+
+ def test_set_after(self):
+ T.set_fields(self.doc, "t-p1", after=[])
+ self.assertEqual(self.doc.item("t-p1").body, [" - Steps: a", " Ref: docs/a.md"])
+ T.set_fields(self.doc, "t-p1", after=["t-p0", "t-p2"])
+ self.assertEqual(self.doc.item("t-p1").body,
+ [" - Steps: a", " - After: [[t-p0]], [[t-p2]]", " Ref: docs/a.md"])
+
+ def test_set_done(self):
+ doc = T.parse("## Pending\n\n- **t-a** [P1] (1h): A.\n - Steps: a\n - note\n Model: sonnet\n Ref: x.md\n")
+ T.set_fields(doc, "t-a", done="a works")
+ self.assertEqual(doc.item("t-a").body,
+ [" - Steps: a", " - note", " Done: a works", " Model: sonnet", " Ref: x.md"])
+ T.set_fields(doc, "t-a", done=" b works ")
+ self.assertEqual(doc.item("t-a").body[2], " Done: b works")
+ T.set_fields(doc, "t-a", done="")
+ self.assertEqual(doc.item("t-a").body, [" - Steps: a", " - note", " Model: sonnet", " Ref: x.md"])
+
+ def test_set_refs(self):
+ T.set_fields(self.doc, "t-p1", refs=["docs/b.md#x (rows)", "DESIGN.md"])
+ self.assertEqual(self.doc.item("t-p1").body[-1], " Ref: docs/b.md#x (rows), DESIGN.md")
+ T.set_fields(self.doc, "t-p0", refs=["docs/a.md"])
+ self.assertEqual(self.doc.item("t-p0").body, [" Ref: docs/a.md"])
+ T.set_fields(self.doc, "t-p0", refs=[])
+ self.assertEqual(self.doc.item("t-p0").body, [])
+
+ def test_add_note_goes_before_after_and_ref(self):
+ T.add_note(self.doc, "t-p1", "found the cause")
+ self.assertEqual(self.doc.item("t-p1").body,
+ [" - Steps: a", " - found the cause", " - After: [[t-p0]]", " Ref: docs/a.md"])
+
+ def test_set_body_keeps_after_and_ref(self):
+ T.set_body(self.doc, "t-p1", ["- Steps: b", " - sub", "", "- Done: c"])
+ self.assertEqual(self.doc.item("t-p1").body,
+ [" - Steps: b", " - sub", "", " - Done: c", " - After: [[t-p0]]", " Ref: docs/a.md"])
+
+ def test_set_body_bare_after_ids_become_links(self):
+ T.set_body(self.doc, "t-p1", ["- Done: c", "- After: t-a, [[t-b]]"])
+ self.assertEqual(self.doc.item("t-p1").body[:2], [" - Done: c", " - After: [[t-a]], [[t-b]]"])
+
+ def test_tick_by_number_and_text(self):
+ self.assertEqual(T.tick(self.doc, "t-play", "1"), "keyboard works")
+ self.assertEqual(T.tick(self.doc, "t-play", "gait"), "gait feels right")
+ self.assertEqual(self.doc.item("t-play").body,
+ [" - [x] keyboard works", " - [x] motion smooth", " - [x] gait feels right"])
+
+ def test_tick_errors(self):
+ with self.assertRaisesRegex(T.TaskError, "already ticked"):
+ T.tick(self.doc, "t-play", "2")
+ with self.assertRaisesRegex(T.TaskError, "no box matches 'jump'"):
+ T.tick(self.doc, "t-play", "jump")
+ with self.assertRaisesRegex(T.TaskError, "no box 7"):
+ T.tick(self.doc, "t-play", "7")
+
+ def test_unblock(self):
+ self.assertEqual(T.unblock(self.doc, "a-key"), ["t-p2"])
+ self.assertIsNone(self.doc.item("t-p2").status)
+
+
+class PickTest(unittest.TestCase):
+ def test_skips_blocked_and_open_dependencies(self):
+ doc = T.parse(EDIT)
+ T.remove(doc, "t-p0")
+ T.remove(doc, "t-p1")
+ item, skipped = L.pick(doc, set(), L.DEFAULT_LANES, "1h")
+ self.assertIsNone(item)
+ self.assertEqual([(i.id, why) for i, why in skipped],
+ [("t-p2", "blocked: a-key"), ("t-p3", "after: t-gone")])
+
+ def test_archived_dependency_counts_as_done(self):
+ doc = T.parse(EDIT)
+ T.remove(doc, "t-p0")
+ T.remove(doc, "t-p1")
+ item, skipped = L.pick(doc, {"t-gone"}, L.DEFAULT_LANES, "1h")
+ self.assertEqual(item.id, "t-p3")
+ self.assertEqual([i.id for i, _ in skipped], ["t-p2"])
+
+ def test_first_item_when_free(self):
+ item, skipped = L.pick(T.parse(EDIT), set(), L.DEFAULT_LANES, "1h")
+ self.assertEqual((item.id, skipped), ("t-p0", []))
+
+ def test_open_dependency_in_file(self):
+ doc = T.parse(EDIT)
+ T.move_to(doc, "t-p0", "deferred")
+ item, skipped = L.pick(doc, set(), L.DEFAULT_LANES, "1h")
+ self.assertIsNone(item)
+ self.assertEqual(skipped[0][1], "after: t-p0")
+
+
+class SliceTest(unittest.TestCase):
+ def test_open_slices(self):
+ doc = T.parse("## Pending\n\n- **t-p** [P1] (10h): Parent.\n\n- **t-p-2** [P1] (1h): Two.\n\n"
+ "- **t-px-1** [P1] (1h): Other.\n")
+ self.assertEqual(T.open_slices(doc, "t-p"), ["t-p-2"])
+
+ def test_open_slices_by_name(self):
+ doc = T.parse(SLICED)
+ self.assertEqual(T.open_slices(doc, "t-p"), ["t-map", "t-p-3"])
+ self.assertEqual(T.parent_of(doc, "t-map"), "t-p")
+ self.assertEqual(T.parent_of(doc, "t-p-3"), "t-p")
+ self.assertIsNone(T.parent_of(doc, "t-q"))
+
+ def test_pick_skips_parent_with_open_slices(self):
+ item, skipped = L.pick(T.parse(SLICED), {"t-done"}, L.DEFAULT_LANES, "1h")
+ self.assertEqual(item.id, "t-q")
+ self.assertEqual([(i.id, why) for i, why in skipped],
+ [("t-map", "after: t-x"), ("t-p-3", "after: t-map"), ("t-p", "open slices: t-map, t-p-3")])
+
+
+class RenameTest(unittest.TestCase):
+ def test_rename_changes_id_and_links(self):
+ doc = T.parse(SLICED)
+ n = T.rename(doc, "t-map", "t-map-scaffold", taken={"t-old"})
+ self.assertEqual(n, 2)
+ self.assertEqual(T.render(doc), SLICED.replace("t-map", "t-map-scaffold"))
+
+ def test_rename_awaiting_updates_blocked_status(self):
+ doc = T.parse("## Awaiting your decision\n\n- **a-a-foo**: Foo?\n\n## Pending\n\n"
+ "- **t-a** [P1] (1h) (blocked: [[a-a-foo]]): A.\n")
+ self.assertEqual(T.rename(doc, "a-a-foo", "a-foo", taken=set()), 1)
+ self.assertEqual(doc.item("t-a").blocked_on, "a-foo")
+
+ def test_rename_refusals(self):
+ doc = T.parse(SLICED)
+ for new, err in [("t-q", "id 't-q' already exists"), ("t-old", "id 't-old' already used in the archive"),
+ ("a-map", "a rename keeps the kind"), ("t-Bad", "bad id 't-Bad'")]:
+ with self.assertRaisesRegex(T.TaskError, err):
+ T.rename(doc, "t-map", new, taken={"t-old"})
+ with self.assertRaisesRegex(T.TaskError, "unknown id 't-none'"):
+ T.rename(doc, "t-none", "t-x", taken=set())
+ self.assertEqual(T.render(doc), SLICED)
+
+
+SLICED = """\
+## Pending
+
+- **t-p** [P1] (10h): Parent.
+ - Slices: [[t-done]], [[t-map]], [[t-p-3]]
+
+- **t-map** [P1] (1h): Map.
+ - After: [[t-x]]
+
+- **t-p-3** [P1] (1h): Three.
+ - After: [[t-map]]
+
+- **t-q** [P2] (1h): Free.
+"""
+
+
+class ArchiveTest(unittest.TestCase):
+ def test_line(self):
+ item = T.Item(id="t-x", prio=1, effort="1h", text="Cave seams. Close the slit.")
+ self.assertEqual(T.archive_line("2026-09-29", item, "closed with side walls"),
+ "- 2026-09-29 **t-x** Cave seams — closed with side walls")
+
+ def test_prepend(self):
+ self.assertEqual(T.archive_prepend("# Archive (newest first)\n\n- old one\n- older\n", "- new"),
+ "# Archive (newest first)\n\n- new\n- old one\n- older\n")
+
+ def test_prepend_to_empty_list(self):
+ self.assertEqual(T.archive_prepend("# Archive (newest first)\n", "- new"),
+ "# Archive (newest first)\n\n- new\n")
+
+ def test_ids(self):
+ text = "# A\n\n- 2026-09-29 **t-x** X — done\n- old line without id\n- 2026-09-01 **t-y-2** Y — z\n"
+ self.assertEqual(T.archive_ids(text), {"t-x", "t-y-2"})
+
+
+class ParseBlockTest(unittest.TestCase):
+ def test_block_is_reindented(self):
+ item = T.parse_block("- **t-n** [P2] (1h): New. Goal.\n - Steps: a\n - sub\n Ref: docs/a.md\n")
+ self.assertEqual(item.lines(), ["- **t-n** [P2] (1h): New. Goal.", " - Steps: a", " - sub", " Ref: docs/a.md"])
+
+ def test_old_numbered_format_is_refused(self):
+ with self.assertRaisesRegex(T.TaskError, "old numbered format"):
+ T.parse_block("3. **[P2] Old** (Effort: 1h) — goal.")
+
+ def test_bad_header_is_refused(self):
+ with self.assertRaisesRegex(T.TaskError, "bad header"):
+ T.parse_block("- **t-n** no colon")
diff --git a/tests/test_usage.py b/tests/test_usage.py
new file mode 100644
index 0000000..f2d55e3
--- /dev/null
+++ b/tests/test_usage.py
@@ -0,0 +1,337 @@
+import json
+import os
+import subprocess
+import sys
+import tempfile
+import unittest
+from pathlib import Path
+
+HERE = Path(__file__).resolve().parent.parent
+WF = HERE / "wf.py"
+sys.path.insert(0, str(HERE))
+
+from wflib import usage # noqa: E402
+
+
+def entry(mid, req, model, ts, inp=0, cw5=0, cw1h=0, cr=0, out=0, type="assistant", stop=None, content=None,
+ uuid=None):
+ u = {"input_tokens": inp, "cache_creation_input_tokens": cw5 + cw1h, "cache_read_input_tokens": cr,
+ "output_tokens": out}
+ if cw1h:
+ u["cache_creation"] = {"ephemeral_5m_input_tokens": cw5, "ephemeral_1h_input_tokens": cw1h}
+ d = {"type": type, "requestId": req, "timestamp": ts,
+ "message": {"id": mid, "model": model, "role": "assistant", "usage": u}}
+ if stop:
+ d["message"]["stop_reason"] = stop
+ if content is not None:
+ d["message"]["content"] = content
+ if uuid:
+ d["uuid"] = uuid
+ return json.dumps(d)
+
+
+# Subagent transcripts: stop_reason always null, output_tokens = message_start placeholder.
+# e1: thinking (signature 1800 chars) + text 1000 chars + tool_use (input json 100 chars), placeholder 8.
+# by hand: 0.3*(1800-800) + 0.3*1000 + (30 + 0.44*100) = 300 + 300 + 74 = 674
+# e2: same tool_use again but stop_reason set (final): reported 50 is trusted.
+# e3: no content, placeholder 9 kept (nothing to estimate).
+TOOL = {"type": "tool_use", "id": "t1", "name": "Bash", "input": {"a": "y" * 91}} # json.dumps → 100 chars
+SUB = "\n".join([
+ entry("e1", "q1", "claude-sonnet-5-5", "2026-10-04T10:00:00Z", out=8, uuid="u1",
+ content=[{"type": "thinking", "thinking": "", "signature": "z" * 1800}]),
+ entry("e1", "q1", "claude-sonnet-5-5", "2026-10-04T10:00:01Z", out=8, uuid="u2",
+ content=[{"type": "text", "text": "x" * 1000}]),
+ entry("e1", "q1", "claude-sonnet-5-5", "2026-10-04T10:00:01Z", out=8, uuid="u2",
+ content=[{"type": "text", "text": "x" * 1000}]), # duplicated line: counted once
+ entry("e1", "q1", "claude-sonnet-5-5", "2026-10-04T10:00:02Z", out=8, uuid="u3", content=[TOOL]),
+ entry("e2", "q2", "claude-sonnet-5-5", "2026-10-04T10:01:00Z", out=50, stop="tool_use", content=[TOOL]),
+ entry("e3", "q3", "claude-sonnet-5-5", "2026-10-04T10:02:00Z", out=9),
+]) + "\n"
+
+
+# main session, opus 5.5: m1 streamed as 2 entries (same usage), m2 as 2 entries (out 4000 then 10000).
+# by hand: m1 = 1000*4 + 100000*5 + 2000*20 = 544000 µ$; m2 = 50000*8 + 1e6*0.2 + 10000*20 = 800000 µ$
+MAIN = "\n".join([
+ json.dumps({"type": "user", "timestamp": "2026-10-04T09:00:00Z", "message": {"role": "user", "content": "hi"}}),
+ entry("m1", "r1", "claude-opus-5-5", "2026-10-04T09:00:01Z", inp=1000, cw5=100000, out=2000),
+ entry("m1", "r1", "claude-opus-5-5", "2026-10-04T09:00:02Z", inp=1000, cw5=100000, out=2000),
+ "not json",
+ json.dumps({"type": "attachment", "timestamp": "2026-10-04T09:00:03Z"}),
+ entry("m2", "r2", "claude-opus-5-5", "2026-10-04T09:01:00Z", cw1h=50000, cr=1000000, out=4000),
+ entry("m2", "r2", "claude-opus-5-5", "2026-10-04T09:01:05Z", cw1h=50000, cr=1000000, out=10000),
+ entry("m9", "r9", "<synthetic>", "2026-10-04T09:02:00Z", out=5),
+]) + "\n"
+
+# subagent, sonnet 5.5: s1 = 10*2 + 20000*2.5 + 30000*0.2 + 500*10 = 61020 µ$; s2 = 50000*0.2 + 1500*10 = 25000 µ$
+AGENT = "\n".join([
+ entry("s1", "q1", "claude-sonnet-5-5", "2026-10-04T10:00:00Z", inp=10, cw5=20000, cr=30000, out=500),
+ entry("s2", "q2", "claude-sonnet-5-5", "2026-10-04T12:00:00Z", cr=50000, out=1500),
+ entry("s3", "q3", "claude-mystery-1", "2026-10-04T12:30:00Z", inp=7, out=3),
+]) + "\n"
+
+SID = "11111111-2222-3333-4444-555555555555"
+
+
+class Parse(unittest.TestCase):
+ def test_dedupes_streamed_entries_and_takes_last_output(self):
+ got = usage.parse(MAIN)
+ self.assertEqual(list(got), ["claude-opus-5-5"])
+ u = got["claude-opus-5-5"]
+ self.assertEqual((u.turns, u.inp, u.cw5, u.cw1h, u.cr, u.out), (2, 1000, 100000, 50000, 1000000, 12000))
+ self.assertEqual(u.cw, 150000)
+ self.assertAlmostEqual(usage.cost("claude-opus-5-5", u), 1.344)
+
+ def test_models_kept_apart_unknown_price_none(self):
+ got = usage.parse(AGENT)
+ s = got["claude-sonnet-5-5"]
+ self.assertEqual((s.turns, s.cr, s.out), (2, 80000, 2000))
+ self.assertAlmostEqual(usage.cost("claude-sonnet-5-5", s), 0.08602)
+ self.assertIsNone(usage.cost("claude-mystery-1", got["claude-mystery-1"]))
+
+ def test_since_until_filter_on_utc_timestamp(self):
+ s = usage.parse(AGENT, since="2026-10-04T11:00", until="2026-10-04T12:10")["claude-sonnet-5-5"]
+ self.assertEqual((s.turns, s.cr), (1, 50000))
+ self.assertAlmostEqual(usage.cost("claude-sonnet-5-5", s), 0.025)
+
+ def test_dated_model_id_priced_by_prefix(self):
+ u = usage.Usage(turns=1, inp=1000000, out=1000000)
+ self.assertAlmostEqual(usage.cost("claude-haiku-4-5-20251001", u), 6.0)
+ self.assertAlmostEqual(usage.cost("claude-opus-5", u), 30.0)
+
+ def test_subagent_output_estimated_from_content(self):
+ u = usage.parse(SUB)["claude-sonnet-5-5"]
+ self.assertEqual((u.turns, u.out, u.est), (3, 674 + 50 + 9, 1))
+
+ def test_log_line_marks_estimate(self):
+ line = usage.log_line("T", "p", "t-x", "1h", "done", "a1", usage.parse(SUB))
+ self.assertIn(" out=733 est=1 usd=", line)
+ self.assertNotIn(" est=", usage.log_line("T", "p", "t-x", "1h", "done", "a1", usage.parse(AGENT)))
+
+ def test_log_line_lane_and_model(self):
+ by = usage.parse(SUB)
+ self.assertIn(" lane=fast model=sonnet effort=", usage.log_line("T", "p", "t-x", "1h", "done", "a1", by, lane="fast"))
+ self.assertIn(" lane=sonnet model=sonnet effort=", usage.log_line("T", "p", "t-x", "1h", "done", "a1", by))
+
+ def test_report_lane_model_key(self):
+ es = [{"lane": "fast", "model": "sonnet", "outcome": "done", "usd": 1.0, "turns": 10, "effort": "1h"},
+ {"lane": "fast", "model": "opus", "outcome": "done", "usd": 2.0, "turns": 20, "effort": "1h"},
+ {"lane": "opus", "outcome": "done", "usd": 3.0, "turns": 30, "effort": "1h"}]
+ self.assertEqual(usage.report(es, ("1h",)), [
+ ("fast/opus", "all", 1, 1, 0, 2.0, 2.0, 2.0, 20, None), ("fast/opus", "1h", 1, 1, 0, 2.0, 2.0, 2.0, 20, None),
+ ("fast/sonnet", "all", 1, 1, 0, 1.0, 1.0, 1.0, 10, None), ("fast/sonnet", "1h", 1, 1, 0, 1.0, 1.0, 1.0, 10, None),
+ ("opus", "all", 1, 1, 0, 3.0, 3.0, 3.0, 30, None), ("opus", "1h", 1, 1, 0, 3.0, 3.0, 3.0, 30, None)])
+
+ def test_fmt(self):
+ self.assertEqual([usage.fmt(n) for n in (999, 1500, 8948413)], ["999", "1.5k", "8.95M"])
+
+
+def tcall(mid, ctx, name, inp, tid):
+ return json.dumps({"type": "assistant", "timestamp": "2026-10-04T10:00:00Z", "message": {
+ "id": mid, "model": "claude-sonnet-5-5", "usage": {"input_tokens": 10, "cache_read_input_tokens": ctx - 10},
+ "content": [{"type": "tool_use", "id": tid, "name": name, "input": inp}]}})
+
+
+def tres(tid, text):
+ return json.dumps({"type": "user", "message": {"role": "user", "content": [
+ {"type": "tool_result", "tool_use_id": tid, "content": text}]}})
+
+
+# agent A: ctx 1000 Read(400 chars=100 tok), 2000 grep(800=200), 3000 Edit, 4000 git commit.
+# exp 3000 / mut 3000 / book 4000 of 10000; first edit after 2 calls, pre 3000 = 30%, context +2000
+EXA = "\n".join([
+ tcall("a1", 1000, "Read", {"file_path": "/w/proj/.worktrees/x/src/a.py"}, "t1"), tres("t1", "x" * 400),
+ tcall("a2", 2000, "Bash", {"command": "grep -n foo src/a.py 2>/dev/null"}, "t2"), tres("t2", "y" * 800),
+ tcall("a3", 3000, "Edit", {"file_path": "/w/proj/src/a.py"}, "t3"), tres("t3", "ok"),
+ tcall("a4", 4000, "Bash", {"command": "git commit -m x"}, "t4"), tres("t4", "done"),
+]) + "\n"
+# agent B: 4 exploring calls of 1000 each, reads src/a.py (100 tok) + sed -n of b.py (400 chars)
+EXB = "\n".join([
+ tcall("b1", 1000, "Read", {"file_path": "/w/proj/src/a.py"}, "u1"), tres("u1", "x" * 400),
+ tcall("b2", 1000, "Bash", {"command": "sed -n 1,9p src/b.py"}, "u2"), tres("u2", "z" * 400),
+ tcall("b3", 1000, "Bash", {"command": "ls"}, "u3"), tres("u3", "q"),
+ tcall("b4", 1000, "Glob", {}, "u4"), tres("u4", "q"),
+]) + "\n"
+
+
+class Explore(unittest.TestCase):
+ def test_one_agent(self):
+ e = usage.explore(EXA, root="/w/proj")
+ self.assertEqual((e.calls, e.first_edit, e.pre, e.ctx_growth), (4, 2, 3000, 2000))
+ self.assertEqual(e.cost, {"exp": 3000, "mut": 3000, "book": 4000})
+ self.assertEqual(e.results, {"Read": 100, "grep": 200, "Edit": 0, "other": 1})
+ self.assertEqual(e.files, {"src/a.py": 300})
+
+ def test_call_kind_book_is_writes_only(self):
+ kind = lambda c: usage.call_kind({"name": "Bash", "input": {"command": c}})
+ for c in ("git worktree list", "git branch --list 'fast/*'", "git branch --show-current", "git status --short"):
+ self.assertEqual(kind(c), "exp", c)
+ for c in ("git worktree add .w/x -b x master", "git worktree remove .w/x", "git branch -d x",
+ "python3 /projects/public/workflow/wf.py finish t-x -m ok", "git switch -c x"):
+ self.assertEqual(kind(c), "book", c)
+
+ def test_since_drops_early_calls(self):
+ e = usage.explore(EXA.replace("10:00:00Z", "09:00:00Z", 2), since="2026-10-04T10")
+ self.assertEqual(e.calls, 2)
+
+ def test_report(self):
+ a, b = usage.explore(EXA, root="/w/proj"), usage.explore(EXB, root="/w/proj")
+ r = usage.explore_report([("A", a), ("B", b)], min_pre=4)
+ self.assertEqual(r["rows"], [("A", 4, 2, 0.3, 0.3), ("B", 4, 4, 1.0, 0.0)])
+ self.assertEqual((r["total"], r["cost"]["exp"]), (14000, 7000))
+ self.assertEqual((r["pre_n"], r["pre_share"], r["pre_calls"], r["pre_ctx"]), (1, 0.3, 2, 2000))
+ self.assertEqual(r["results"]["sed-cat"], 100)
+ self.assertEqual(r["files"], [("src/a.py", 2, 400)])
+ self.assertEqual(usage.explore_report([("A", a)], min_calls=5)["rows"], [])
+
+
+class Cli(unittest.TestCase):
+ def setUp(self):
+ self.tmp = tempfile.TemporaryDirectory()
+ self.home = Path(self.tmp.name)
+ proj = self.home / "projects" / "-work-demo"
+ sub = proj / SID / "subagents"
+ sub.mkdir(parents=True)
+ (proj / f"{SID}.jsonl").write_text(MAIN)
+ (sub / "agent-abc123.jsonl").write_text(AGENT)
+ (sub / "agent-abc123.meta.json").write_text(json.dumps({"description": "wf-worker sonnet t-x"}))
+
+ def tearDown(self):
+ self.tmp.cleanup()
+
+ def wf(self, *args, sid="", cwd=None):
+ env = {**os.environ, "CLAUDE_CONFIG_DIR": str(self.home), "CLAUDE_CODE_SESSION_ID": sid}
+ r = subprocess.run([sys.executable, str(WF), "usage", *args], capture_output=True, text=True,
+ cwd=cwd or self.tmp.name, env=env, timeout=30)
+ return r.returncode, r.stdout, r.stderr
+
+ def test_explore_skips_main_and_short_agents(self):
+ code, out, err = self.wf("--session", SID[:8], "--explore")
+ self.assertEqual((code, err), (0, ""))
+ self.assertEqual(out.strip(), "no subagent with >= 4 calls")
+
+ def test_session_rows_and_total(self):
+ code, out, err = self.wf("--session", SID[:8])
+ self.assertEqual((code, err), (0, ""))
+ rows = [l.split() for l in out.strip().split("\n")[1:]]
+ self.assertEqual(rows[0], ["main", "opus-5-5", "2", "1.0k", "150.0k", "1.00M", "12.0k", "1.34"])
+ self.assertEqual(rows[1], ["abc123", "wf-worker", "sonnet", "t-x",
+ "sonnet-5-5", "2", "10", "20.0k", "80.0k", "2.0k", "0.09"])
+ self.assertEqual(rows[2][-7:], ["mystery-1", "1", "7", "0", "0", "3", "?"])
+ self.assertEqual(rows[3], ["total", "5", "1.0k", "170.0k", "1.08M", "14.0k", "1.43+"])
+
+ def test_session_from_env(self):
+ code, out, _ = self.wf(sid=SID)
+ self.assertEqual(code, 0)
+ self.assertIn("abc123", out)
+
+ def test_agent_only(self):
+ code, out, err = self.wf("--agent", "agent-abc123", "--since", "2026-10-04T11:00")
+ self.assertEqual((code, err), (0, ""))
+ lines = out.strip().split("\n")
+ self.assertNotIn("main", out)
+ self.assertEqual(lines[1].split()[-7:], ["sonnet-5-5", "1", "0", "0", "50.0k", "1.5k", "0.03"])
+
+ def test_log_appends_one_key_value_line(self):
+ root = self.home / "demo"
+ root.mkdir()
+ (root / "workflow.toml").write_text("format = 1\n")
+ code, out, err = self.wf("--agent", "abc123", "--log", "t-x", "done", "--effort", "1h", cwd=root)
+ self.assertEqual((code, err), (0, ""))
+ code, out, err = self.wf("--agent", "abc123", "--log", "t-x", "handback", cwd=root)
+ log = (root / "out" / "wf-cost.log").read_text().split("\n")
+ self.assertEqual(len(log), 3) # two lines + trailing newline
+ # lane = model with most turns (sonnet 2 vs mystery 1); usd = known models only
+ self.assertRegex(log[0], r"^\d{4}-\d\d-\d\dT\d\d:\d\d:\d\dZ project=demo task=t-x lane=sonnet model=sonnet effort=1h "
+ r"outcome=done turns=3 in=17 cw=20000 cr=80000 out=2003 usd=0\.0860 agent=abc123$")
+ self.assertIn(" effort=- outcome=handback ", log[1])
+ self.assertEqual(out, log[1] + "\n")
+
+ def test_estimated_out_marked_with_tilde(self):
+ other = self.home / "projects" / "-work-demo" / "other-session" / "subagents"
+ other.mkdir(parents=True)
+ (other / "agent-est1.jsonl").write_text(SUB)
+ code, out, err = self.wf("--agent", "est1")
+ self.assertEqual((code, err), (0, ""))
+ self.assertEqual(out.strip().split("\n")[1].split()[-3:], ["0", "~733", "0.01"])
+
+ def test_log_needs_agent(self):
+ code, out, err = self.wf("--log", "t-x", "done")
+ self.assertEqual((code, err), (2, "wf: --log needs --agent\n"))
+
+ def test_log_effort_must_be_an_estimate(self):
+ code, out, err = self.wf("--agent", "abc123", "--log", "t-x", "done", "--effort", "medium")
+ self.assertEqual(code, 2)
+ self.assertIn("invalid choice: 'medium'", err)
+
+ def test_missing_agent_one_line_error(self):
+ code, out, err = self.wf("--agent", "nope")
+ self.assertEqual((code, out), (1, ""))
+ self.assertEqual(err, "wf: no transcript for agent nope\n")
+
+
+LOG_A = """\
+2026-10-01T10:00:00Z project=a task=t-1 lane=sonnet effort=1h outcome=done turns=10 usd=0.50
+2026-10-01T11:00:00Z project=a task=t-2 lane=sonnet effort=1h outcome=handback turns=30 usd=1.50
+garbage line
+"""
+LOG_B = """\
+2026-10-02T10:00:00Z project=b task=t-3 lane=sonnet effort=<1h outcome=done turns=20 usd=0.40
+2026-10-02T11:00:00Z project=b task=t-4 lane=opus effort=5h outcome=done turns=80 usd=3.00
+2026-10-03T11:00:00Z project=b task=t-5 lane=opus effort=1h outcome=done turns=40 usd=2.00
+"""
+
+
+class Report(unittest.TestCase):
+ def test_rows_by_lane_then_effort(self):
+ entries = usage.parse_log(LOG_A + LOG_B)
+ self.assertEqual(len(entries), 5)
+ rows = usage.report(entries, ("<1h", "1h", "5h"))
+ # lane, effort, n, done, other, median $, total $, $/done, median turns — all by hand
+ self.assertEqual(rows, [
+ ("opus", "all", 2, 2, 0, 2.5, 5.0, 2.5, 60, None),
+ ("opus", "1h", 1, 1, 0, 2.0, 2.0, 2.0, 40, None),
+ ("opus", "5h", 1, 1, 0, 3.0, 3.0, 3.0, 80, None),
+ ("sonnet", "all", 3, 2, 1, 0.5, 2.4, 1.2, 20, None),
+ ("sonnet", "<1h", 1, 1, 0, 0.4, 0.4, 0.4, 20, None),
+ ("sonnet", "1h", 2, 1, 1, 1.0, 2.0, 2.0, 20, None),
+ ])
+
+ def test_duration(self):
+ line = usage.log_line("T", "p", "t-x", "1h", "done", "a1", usage.parse(SUB), dur=125)
+ self.assertIn(" dur=125 agent=a1", line)
+ self.assertNotIn("dur=", usage.log_line("T", "p", "t-x", "1h", "done", "a1", usage.parse(SUB)))
+ log = ("2026-10-04T10:00:00Z project=a task=t-6 lane=fast effort=<1h outcome=done turns=1 usd=1 dur=100\n"
+ "2026-10-04T10:00:00Z project=a task=t-7 lane=fast effort=<1h outcome=done turns=1 usd=1 dur=300\n"
+ "2026-10-04T10:00:00Z project=a task=t-8 lane=fast effort=<1h outcome=done turns=1 usd=1\n")
+ self.assertEqual(usage.report(usage.parse_log(log), ("<1h",))[0][9], 200)
+
+ def test_done_gate_red_counts_as_done(self):
+ log = ("2026-10-04T10:00:00Z project=a task=t-6 lane=fast effort=<1h outcome=done+gate-red turns=12 usd=1.00\n"
+ "2026-10-04T11:00:00Z project=a task=t-7 lane=fast effort=<1h outcome=donex turns=8 usd=0.50\n")
+ rows = usage.report(usage.parse_log(log), ("<1h",))
+ self.assertEqual(rows[0], ("fast", "all", 2, 1, 1, 0.75, 1.5, 1.5, 10, None))
+
+ def test_since_and_no_done(self):
+ entries = usage.parse_log(LOG_A + LOG_B, since="2026-10-01T10:30")
+ rows = usage.report(entries, ("1h",))
+ self.assertIn(("sonnet", "all", 2, 1, 1, 0.95, 1.9, 1.9, 25, None), rows)
+ rows = usage.report(usage.parse_log(LOG_A, since="2026-10-01T10:30"), ("1h",))
+ self.assertEqual(rows[0], ("sonnet", "all", 1, 0, 1, 1.5, 1.5, None, 30, None))
+
+ def test_cli_reads_every_project(self):
+ with tempfile.TemporaryDirectory() as tmp:
+ for name, text in (("a", LOG_A), ("b", LOG_B)):
+ (Path(tmp) / name / "out").mkdir(parents=True)
+ (Path(tmp) / name / "workflow.toml").write_text("format = 1\n")
+ (Path(tmp) / name / "out" / "wf-cost.log").write_text(text)
+ r = subprocess.run([sys.executable, str(WF), "usage", "--report"], capture_output=True, text=True,
+ cwd=tmp, env={**os.environ, "WF_ROOT": tmp}, timeout=30)
+ self.assertEqual((r.returncode, r.stderr), (0, ""))
+ lines = [l.split() for l in r.stdout.strip().split("\n")]
+ self.assertEqual(lines[0], ["lane", "effort", "n", "done", "other", "med$", "total$", "$/done", "med_turns", "med_dur"])
+ self.assertEqual(lines[1], ["opus", "all", "2", "2", "0", "2.50", "5.00", "2.50", "60", "-"])
+ self.assertEqual(lines[6], ["sonnet", "1h", "2", "1", "1", "1.00", "2.00", "2.00", "20", "-"])
+
+
+if __name__ == "__main__":
+ unittest.main()