diff options
| author | godosa <godosa@godosa.eu> | 2026-10-07 07:27:17 +0200 |
|---|---|---|
| committer | godosa <godosa@godosa.eu> | 2026-10-07 07:27:17 +0200 |
| commit | 81d4e80fd5aabe4e80f58e960affa795cf7d34ec (patch) | |
| tree | e98eeac2af6af63aa4287bba1f6d4a3af26b5727 /tests | |
| download | workflow-81d4e80fd5aabe4e80f58e960affa795cf7d34ec.tar.gz workflow-81d4e80fd5aabe4e80f58e960affa795cf7d34ec.zip | |
workflow: initial public history
Diffstat (limited to 'tests')
| -rw-r--r-- | tests/test_areas.py | 319 | ||||
| -rw-r--r-- | tests/test_batch.py | 534 | ||||
| -rw-r--r-- | tests/test_check.py | 304 | ||||
| -rw-r--r-- | tests/test_claims.py | 182 | ||||
| -rw-r--r-- | tests/test_cli.py | 752 | ||||
| -rw-r--r-- | tests/test_cloud.py | 1174 | ||||
| -rw-r--r-- | tests/test_config.py | 134 | ||||
| -rw-r--r-- | tests/test_ctx_hint.py | 98 | ||||
| -rw-r--r-- | tests/test_finish.py | 210 | ||||
| -rw-r--r-- | tests/test_gate.py | 74 | ||||
| -rw-r--r-- | tests/test_human_done.py | 40 | ||||
| -rw-r--r-- | tests/test_lanes.py | 408 | ||||
| -rw-r--r-- | tests/test_ledgers.py | 26 | ||||
| -rw-r--r-- | tests/test_lock.py | 30 | ||||
| -rw-r--r-- | tests/test_merge.py | 159 | ||||
| -rw-r--r-- | tests/test_migrate.py | 160 | ||||
| -rw-r--r-- | tests/test_model.py | 247 | ||||
| -rw-r--r-- | tests/test_orch.py | 227 | ||||
| -rw-r--r-- | tests/test_publish_snapshot.py | 122 | ||||
| -rw-r--r-- | tests/test_refs.py | 109 | ||||
| -rw-r--r-- | tests/test_res.py | 751 | ||||
| -rw-r--r-- | tests/test_res_io.py | 815 | ||||
| -rw-r--r-- | tests/test_runner.py | 81 | ||||
| -rw-r--r-- | tests/test_search.py | 78 | ||||
| -rw-r--r-- | tests/test_sessions_field.py | 197 | ||||
| -rw-r--r-- | tests/test_setup.py | 164 | ||||
| -rw-r--r-- | tests/test_start.py | 109 | ||||
| -rw-r--r-- | tests/test_tasks.py | 523 | ||||
| -rw-r--r-- | tests/test_usage.py | 337 |
29 files changed, 8364 insertions, 0 deletions
diff --git a/tests/test_areas.py b/tests/test_areas.py new file mode 100644 index 0000000..558230d --- /dev/null +++ b/tests/test_areas.py @@ -0,0 +1,319 @@ +import subprocess +import tempfile +import unittest +from pathlib import Path + +from test_cli import Cli +from test_claims import git +from test_merge import IDENT +from wflib import areas # noqa: E402 (tests/ run with the repo root on sys.path) + +NOTES = """\ +# CLAUDE.md + +## Areas +### Parser +- Code map: `parse_item`, `NOPE_GONE` +- Test recipe: python3 -m unittest tests.test_p +- Paths: src/ +- Checked: {sha} + +### Docs +- Code map: `intro` + +## Gotchas +- x +""" + + +def build(root: Path) -> tuple[str, str]: + """Repo with src/p.py + docs.md, then 3 commits touching src/; returns (sha1, notes).""" + git(root, "init", "-q", "-b", "master") + (root / "src").mkdir(exist_ok=True) + (root / "src" / "p.py").write_text("def parse_item(): pass\n") + (root / "docs.md").write_text("intro\n") + (root / ".gitignore").write_text(".wf/\n") + git(root, "add", "-A") + git(root, "-c", "user.name=t", "-c", "user.email=t@t", "commit", "-qm", "c1") + sha1 = subprocess.run(["git", "rev-parse", "--short", "HEAD"], cwd=root, capture_output=True, text=True).stdout.strip() + for n in "abc": + (root / "src" / f"{n}.txt").write_text(n) + git(root, "add", "-A") + git(root, "-c", "user.name=t", "-c", "user.email=t@t", "commit", "-qm", n) + return sha1, NOTES.format(sha=sha1) + + +class AreasTest(unittest.TestCase): + def setUp(self): + self.tmp = tempfile.TemporaryDirectory() + self.root = Path(self.tmp.name).resolve() + self.sha1, self.notes = build(self.root) + + def tearDown(self): + self.tmp.cleanup() + + def test_parse(self): + a = areas.parse(NOTES.format(sha="abc1234")) + self.assertEqual([(x.name, x.slug, x.anchors, x.paths, x.checked) for x in a], [ + ("Parser", "parser", ["parse_item", "NOPE_GONE"], ["src/"], "abc1234"), + ("Docs", "docs", ["intro"], [], None)]) + + def test_missing_and_commits(self): + parser, docs = areas.parse(self.notes) + self.assertEqual(areas.missing(self.root, parser), ["NOPE_GONE"]) + self.assertEqual(areas.commits_since(self.root, parser), 3) + self.assertEqual(areas.missing(self.root, docs), []) + self.assertIsNone(areas.commits_since(self.root, docs)) + + def test_stale(self): + parser, docs = areas.parse(self.notes) + self.assertEqual(areas.status(self.root, parser, 20), (True, ["NOPE_GONE"], 3)) + self.assertEqual(areas.status(self.root, docs, 20), (False, [], None)) + fixed = areas.parse(self.notes.replace(", `NOPE_GONE`", ""))[0] + self.assertEqual(areas.status(self.root, fixed, 3), (True, [], 3)) + self.assertEqual(areas.status(self.root, fixed, 4), (False, [], 3)) + + def test_unknown_checked_sha(self): + a = areas.parse(NOTES.format(sha="deadbee"))[0] + self.assertIsNone(areas.commits_since(self.root, a)) + + def test_mark(self): + out = areas.mark(NOTES.format(sha="abc1234"), "Docs", "f00ba44") + self.assertIn("### Docs\n- Code map: `intro`\n- Checked: f00ba44\n\n## Gotchas", out) + out = areas.mark(out, "Parser", "1111111") + self.assertIn("- Paths: src/\n- Checked: 1111111\n", out) + + +class MatchTest(unittest.TestCase): + def test_matches(self): + parser, docs = areas.parse(NOTES.format(sha="abc1234")) + self.assertTrue(areas.matches(parser, "Fix src/p.py crash")) + self.assertTrue(areas.matches(parser, "see x.py/src/p.py")) + self.assertTrue(areas.matches(parser, "parse_item() loops")) + self.assertTrue(areas.matches(parser, "the parser area")) + self.assertFalse(areas.matches(parser, "parse_items and srcs")) + self.assertFalse(areas.matches(docs, "introduction")) + self.assertTrue(areas.matches(docs, "docs intro")) + self.assertFalse(areas.matches(parser, "Fix src/p.py crash", {"src"})) + + def test_shared_paths(self): + a = areas.parse("## Areas\n### A\n- Paths: x.py lib/\n### B\n- Paths: ./x.py y.py\n### C\n- Paths: lib\n") + self.assertEqual(areas.shared_paths(a), {"x.py", "lib"}) + + def test_block(self): + text = NOTES.format(sha="abc1234") + self.assertEqual(areas.block(text, "Docs"), "### Docs\n- Code map: `intro`") + self.assertEqual(areas.block(text, "Parser").splitlines()[-1], "- Checked: abc1234") + self.assertEqual(areas.block(text, "X"), "") + + +class CodeRootTest(Cli): + """code_root: area anchors / Checked / staleness live in another repo than the project.""" + + def setUp(self): + super().setUp() + self.code = self.root.parent / "code" + self.code.mkdir() + self.sha1, notes = build(self.code) + (self.root / "CLAUDE.md").write_text(notes) + with open(self.root / "workflow.toml", "a") as f: + f.write('code_root = "../code"\n') + git(self.root, "init", "-q", "-b", "master") + (self.root / "engine").mkdir() + (self.root / "engine" / "x.py").write_text("x") + + def test_areas_use_code_root(self): + self.assertEqual(self.ok("areas"), f"Parser: stale · missing: NOPE_GONE · 3 commits since {self.sha1}\n" + "Docs: ok · no Checked\n") + + def test_mark_uses_code_head(self): + self.ok("areas", "--mark", "Docs") + head = subprocess.run(["git", "rev-parse", "--short", "HEAD"], cwd=self.code, capture_output=True, + text=True).stdout.strip() + self.assertIn(f"- Checked: {head}", (self.root / "CLAUDE.md").read_text()) + + def test_check_uses_code_root(self): + out = self.ok("check") + self.assertIn("area Parser: anchor NOPE_GONE not found", out) + self.assertNotIn("Checked", out) + self.assertNotIn("anchor intro", out) + + def test_no_uncovered_nudge(self): + self.assertNotIn("uncovered", self.ok("areas")) + + +class CtxAreaTest(Cli): + tasks_text = ("# T\n\n## Pending\n\n- **t-a** [P1] (<1h): A. Fix src/p.py.\n\n" + "- **t-b** [P1] (<1h): B. Unrelated.\n") + + def setUp(self): + super().setUp() + (self.root / "CLAUDE.md").write_text(NOTES.format(sha="abc1234")) + + def test_ctx_prints_matching_area(self): + out = self.ok("ctx", "t-a") + self.assertIn("\nArea Parser (CLAUDE.md):\n### Parser\n- Code map: `parse_item`, `NOPE_GONE`\n" + "- Test recipe: python3 -m unittest tests.test_p\n", out) + self.assertNotIn("### Docs", out) + + def test_ctx_no_match(self): + self.assertNotIn("Area ", self.ok("ctx", "t-b")) + + def test_ctx_shared_path_no_match(self): + (self.root / "CLAUDE.md").write_text(NOTES.format(sha="abc1234").replace("- Code map: `intro`", + "- Code map: `intro`\n- Paths: src/")) + self.assertNotIn("Area ", self.ok("ctx", "t-a")) + + +class AreasCliTest(Cli): + def setUp(self): + super().setUp() + sha1, notes = build(self.root) + self.sha1 = sha1 + (self.root / "CLAUDE.md").write_text(notes) + + def test_areas_output(self): + out = self.ok("areas") + self.assertEqual(out, f"Parser: stale · missing: NOPE_GONE · 3 commits since {self.sha1}\n" + "Docs: ok · no Checked\n") + + def test_mark_then_show(self): + self.ok("areas", "--mark", "Docs") + head = subprocess.run(["git", "rev-parse", "--short", "HEAD"], cwd=self.root, capture_output=True, + text=True).stdout.strip() + self.assertEqual(self.ok("areas", "Docs"), f"Docs: ok · 0 commits since {head}\n") + + def test_mark_unique_prefix(self): + self.ok("areas", "--mark", "Do") + head = subprocess.run(["git", "rev-parse", "--short", "HEAD"], cwd=self.root, capture_output=True, + text=True).stdout.strip() + self.assertEqual(self.ok("areas", "Doc"), f"Docs: ok · 0 commits since {head}\n") + + def test_ambiguous_prefix(self): + txt = (self.root / "CLAUDE.md").read_text().replace("### Docs", "### Parser2 (x)") + (self.root / "CLAUDE.md").write_text(txt) + err = self.fails("areas", "--mark", "Par", code=2) + self.assertIn("ambiguous area 'Par'", err) + self.assertIn("Parser, Parser2 (x)", err) + self.ok("areas", "--mark", "Parser") + + def test_unknown_area(self): + err = self.fails("areas", "X", code=2) + self.assertIn("wf: no area 'X' in CLAUDE.md (Parser, Docs)", err) + + def test_check_warnings(self): + out = self.ok("check") + self.assertIn("area Parser: anchor NOPE_GONE not found", out) + (self.root / "CLAUDE.md").write_text(NOTES.format(sha="deadbee").replace("### Parser", "### X")) + self.assertIn("area X: Checked deadbee unknown", self.ok("check")) + + +class AutoTaskTest(Cli): + tasks_text = "# T\n\n## Pending\n\n- **t-a** [P1] (<1h): A. Do.\n\n- **t-b** [P1] (<1h): B. Do.\n" + + def setUp(self): + super().setUp() + sha1, notes = build(self.root) + (self.root / "CLAUDE.md").write_text(notes) + + def test_auto_task(self): + out = self.ok("done", "t-a", "-m", "x") + self.assertIn("added t-map-parser (area map stale)", out) + shown = self.ok("show", "t-map-parser") + self.assertIn("- **t-map-parser** [P1] (<1h): Refresh Parser area map.", shown) + self.assertIn("Steps: wf areas Parser; fix missing anchors + test recipe from git log --stat " + "<Checked>..HEAD -- <paths>; wf areas --mark Parser.", shown) + self.assertIn("Done: wf areas Parser shows ok.", shown) + self.assertIn("Model: sonnet", shown) + + def test_auto_task_once(self): + self.ok("done", "t-a", "-m", "x") + out = self.ok("done", "t-b", "-m", "y") + self.assertNotIn("added", out) + self.assertEqual(self.ok("list").count("t-map-parser"), 1) + + def test_archived_gets_suffix(self): + self.ok("done", "t-a", "-m", "x") + out = self.ok("done", "t-map-parser", "-m", "z") + self.assertIn("added t-map-parser-2 (area map stale)", out) + + def test_no_areas_file(self): + (self.root / "CLAUDE.md").unlink() + self.assertNotIn("added", self.ok("done", "t-a", "-m", "x")) + + +class UncoveredTest(unittest.TestCase): + def test_uncovered_groups_two_levels(self): + a = areas.parse(NOTES.format(sha="abc1234")) + files = ["src/p.py", "lib/x/y/z.py", "lib/x/w.py", "lib/q.py", "README.md", "tests/t.py", + ".github/ci.yml", "TASKS.md"] + self.assertEqual(areas.uncovered(files, a, ["tests"]), [("lib/x", 2), ("lib", 1)]) + + def test_paths_globs_and_files(self): + a = areas.parse("## Areas\n### A\n- Paths: app/*.py tools/run.sh\n") + self.assertEqual(areas.uncovered(["app/m.py", "tools/run.sh", "tools/other.sh"], a, []), [("tools", 1)]) + + def test_no_paths_anywhere_no_nudge(self): + a = areas.parse("## Areas\n### A\n- Code map: `x`\n") + self.assertEqual(areas.uncovered(["lib/q.py"], a, []), []) + + +class AnchorPathsTest(unittest.TestCase): + def test_anchor_folders_stand_in(self): + with tempfile.TemporaryDirectory() as d: + root = Path(d) + build(root) + a = areas.parse("## Areas\n### A\n- Code map: `parse_item`, `intro`\n### B\n- Paths: lib/\n") + self.assertEqual(areas.with_anchor_paths(root, a[0]).paths, ["docs.md", "src"]) + self.assertEqual(areas.with_anchor_paths(root, a[1]).paths, ["lib/"]) + + +class UncoveredCliTest(Cli): + tasks_text = "# T\n\n## Pending\n\n- **t-a** [P1] (<1h): A. Do.\n\n- **t-b** [P1] (<1h): B. Do.\n" + + def setUp(self): + super().setUp() + build(self.root) + notes = NOTES.format(sha="x").replace(", `NOPE_GONE`", "").replace("- Checked: x\n", "") + (self.root / "CLAUDE.md").write_text(notes) + git(self.root, "add", "-A") + git(self.root, "-c", "user.name=t", "-c", "user.email=t@t", "commit", "-qm", "notes") + (self.root / "engine" / "net").mkdir(parents=True) + (self.root / "engine" / "net" / "sock.py").write_text("x") + (self.root / "docs.md").write_text("changed") + + def test_done_adds_map_task(self): + out = self.ok("done", "t-a", "-m", "x") + self.assertIn("added t-map-engine-net (uncovered area: engine/net, 1 file)", out) + shown = self.ok("show", "t-map-engine-net") + self.assertIn("- **t-map-engine-net** [P1] (<1h): Map engine/net area. " + "No area's Paths covers it.", shown) + self.assertIn("Done: wf areas lists the area ok and no longer names engine/net uncovered.", shown) + self.assertIn("Model: sonnet", shown) + self.assertNotIn("added", self.ok("done", "t-b", "-m", "y")) + + def test_areas_names_uncovered(self): + self.assertIn("uncovered: engine/net (1 file)", self.ok("areas")) + + def test_branch_diff_counts(self): + git(self.root, "switch", "-qc", "topic") + git(self.root, "add", "-A") + git(self.root, "-c", "user.name=t", "-c", "user.email=t@t", "commit", "-qm", "work") + self.assertIn("uncovered: engine/net (1 file)", self.ok("areas")) + + def test_no_paths_uses_anchor_folders(self): + (self.root / "CLAUDE.md").write_text("## Areas\n### P\n- Code map: `parse_item`\n") + (self.root / "src" / "n.py").write_text("x") + out = self.ok("areas") + self.assertIn("uncovered: engine/net (1 file)", out) + self.assertNotIn("uncovered: src", out) + + def test_ignore_config(self): + with open(self.root / "workflow.toml", "a") as f: + f.write('area_ignore = ["engine"]\n') + self.assertNotIn("uncovered", self.ok("areas")) + self.assertNotIn("added", self.ok("done", "t-a", "-m", "x")) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_batch.py b/tests/test_batch.py new file mode 100644 index 0000000..2f0e834 --- /dev/null +++ b/tests/test_batch.py @@ -0,0 +1,534 @@ +import io +import shlex +import subprocess +import sys +import tempfile +import threading +import unittest +from contextlib import redirect_stderr, redirect_stdout +from pathlib import Path + +HERE = Path(__file__).resolve().parent.parent +sys.path.insert(0, str(HERE)) +sys.path.insert(0, str(HERE / "tests")) + +import wf_res # noqa: E402 +from test_res_io import Box, hist_rec, seed_history # noqa: E402 + +OUT = "out/wf-batch-2026-10-01-1400.md" + + +class BatchBase(unittest.TestCase): + def setUp(self): + self._tmp = tempfile.TemporaryDirectory() + self.box = Box(Path(self._tmp.name)) + bin_ = self.box.tmp / "bin" + bin_.mkdir() + self.claude = bin_ / "claude" + self.claude.write_text("#!/bin/sh\n") + self.claude.chmod(0o755) + self.box.caller["PATH"] = f"{bin_}:/usr/bin" + + def tearDown(self): + self._tmp.cleanup() + + def batch(self, *argv): + out, err = io.StringIO(), io.StringIO() + with redirect_stdout(out), redirect_stderr(err): + code = wf_res.batch_main(list(argv), self.box.env) + return code, out.getvalue(), err.getvalue() + + def unit_argv(self): + return self.box.fake.ran("systemd-run")[0] + + +class Start(BatchBase): + def test_starts_claude_print_unit(self): + code, out, err = self.batch("2", "--lanes", "fast,slow") + self.assertEqual((code, err), (0, "")) + log = self.box.env.state / "logs" / "r-1.log" + self.assertEqual(out, f"r-1 started; log {log}; ETA ~20:00 (estimate: not killed when over)\n" + f"summary: {OUT} · status: wf batch --status\n") + argv = self.unit_argv() + i = argv.index(str(self.claude)) + self.assertEqual(argv[i - 1], "sh") + self.assertEqual(argv[i + 1:-1], ["-p", "--model", "opus", "--permission-mode", "auto", + "--permission-prompts", "none"]) + self.assertIn("--setenv=CLAUDE_CODE_PRINT_BG_WAIT_CEILING_MS=21600000", argv) + self.assertIn(f"MemoryMax={4 * 1024 ** 3}", argv) + e = self.box.ledger().get("r-1") + self.assertEqual((e.title, e.est_min, e.project), ("wf-batch", 360, "proj")) + + def test_prompt_from_template(self): + self.batch("2", "--lanes", "fast,slow") + prompt = self.unit_argv()[-1] + self.assertIn("for up to 2 tasks (lanes: fast, slow)", prompt) + self.assertIn(f"append one line per task to {OUT}", prompt) + self.assertIn("/projects/public/workflow/docs/orchestrator.md", prompt.replace(str(HERE), "/projects/public/workflow")) + self.assertIn("foreground", prompt) + self.assertNotIn("{", prompt) + + def test_dry_run_writes_no_summary(self): + self.batch("2", "--dry-run") + self.assertFalse((self.box.env.root() / OUT).exists()) + + def test_existing_summary_not_overwritten(self): + f = self.box.env.root() / OUT + f.parent.mkdir(exist_ok=True) + f.write_text("live\n") + self.batch("2") + self.assertTrue(f.read_text().startswith("live\n# wf-batch")) + + def test_size_lane(self): + code, out, err = self.batch("2", "--lanes", "fast", "--dry-run") + self.assertEqual((code, err), (0, "")) + self.assertIn("lanes: fast", out) + + def test_default_lanes(self): + self.batch("3") + self.assertIn("for up to 3 tasks (lanes: every lane with ready tasks, see wf lanes)", self.unit_argv()[-1]) + + def test_for_sets_ceiling_and_model(self): + self.batch("1", "--for", "2h", "--model", "sonnet", "--mem", "2G") + argv = self.unit_argv() + self.assertIn("--setenv=CLAUDE_CODE_PRINT_BG_WAIT_CEILING_MS=7200000", argv) + self.assertEqual(argv[argv.index("--model") + 1], "sonnet") + self.assertIn(f"MemoryMax={2 * 1024 ** 3}", argv) + + def test_no_claude(self): + self.box.caller["PATH"] = "/nonexistent" + code, out, err = self.batch("2") + self.assertEqual((code, out, err), (1, "", "wf: claude not found on PATH\n")) + self.assertEqual(self.box.fake.ran("systemd-run"), []) + + def test_bad_count(self): + code, _, err = self.batch("0") + self.assertEqual(code, 2) + self.assertIn("N must be ≥ 1", err) + + def test_summary_header_written(self): + code, out, err = self.batch("2", "--lanes", "fast") + self.assertEqual(code, 0) + text = (self.box.env.root() / OUT).read_text() + self.assertEqual(text, "# wf-batch proj 2026-10-01 14:00 (N=2, lanes: fast)\n") + + def test_dry_run(self): + code, out, err = self.batch("2", "--dry-run") + self.assertEqual((code, err), (0, "")) + self.assertIn("CLAUDE_CODE_PRINT_BG_WAIT_CEILING_MS=21600000 wf res run --mem 4G --for 6h --title wf-batch -- ", out) + self.assertIn(f"{self.claude} -p --model opus", out) + self.assertEqual(self.box.fake.ran("systemd-run"), []) + + def test_mem_scales_with_n(self): + for args, mem in ((("4",), "6G"), (("1",), "4G"), (("9",), "8G"), (("4", "--mem", "3G"), "3G")): + code, out, err = self.batch(*args, "--dry-run") + self.assertIn(f"wf res run --mem {mem} --for", out, args) + + def test_defaults_from_history(self): + # peaks 2,3,4 → p95 4 ×1.15 = 4.6 GB; durations 20,30,40 → p90 40 ×1.5 = 60 min; ceiling stays 6h + seed_history(self.box, [hist_rec(f"r-{i}", title="wf-batch", peak=p, minutes=m) + for i, (p, m) in enumerate(((2.0, 20.0), (3.0, 30.0), (4.0, 40.0)))]) + code, out, err = self.batch("4", "--dry-run") + self.assertEqual((code, err), (0, "")) + self.assertIn("CEILING_MS=21600000 wf res run --mem 4.6G --for 60m --title wf-batch -- ", out) + out = self.batch("4", "--mem", "3G", "--for", "2h", "--dry-run")[1] + self.assertIn("CEILING_MS=7200000 wf res run --mem 3G --for 2h --title wf-batch -- ", out) + + def test_busy_exit_3(self): + self.box.available_gb = 1 + code, out, _ = self.batch("2") + self.assertEqual(code, 3) + self.assertIn("busy", out) + + def test_force_past_unused_claim(self): + self.box.wf("run", "--mem", "10G", "--for", "40m", "--title", "big", "--", "x") + code, out, _ = self.batch("2") + self.assertEqual(code, 3) + code, out, err = self.batch("2", "--force") + self.assertEqual(code, 0, (out, err)) + self.assertIn("r-2 started", out) + + +class Status(BatchBase): + def test_none(self): + code, out, err = self.batch("--status") + self.assertEqual((code, out, err), (0, "no batch in this project\n", "")) + + def test_running_and_summary(self): + self.batch("2") + md = self.box.tmp / "proj" / OUT + with md.open("a") as f: + f.write("t-a sonnet done abc123\n") + (md.parent / "wf-batch-2026-09-30-0100.md").write_text("old\n") + code, out, _ = self.batch("--status") + self.assertEqual(code, 0) + lines = out.splitlines() + self.assertTrue(lines[0].startswith('r-1 proj "wf-batch" running'), lines[0]) + self.assertIn(f"log {self.box.env.state / 'logs' / 'r-1.log'}", lines[1]) + self.assertEqual(lines[2:], [f"{OUT}:", "# wf-batch proj 2026-10-01 14:00 (N=2, lanes: every lane with ready tasks, see wf lanes)", + "t-a sonnet done abc123"]) + + def test_orch_log_progress(self): + self.batch("2") + out_dir = self.box.tmp / "proj" / "out" + (out_dir / "wf-orch.log").write_text( + "2026-10-01 13:00 fast sonnet t-old done aaa\n2026-10-01 14:05 fast sonnet t-new done bbb\n") + _, out, _ = self.batch("--status") + self.assertIn("finished since batch start", out) + self.assertIn("t-new done bbb", out) + self.assertNotIn("t-old", out) + + def test_stop_creates_file_and_status_reports(self): + code, out, err = self.batch("--stop") + self.assertEqual((code, err), (0, "")) + f = self.box.tmp / "proj" / "out" / "wf-batch.stop" + self.assertTrue(f.exists()) + self.assertIn("out/wf-batch.stop", out) + code, out, _ = self.batch("--status") + self.assertIn("stop requested: out/wf-batch.stop", out) + + def test_stale_stop_cleared_on_start(self): + f = self.box.tmp / "proj" / "out" / "wf-batch.stop" + f.parent.mkdir(exist_ok=True) + f.touch() + code, out, err = self.batch("2") + self.assertEqual(code, 0) + self.assertIn("stale out/wf-batch.stop cleared", err) + + def test_stop_consumed_on_exit(self): + f = self.box.tmp / "proj" / "out" / "wf-batch.stop" + f.parent.mkdir(exist_ok=True) + orig = wf_res.cmd_run + + def fake(env, cfg, args): + f.touch() + return 0 + wf_res.cmd_run = fake + try: + code, _, _ = self.batch("2") + finally: + wf_res.cmd_run = orig + self.assertEqual(code, 0) + self.assertFalse(f.exists()) + + def test_prompt_has_stop_check(self): + code, out, _ = self.batch("2", "--dry-run") + self.assertIn("out/wf-batch.stop", out) + self.assertIn("stopped: stop file", out) + + def test_prompt_cloud_lane(self): + code, out, _ = self.batch("2", "--dry-run") + self.assertIn("wf orch pick cloud", out) + self.assertIn("never run `wf cloud pull`", out) + self.assertIn("sidecar pulls every 10 min", out) + self.assertIn(f"-- {self.claude} -p", out) # not a cloud project: plain claude -p job + + def test_prompt_stale_branch_once(self): + code, out, _ = self.batch("2", "--dry-run") + self.assertIn("pre-existing branch", out) + self.assertIn("ONCE", out) + + def test_prompt_checks_stop_before_every_spawn(self): + code, out, _ = self.batch("2", "--dry-run") + self.assertIn("Before EVERY spawn", out) + self.assertIn(f"test -e {self.box.tmp / 'proj' / 'out' / 'wf-batch.stop'}", out) + self.assertNotIn("Each round: first", out) + code, out, _ = self.batch("--stop") + self.assertIn("before its next spawn", out) + + def test_n_required_without_status(self): + code, _, err = self.batch() + self.assertEqual(code, 2) + + +PREP_TASKS = """\ +## Awaiting your decision + +## Pending + +- **t-ready** [P1] (<1h): Ready. + Done: works. + +- **t-low** [P3] (<1h): Low, no Done. + +- **t-hi** [P1] (1h): High, no Done. + +- **t-mid** [P2] (1h): Mid, no Done. + +## Needs human + +## Deferred +""" + + +class Prep(BatchBase): + def setUp(self): + super().setUp() + root = self.box.env.root() + (root / "workflow.toml").write_text('format = 1\ntasks = "TASKS.md"\narchive = "archive.md"\n') + (root / "TASKS.md").write_text(PREP_TASKS) + (root / "archive.md").write_text("# Archive\n") + + def test_prompt_lists_targets(self): + code, out, err = self.batch("2", "--prep") + self.assertEqual((code, err), (0, "")) + prompt = self.unit_argv()[-1] + self.assertIn("Tasks: t-hi t-mid.", prompt) + self.assertIn("never implement", prompt) + self.assertIn("wf set <id> --done", prompt) + self.assertIn("wf add -s awaiting", prompt) + self.assertIn(f"to {OUT}", prompt) + for ph in ("never invert recorded original behaviour", "wf add --parent <id> slices", + "concrete files/runs", "[opus: <why>]", "[Model: <m>]"): + self.assertIn(ph, prompt) + self.assertNotIn("{", prompt) + self.assertEqual((self.box.env.root() / OUT).read_text(), + "# wf-batch proj 2026-10-01 14:00 (prep K=2: t-hi t-mid)\n") + self.assertEqual(self.box.ledger().get("r-1").title, "wf-batch") + + def test_lanes_filter(self): + self.batch("5", "--prep", "--lanes", "fast") + self.assertIn("Tasks: t-low.", self.unit_argv()[-1]) + + def test_prep_alias_all(self): + out, err = io.StringIO(), io.StringIO() + with redirect_stdout(out), redirect_stderr(err): + code = wf_res.prep_main(["all"], self.box.env) + self.assertEqual((code, err.getvalue()), (0, "")) + self.assertIn("Tasks: t-hi t-mid", self.unit_argv()[-1]) + + def test_nothing_to_prep_starts_nothing(self): + code, out, err = self.batch("3", "--prep", "--lanes", "nolane") + self.assertEqual((code, out, err), (0, "prep: no pending task without Done (lanes: nolane); nothing started\n", "")) + self.assertEqual(self.box.fake.ran("systemd-run"), []) + self.assertFalse((self.box.env.root() / OUT).exists()) + + def test_prep_mem_default_small(self): + for args, mem in ((("3", "--prep"), "1G"), (("3", "--prep", "--mem", "2G"), "2G")): + code, out, err = self.batch(*args, "--dry-run") + self.assertIn(f"wf res run --mem {mem} --for", out, args) + + def test_dry_run(self): + code, out, err = self.batch("1", "--prep", "--dry-run") + self.assertEqual((code, err), (0, "")) + self.assertIn("--title wf-batch -- ", out) + self.assertIn("t-hi", out) + self.assertEqual(self.box.fake.ran("systemd-run"), []) + + def test_no_project(self): + (self.box.env.root() / "workflow.toml").unlink() + code, out, err = self.batch("1", "--prep") + self.assertEqual(code, 1) + self.assertTrue(err.startswith("wf: "), err) + + +class Dispatch(unittest.TestCase): + def test_wf_forwards_batch(self): + r = subprocess.run([sys.executable, str(HERE / "wf.py"), "batch", "-h"], capture_output=True, text=True) + self.assertEqual(r.returncode, 0) + self.assertIn("usage: wf batch", r.stdout) + + def test_listed_in_wf_help(self): + r = subprocess.run([sys.executable, str(HERE / "wf.py"), "-h"], capture_output=True, text=True) + self.assertIn("batch", r.stdout) + + +if __name__ == "__main__": + unittest.main() + + +FAKE_WF = """import sys, pathlib +d = pathlib.Path(sys.argv[0]).parent +with open(d / "calls.log", "a") as f: + f.write(" ".join(sys.argv[1:]) + "\\n") +if sys.argv[1:3] == ["cloud", "pull"] and (d / "clear").exists(): + for rec in (d / ".wf" / "cloud").glob("*.json"): + rec.unlink() +if sys.argv[1:3] == ["cloud", "pull"] and not (d / "pulled").exists(): + (d / "pulled").touch() + print("t-a: done (session_1, $0.40 usage): merged abc1234") + print("archive session_1 failed: x; archive by hand (wf cloud archive --ended)") + print("report: commit abc1234") + print("t-b: running (session_2, 12 min)") + print("t-c: handback (session_3, $0.10 usage): cloud handback: stuck") + print("t-d: pulled by another wf cloud pull, skipped") + sys.exit(1) +""" + + +class Sidecar(unittest.TestCase): + """wf batch on a cloud = true project: the job runs the orchestrator under the pull sidecar.""" + + def setUp(self): + self._tmp = tempfile.TemporaryDirectory() + self.root = Path(self._tmp.name) + (self.root / "out").mkdir() + self.summary = self.root / OUT + self.summary.write_text("# wf-batch\n") + self.fake = self.root / "fakewf.py" + self.fake.write_text(FAKE_WF) + + def tearDown(self): + self._tmp.cleanup() + + def run_sidecar(self, child_s, every=0.2, rc=7, **kw): + child = [sys.executable, "-c", f"import time, sys; time.sleep({child_s}); sys.exit({rc})"] + out = io.StringIO() + with redirect_stdout(out): + code = wf_res.sidecar(child, self.root, self.summary, every, [sys.executable, str(self.fake)], **kw) + return code, out.getvalue() + + def records(self, *ids): + d = self.root / ".wf" / "cloud" + d.mkdir(parents=True) + for i in ids: + (d / f"{i}.json").write_text("{}") + + def tail_line(self): + return [ln[6:] for ln in self.summary.read_text().splitlines()[1:] if "tail" in ln] + + def test_tail_ends_on_empty_records(self): + self.records("t-b") + (self.root / "clear").touch() + code, out = self.run_sidecar(0.05, every=0.1) + self.assertEqual(code, 7) + self.assertEqual(len([c for c in self.calls() if c.startswith("cloud pull")]), 1) + self.assertIn("orch post t-a cloud --result done --no-pick --commit abc1234", self.calls()) + self.assertEqual(self.tail_line(), ["cloud tail ended: no cloud records left; 1 pulls; left: none (sidecar)"]) + + def test_tail_ends_on_stop_file(self): + self.records("t-b", "t-e") + threading.Timer(0.5, (self.root / wf_res.STOP_FILE).touch).start() + code, _ = self.run_sidecar(0.05, every=0.2) + pulls = [c for c in self.calls() if c.startswith("cloud pull")] + self.assertGreaterEqual(len(pulls), 1) # pulled in the tail until the stop file + self.assertEqual(self.tail_line(), [f"cloud tail ended: stop file; {len(pulls)} pulls; left: t-b t-e (sidecar)"]) + self.assertEqual(code, 7) + + def test_tail_ends_on_cap(self): + self.records("t-b") + shrunk = [] + code, out = self.run_sidecar(0.05, every=0.1, cap=0.35, shrink=lambda: shrunk.append(1) or "shrunk") + self.assertEqual((code, shrunk), (7, [1])) # reservation shrunk once, entering the tail + self.assertIn("shrunk", out) + pulls = [c for c in self.calls() if c.startswith("cloud pull")] + self.assertGreaterEqual(len(pulls), 2) + self.assertTrue(all(c.startswith("cloud pull") or c.startswith("orch post") for c in self.calls())) # no picks + self.assertEqual(self.tail_line(), + [f"cloud tail ended: {0.35 / 3600:g}h cap; {len(pulls)} pulls; left: t-b (sidecar)"]) + + def test_no_tail_without_records(self): + shrunk = [] + code, _ = self.run_sidecar(0.05, every=0.1, shrink=lambda: shrunk.append(1) or "") + self.assertEqual((code, shrunk, self.calls(), self.tail_line()), (7, [], [], [])) + + def test_shrink_reservation(self): + b = BatchBase("run") + b.setUp() + try: + b.batch("4", "--mem", "6G") + msg = wf_res.shrink_reservation(b.box.env, "r-1") + e = b.box.ledger().get("r-1") + self.assertEqual((e.mem_gb, e.cpus, e.state), (0.2, 1, "running")) + self.assertIn("r-1 reservation 6", msg) + self.assertIn("not running", wf_res.shrink_reservation(b.box.env, "r-9")) + finally: + b.tearDown() + + def calls(self): + f = self.root / "calls.log" + return f.read_text().splitlines() if f.exists() else [] + + def test_pulls_and_posts_while_child_lives(self): + code, out = self.run_sidecar(1.0) + self.assertEqual(code, 7) # the orchestrator's exit code + calls = self.calls() + pulls = [c for c in calls if c.startswith("cloud pull")] + self.assertGreaterEqual(len(pulls), 2) # repeats every interval while the child lives + self.assertEqual(pulls[0], f"cloud pull --all --project {self.root}") + self.assertEqual([c for c in calls if c.startswith("orch post")], + ["orch post t-a cloud --result done --no-pick --commit abc1234", + "orch post t-c cloud --result handback --no-pick"]) + lines = self.summary.read_text().splitlines()[1:] + self.assertEqual([ln[6:] for ln in lines], ["cloud t-a done abc1234 (sidecar pull)", + "cloud t-c handback - (sidecar pull)"]) + self.assertIn("t-a: done", out) # pull output lands in the job log + + def test_stops_with_child(self): + code, _ = self.run_sidecar(0.1, every=5, rc=0) + self.assertEqual((code, self.calls()), (0, [])) + + def test_stop_file_ends_pulls(self): + (self.root / wf_res.STOP_FILE).touch() + code, out = self.run_sidecar(0.7) + self.assertEqual((code, self.calls()), (7, [])) + self.assertIn("sidecar: stop file, no more pulls", out) + + def test_cloud_project_job_wraps_claude(self): + b = BatchBase("run") + b.setUp() + try: + root = b.box.env.root() + (root / "workflow.toml").write_text('format = 1\ntasks = "TASKS.md"\narchive = "a.md"\ncloud = true\n') + code, out, _ = b.batch("2", "--dry-run") + self.assertEqual(code, 0) + argv = shlex.split(out.split(" -- ", 1)[1]) + self.assertEqual(argv[:9], [sys.executable, str(HERE / "wf_res.py"), "batch-sidecar", "--root", str(root), + "--summary", OUT, "--every", "600"]) + self.assertEqual(argv[9:12], ["--", str(b.claude), "-p"]) + finally: + b.tearDown() + + def test_main_entry(self): + r = subprocess.run([sys.executable, str(HERE / "wf_res.py"), "batch-sidecar", "--root", str(self.root), + "--summary", OUT, "--every", "5", "--wf", f"{sys.executable} {self.fake}", "--", + sys.executable, "-c", "import sys; sys.exit(3)"], capture_output=True, text=True) + self.assertEqual((r.returncode, r.stderr), (3, "")) + + +LOG_ROWS = "".join(f"2026-10-01T1{i}:00:00 slow opus t-{i} done abc {m}m00s\n" for i, m in enumerate((20, 25, 31))) + + +class Fit(BatchBase): + def log(self, text): + (self.box.env.root() / "out").mkdir(exist_ok=True) + (self.box.env.root() / "out" / "wf-orch.log").write_text(text) + + def test_left_shrinks_k_and_sets_for(self): + self.log(LOG_ROWS) # p90 of 20,25,31 = 31 min + code, out, err = self.batch("4", "--left", "1h40m", "--dry-run") + self.assertEqual((code, err), (0, "")) + self.assertIn("fit: 3 of 4 tasks (left 1h40m, task p90 31m from 3 runs)\n", out) + self.assertIn("CEILING_MS=6000000 wf res run --mem 4.5G --for 1h40m --title wf-batch -- ", out) + self.assertIn("for up to 3 tasks", out) + + def test_left_default_30m(self): + code, out, err = self.batch("4", "--left", "45m", "--dry-run") + self.assertIn("fit: 1 of 4 tasks (left 45m, task p90 30m default (0 runs))\n", out) + self.assertIn("--for 45m", out) + + def test_left_too_short_starts_nothing(self): + self.log(LOG_ROWS) + code, out, err = self.batch("4", "--left", "30m") + self.assertEqual((code, err), (0, "")) + self.assertEqual(out, "fit: 0 of 4 tasks (left 30m, task p90 31m from 3 runs); nothing started\n") + self.assertEqual(self.box.fake.ran("systemd-run"), []) + + def test_prompt_has_deadline_check(self): + code, out, _ = self.batch("2", "--for", "2h", "--dry-run") + self.assertIn("wf batch --time-left 2026-10-01T16:00+00:00", out) + self.assertIn("stopped: deadline", out) + + def test_time_left(self): + self.log(LOG_ROWS) + code, out, err = self.batch("--time-left", "2026-10-01T15:00+00:00") + self.assertEqual((code, out, err), (0, "time left 1h, task p90 31m (3 runs): spawn\n", "")) + code, out, err = self.batch("--time-left", "2026-10-01T14:30+00:00") + self.assertEqual((code, out), (0, "time left 30m, task p90 31m (3 runs): stop\n")) + code, out, err = self.batch("--time-left", "2026-10-01T13:00+00:00") + self.assertEqual((code, out), (0, "time left 0m, task p90 31m (3 runs): stop\n")) + + def test_time_left_bad(self): + code, out, err = self.batch("--time-left", "soon") + self.assertEqual(code, 1) + self.assertIn("wf: --time-left", err) diff --git a/tests/test_check.py b/tests/test_check.py new file mode 100644 index 0000000..e4f3d3f --- /dev/null +++ b/tests/test_check.py @@ -0,0 +1,304 @@ +import os +import sys +import time +import tempfile +import unittest +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) +from wflib import check as K +from wflib import config as C + +TOML = 'format = 1\ntasks = "TASKS.md"\narchive = "tasks/archive.md"\ndocs = ["DESIGN.md", "docs/"]\n' +ANCHORS = '[anchors]\nindex = "DESIGN.md"\nindex_section = "Subsystems"\nspecs = "docs/specs"\n' +ARCHIVE = "# Archive (newest first)\n\n- 2026-09-01 **t-done** Done thing — ok\n" +DESIGN = """\ +# Design + +## Subsystems + +### terrain + +Heightmap. [spec](docs/specs/terrain.md#terrain) + +### input + +Keys. + +## Notes + +Free text. +""" +SPEC = '# Terrain spec\n\n<a id="terrain"></a>\n## Terrain\n\nText.\n' +CLEAN = """\ +# Tasks + +## Awaiting your decision + +- **a-key**: Key needed. Which one? + +## Pending + +- **t-one** [P1] (1h): One. + - After: [[t-done]] + Ref: DESIGN.md#terrain, docs/specs/terrain.md + +- **t-two** [P2] (5h) (blocked: [[a-key]]): Two. + - After: [[t-one]] + +## Needs human + +## Deferred +""" + + +class Base(unittest.TestCase): + toml = TOML + + def setUp(self): + self.tmp = tempfile.TemporaryDirectory() + self.root = Path(self.tmp.name).resolve() + (self.root / "tasks").mkdir() + (self.root / "docs" / "specs").mkdir(parents=True) + (self.root / "workflow.toml").write_text(self.toml) + (self.root / "tasks" / "archive.md").write_text(ARCHIVE) + (self.root / "DESIGN.md").write_text(DESIGN) + (self.root / "docs" / "specs" / "terrain.md").write_text(SPEC) + self.tasks(CLEAN) + + def tearDown(self): + self.tmp.cleanup() + + def tasks(self, text): + (self.root / "TASKS.md").write_text(text) + + def pending(self, items): + self.tasks("## Awaiting your decision\n\n- **a-key**: Key. Q?\n\n## Pending\n\n" + items + + "\n## Needs human\n\n## Deferred\n") + + def run_check(self): + errors, warnings = K.check(C.load(self.root)) + return [str(e) for e in errors], [str(w) for w in warnings] + + def errors(self): + return self.run_check()[0] + + +class TasksCheckTest(Base): + def test_clean(self): + self.assertEqual(self.run_check(), ([], [])) + + def test_after_prose_and_deferred(self): + self.tasks("## Awaiting your decision\n\n## Pending\n\n" + "- **t-a** [P1] (1h): A.\n After: when ready [[t-d]]\n\n" + "## Needs human\n\n## Deferred\n\n- **t-d** [P3] (1h): D.\n") + w = self.run_check()[1] + self.assertEqual(len(w), 2, w) + self.assertTrue(any("has prose" in x and ":6:" in x for x in w), w) + self.assertTrue(any("'t-d' is deferred" in x for x in w), w) + + def test_after_clean_links(self): + self.pending("- **t-a** [P1] (1h): A.\n After: [[t-done]], [[t-b]]\n\n- **t-b** [P1] (1h): B.\n") + self.assertEqual(self.run_check()[1], []) + + def test_duplicate_id(self): + self.pending("- **t-a** [P1] (1h): A.\n\n- **t-a** [P1] (1h): Again.\n") + self.assertEqual(self.errors(), ["TASKS.md:9: t-a: duplicate id (also line 7)"]) + + def test_id_reused_from_archive(self): + self.pending("- **t-done** [P1] (1h): A.\n") + self.assertEqual(self.errors(), ["TASKS.md:7: t-done: id already used in the archive (ids are never reused)"]) + + def test_bad_grammar(self): + self.pending("- **T-Foo** [P1] (1h): A.\n") + self.assertEqual(self.errors(), ["TASKS.md:7: T-Foo: bad id (want t-… or a-…, lowercase a-z 0-9 -)"]) + + def test_wrong_kind_for_section(self): + self.pending("- **a-x**: A. Q?\n") + self.assertEqual(self.errors(), ["TASKS.md:7: a-x: a- item in Pending (a- items belong in Awaiting, t- elsewhere)"]) + + def test_task_in_awaiting(self): + self.tasks("## Awaiting your decision\n\n- **t-x** [P1] (1h): A.\n\n## Pending\n\n## Needs human\n\n## Deferred\n") + self.assertEqual(self.errors(), + ["TASKS.md:3: t-x: t- item in Awaiting your decision (a- items belong in Awaiting, t- elsewhere)"]) + + def test_missing_priority_and_effort(self): + self.pending("- **t-a**: A.\n") + self.assertEqual(self.errors(), ["TASKS.md:7: t-a: no priority [P0]-[P3]", + "TASKS.md:7: t-a: no effort (want <1h, 1h, 5h, 10h, 100h)"]) + + def test_bad_priority_and_effort(self): + self.pending("- **t-a** [P7] (2h): A.\n") + self.assertEqual(self.errors(), ["TASKS.md:7: t-a: priority P7 (want P0-P3)", + "TASKS.md:7: t-a: effort '2h' (want <1h, 1h, 5h, 10h, 100h)"]) + + def test_dangling_link(self): + self.pending("- **t-a** [P1] (1h): A.\n - see [[t-none]]\n") + self.assertEqual(self.errors(), ["TASKS.md:8: [[t-none]]: no such id in TASKS or archive"]) + + def test_dangling_link_in_prose_section(self): + self.tasks(CLEAN + "\n## Notes\n\nSee [[a-gone]] and `[[t-example]]`.\n") + self.assertEqual(self.errors(), ["TASKS.md:22: [[a-gone]]: no such id in TASKS or archive"]) + + def test_blocked_on_missing_awaiting_item(self): + self.pending("- **t-a** [P1] (1h) (blocked: [[a-none]]): A.\n") + self.assertEqual(self.errors(), ["TASKS.md:7: [[a-none]]: no such id in TASKS or archive", + "TASKS.md:7: t-a: blocked on 'a-none', which is not an open Awaiting item"]) + + def test_after_cycle(self): + self.pending("- **t-a** [P1] (1h): A.\n - After: [[t-b]]\n\n- **t-b** [P1] (1h): B.\n - After: [[t-a]]\n") + self.assertIn("TASKS.md:7: t-a: After: cycle t-a → t-b → t-a", self.errors()) + + def test_item_before_its_open_dependency(self): + self.pending("- **t-a** [P1] (1h): A.\n - After: [[t-b]]\n\n- **t-b** [P1] (1h): B.\n") + self.assertEqual(self.errors(), ["TASKS.md:7: t-a: placed before 't-b', which it is After:"]) + + def test_after_bare_id_is_flagged(self): + self.pending("- **t-a** [P1] (1h): A.\n After: t-b\n\n- **t-b** [P1] (1h): B.\n") + self.assertEqual(self.errors(), ["TASKS.md:8: t-a: After: 't-b' is not a link (want [[t-b]])"]) + + def test_slices_bare_id_is_flagged(self): + self.pending("- **t-a** [P1] (1h): A.\n - Slices: [[t-b]], t-c\n\n- **t-b** [P1] (1h): B.\n") + self.assertEqual(self.errors(), ["TASKS.md:8: t-a: Slices: 't-c' is not a link (want [[t-c]])"]) + + def test_ref_path_missing(self): + self.pending("- **t-a** [P1] (1h): A.\n Ref: docs/none.md\n") + self.assertEqual(self.errors(), ["TASKS.md:8: t-a: Ref 'docs/none.md' does not exist"]) + + def test_ref_anchor_missing(self): + self.pending("- **t-a** [P1] (1h): A.\n Ref: DESIGN.md#nope\n") + self.assertEqual(self.errors(), ["TASKS.md:8: t-a: Ref 'DESIGN.md#nope': no such anchor"]) + + def test_old_numbered_items(self): + self.pending("1. **[P1] Old** (Effort: 1h) — goal.\n - Steps\n") + self.assertEqual(self.errors(), ["TASKS.md:7: old numbered item (wf migrate)"]) + + def test_malformed_header_has_line(self): + self.pending("- **t-ok** [P1] (1h): Fine.\n\n- **t-x** [P1] (1h) no colon\n") + self.assertEqual(self.errors(), + ["TASKS.md:9: t-x: bad header: want '- **id** [Pn] (effort) [(status)]: Title. Goal.'"]) + + def test_item_after_prose_is_flagged(self): + self.pending("- **t-a** [P1] (1h): A.\nstray\n- **t-b** [P1] (1h): B.\n") + self.assertEqual(self.errors(), ["TASKS.md:9: t-b: item after a flush-left prose line (line 8): indent or move the prose"]) + + def test_missing_section(self): + self.tasks("## Pending\n\n## Needs human\n\n## Deferred\n") + self.assertEqual(self.errors(), ["TASKS.md: no '## Awaiting your decision' section"]) + + def test_number_refs_warn(self): + self.pending("- **t-a** [P1] (1h): A, see #12.\n") + self.assertEqual(self.run_check(), ([], ["TASKS.md:7: '#12': number ref (tasks have ids: [[t-…]])"])) + + def test_empty_progress_note_warns(self): + self.pending("- **t-a** [P1] (1h) (in progress: ): A.\n") + self.assertEqual(self.run_check(), ([], ["TASKS.md:7: t-a: in progress without a branch or note"])) + + def test_doc_link_to_unknown_id(self): + (self.root / "docs" / "note.md").write_text("# N\n\nSee [[t-one]], [[t-done]], [[t-lost]], [[wiki-page]].\n") + self.assertEqual(self.errors(), ["docs/note.md:3: [[t-lost]]: no such id in TASKS or archive"]) + + +class StaleAwaitingTest(Base): + def commit(self, date): + import os + import subprocess + env = {**os.environ, "GIT_AUTHOR_DATE": date, "GIT_COMMITTER_DATE": date, + "GIT_AUTHOR_NAME": "t", "GIT_AUTHOR_EMAIL": "t@t", "GIT_COMMITTER_NAME": "t", "GIT_COMMITTER_EMAIL": "t@t"} + for cmd in (["init", "-q"], ["add", "-A"], ["commit", "-q", "-m", "x"]): + subprocess.run(["git", "-C", str(self.root), *cmd], check=True, env=env, capture_output=True) + + def test_old_unreferenced_awaiting_item_warns(self): + self.tasks(CLEAN.replace("- **a-key**: Key needed. Which one?", + "- **a-key**: Key needed. Which one?\n\n- **a-old**: Old question. Still open?")) + self.commit("2020-01-01T00:00:00") + errors, warnings = self.run_check() + self.assertEqual(errors, []) + self.assertEqual(len(warnings), 1) + self.assertRegex(warnings[0], r"^TASKS.md:7: a-old: waiting \d+ days, no task references it$") + + def test_fresh_item_does_not_warn(self): + self.tasks(CLEAN.replace("- **a-key**: Key needed. Which one?", + "- **a-key**: Key needed. Which one?\n\n- **a-new**: New question. Open?")) + import datetime + self.commit(datetime.datetime.now().isoformat(timespec="seconds")) + self.assertEqual(self.run_check(), ([], [])) + + +class ConfigCheckTest(Base): + def test_missing_tasks_file(self): + (self.root / "TASKS.md").unlink() + self.assertEqual(self.errors(), ["workflow.toml: tasks 'TASKS.md' does not exist"]) + + def test_missing_archive_and_docs(self): + (self.root / "tasks" / "archive.md").unlink() + (self.root / "DESIGN.md").unlink() + self.pending("- **t-a** [P1] (1h): A.\n") + self.assertEqual(self.errors(), ["workflow.toml: archive 'tasks/archive.md' does not exist", + "workflow.toml: docs 'DESIGN.md' does not exist"]) + + +class AnchorsCheckTest(Base): + toml = TOML + ANCHORS + + def test_clean(self): + self.assertEqual(self.run_check(), ([], [])) + + def test_task_anchor_not_in_index_section(self): + self.pending("- **t-a** [P1] (1h): A.\n Ref: DESIGN.md#notes\n") + self.assertEqual(self.errors(), ["TASKS.md:8: t-a: Ref 'DESIGN.md#notes' is not a heading under 'Subsystems'"]) + + def test_duplicate_index_slug(self): + (self.root / "DESIGN.md").write_text(DESIGN.replace("### input", "### Terrain")) + self.assertEqual(self.errors(), ["DESIGN.md:9: two 'Subsystems' headings give anchor 'terrain' (also line 5)"]) + + def test_index_link_to_missing_file_and_anchor(self): + (self.root / "DESIGN.md").write_text(DESIGN.replace("Keys.", "Keys. [a](docs/none.md) [b](docs/specs/terrain.md#nope) [c](https://x.y/z)")) + self.assertEqual(self.errors(), ["DESIGN.md:11: link 'docs/none.md': file does not exist", + "DESIGN.md:11: link 'docs/specs/terrain.md#nope': no such anchor"]) + + def test_explicit_spec_anchor_without_index_entry(self): + (self.root / "docs" / "specs" / "combat.md").write_text('# Combat\n\n<a id="combat"></a>\n## Combat\n') + self.assertEqual(self.errors(), ["docs/specs/combat.md:3: explicit anchor 'combat' has no heading under 'Subsystems' in DESIGN.md"]) + + +class ProblemTest(unittest.TestCase): + def test_key_ignores_line(self): + a = K.Problem("TASKS.md", 7, "t-a", "x") + b = K.Problem("TASKS.md", 9, "t-a", "x") + self.assertEqual(a.key, b.key) + self.assertEqual(str(a), "TASKS.md:7: t-a: x") + self.assertEqual(str(K.Problem("workflow.toml", None, "", "y")), "workflow.toml: y") + + +if __name__ == "__main__": + unittest.main() + + +class MergedToolWorktreeTest(unittest.TestCase): + def git(self, root, *a): + import subprocess + subprocess.run(["git", "-C", str(root), "-c", "user.name=t", "-c", "user.email=t@t", *a], + check=True, capture_output=True) + + def test_flags_only_merged_worktrees(self): + with tempfile.TemporaryDirectory() as d: + root = Path(d) / "tool" + root.mkdir() + self.git(root, "init", "-b", "master") + self.git(root, "commit", "--allow-empty", "-m", "a") + self.git(root, "worktree", "add", str(Path(d) / "fresh"), "-b", "fresh") + self.git(root, "worktree", "add", str(Path(d) / "done"), "-b", "done") + self.git(root, "worktree", "add", str(Path(d) / "open"), "-b", "open") + self.git(Path(d) / "open", "commit", "--allow-empty", "-m", "b") + self.git(root, "commit", "--allow-empty", "-m", "c") + old = time.time() - 3 * 3600 + for n in ("done", "open"): + os.utime(Path(d) / n / ".git", (old, old)) + got = K.merged_tool_worktrees(root) + names = sorted(p.message.split("branch ")[1].split(")")[0] for p in got) + self.assertEqual(names, ["done"]) # fresh (<2h) skipped + + def test_not_git(self): + with tempfile.TemporaryDirectory() as d: + self.assertEqual(K.merged_tool_worktrees(Path(d)), []) diff --git a/tests/test_claims.py b/tests/test_claims.py new file mode 100644 index 0000000..e6b5c5d --- /dev/null +++ b/tests/test_claims.py @@ -0,0 +1,182 @@ +import json +import os +import subprocess +import sys +import unittest +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) +from wflib import tasks as T +from wflib import lanes as L +from test_cli import Cli +from test_model import LANES + +FOUR = LANES.replace("## Needs human", "- **t-four** [P3] (1h): Four.\n\n## Needs human") + + +class HeldPickTest(unittest.TestCase): + def test_pick_skips_held(self): + item, skipped = L.pick(T.parse(FOUR), set(), L.DEFAULT_LANES, "1h", None, "opus", + held={"t-one": "opus session uds:/a"}) + self.assertEqual((item.id, [(i.id, why) for i, why in skipped]), + ("t-two", [("t-one", "in progress by opus session uds:/a")])) + + +class ClaimCliTest(Cli): + tasks_text = FOUR + + def env(self, name, pid): + sock = self.root / f"{name}.sock" + sock.write_text("") + return {"CLAUDE_CODE_MESSAGING_SOCKET": str(sock), "CLAUDE_PID": str(pid), "CLAUDE_CODE_SESSION_ID": name} + + def dead_pid(self): + p = subprocess.Popen(["true"]) + p.wait() + return p.pid + + def claim(self, id): + return json.loads((self.root / ".wf" / "claims" / f"{id}.json").read_text()) + + def test_progress_claims_and_other_session_skips(self): + a, b = self.env("a", os.getpid()), self.env("b", os.getppid()) + self.ok("next", "--lane", "fast", "--as", "opus", "--brief", env=a) + self.ok("status", "t-three", "progress", "x", env=a) + c = self.claim("t-three") + self.assertEqual({k: c[k] for k in ("pid", "socket", "lane", "model")}, + {"pid": os.getpid(), "socket": str(self.root / "a.sock"), "lane": "fast", "model": "opus"}) + self.assertEqual(self.ok("next", "--lane", "fast", "--as", "opus", "--brief", env=a).splitlines()[0], + "- **t-three** [P2] (<1h) (in progress: x): Three.") + code, out, err = self.wf("next", "--lane", "fast", "--as", "opus", env=b) + self.assertEqual((code, err), (1, "wf: nothing pickable for fast (opus) in Pending\n")) + self.assertIn(f"===== Skipped =====\n- t-three: in progress by opus session uds:{self.root / 'a.sock'}\n", out) + out = self.ok("next", "--lane", "slow", "--as", "opus", env=b) + self.assertIn("===== Next task =====\n- **t-one**", out) + + def test_dead_claim_ignored(self): + self.ok("status", "t-three", "progress", "x", env=self.env("a", self.dead_pid())) + self.assertEqual(self.ok("next", "--lane", "fast", "--as", "opus", "--brief", env=self.env("b", os.getpid())).splitlines()[0], + "- **t-three** [P2] (<1h) (in progress: x): Three.") + + def test_claim_without_in_progress_status_ignored(self): + self.ok("status", "t-three", "progress", "x", env=self.env("a", os.getppid())) + self.ok("status", "t-three", "clear", env=self.env("b", os.getpid())) + self.assertFalse((self.root / ".wf" / "claims" / "t-three.json").exists()) + self.assertEqual(self.ok("next", "--lane", "fast", "--as", "opus", "--brief", env=self.env("b", os.getpid())).splitlines()[0], + "- **t-three** [P2] (<1h): Three.") + + def test_done_and_blocked_release(self): + a = self.env("a", os.getpid()) + self.ok("status", "t-three", "progress", "x", env=a) + self.ok("done", "t-three", "-m", "ok", env=a) + self.assertFalse((self.root / ".wf" / "claims" / "t-three.json").exists()) + self.ok("status", "t-four", "progress", "x", env=a) + self.ok("add", "-s", "awaiting", "Q?", env=a) + self.ok("status", "t-four", "blocked", "a-q", env=a) + self.assertFalse((self.root / ".wf" / "claims" / "t-four.json").exists()) + + def test_no_env_no_claim(self): + self.ok("status", "t-three", "progress", "x", env={"CLAUDE_PID": "", "CLAUDE_CODE_MESSAGING_SOCKET": ""}) + self.assertFalse((self.root / ".wf" / "claims").exists()) + self.assertFalse((self.root / ".wf" / "sessions").exists()) + + def test_same_lane_live_session_warns_and_keeps_registry(self): + a, b = self.env("a", os.getppid()), self.env("b", os.getpid()) + self.ok("next", "--lane", "fast", "--as", "opus", "--brief", env=a) + out = self.ok("next", "--lane", "fast", "--as", "opus", "--brief", env=b) + self.assertEqual(out.splitlines()[0], + f"another live fast session holds this lane: uds:{self.root / 'a.sock'} " + "(claims keep tasks apart; tell the owner if unintended)") + reg = json.loads((self.root / ".wf" / "sessions" / "fast.json").read_text()) + self.assertEqual(reg["socket"], str(self.root / "a.sock")) + self.ok("next", "--lane", "fast", "--as", "opus", "--brief", env=a) # own re-register: no warning + self.assertNotIn("another live", self.ok("next", "--lane", "fast", "--as", "opus", "--brief", env=a)) + + def test_lanes_unregister_drops_own_record(self): + a, b = self.env("a", os.getpid()), self.env("b", os.getppid()) + self.ok("next", "--lane", "fast", "--as", "opus", "--brief", env=a) + self.ok("next", "--lane", "slow", "--as", "sonnet", "--brief", env=b) + out = self.ok("lanes", "--unregister", env=a) + self.assertFalse((self.root / ".wf" / "sessions" / "fast.json").exists()) + self.assertTrue((self.root / ".wf" / "sessions" / "slow.json").exists()) + self.assertNotIn(str(self.root / "a.sock"), out) + + +def git(cwd, *args): + subprocess.run(["git", "-C", str(cwd), *args], check=True, capture_output=True, + env={**os.environ, "GIT_AUTHOR_NAME": "t", "GIT_AUTHOR_EMAIL": "t@t", "GIT_COMMITTER_NAME": "t", + "GIT_COMMITTER_EMAIL": "t@t"}) + + +class WorktreeTest(Cli): + tasks_text = FOUR + + def setUp(self): + super().setUp() + git(self.root, "init", "-q", "-b", "master") + (self.root / ".gitignore").write_text(".worktrees/\n") + git(self.root, "add", "-A") + git(self.root, "commit", "-qm", "init") + self.wt = self.root / ".worktrees" / "fast" + git(self.root, "worktree", "add", "-q", str(self.wt), "-b", "fast/t-three") + + def test_worktree_writes_main_tree(self): + none = {"CLAUDE_CODE_MESSAGING_SOCKET": ""} + self.wf("status", "t-three", "progress", "x", project=False, cwd=self.wt, env=none) + self.assertIn("(in progress: x)", (self.root / "TASKS.md").read_text()) + self.assertNotIn("in progress", (self.wt / "TASKS.md").read_text()) + code, out, err = self.wf("done", "t-four", "-m", "ok", project=False, cwd=self.wt / "docs", env=none) + self.assertEqual(code, 0, err) + self.assertIn("**t-four**", self.archive()) + self.assertNotIn("t-four", (self.wt / "tasks" / "archive.md").read_text()) + + def test_done_in_worktree_prints_merge_steps(self): + code, out, err = self.wf("done", "t-four", "-m", "ok", project=False, cwd=self.wt, + env={"CLAUDE_CODE_MESSAGING_SOCKET": ""}) + self.assertEqual(code, 0, err) + self.assertTrue(out.endswith( + "worktree mode (branch fast/t-three), after verify: commit your code here (explicit paths), then:\n" + " wf merge (rebase, ff-merge into master, commit TASKS.md tasks/archive.md, push home; " + "conflict → git rebase master, resolve, verify, wf merge again)\n"), out) + self.assertNotIn("worktree mode", self.ok("done", "t-three", "-m", "ok")) + + def env(self, name, pid): + sock = self.root / f"{name}.sock" + sock.write_text("") + return {"CLAUDE_CODE_MESSAGING_SOCKET": str(sock), "CLAUDE_PID": str(pid), "CLAUDE_CODE_SESSION_ID": name} + + def test_next_says_worktree_mode_when_two_sessions_live(self): + son, me = self.env("s", os.getppid()), self.env("o", os.getpid()) + self.assertNotIn("Multi-session", self.ok("next", "--lane", "fast", "--as", "opus", env=me)) + self.ok("next", "--lane", "slow", "--as", "sonnet", "--brief", env=son) + out = self.ok("next", "--lane", "fast", "--as", "opus", env=me) + self.assertIn("===== Multi-session =====\n" + "2 live sessions here: work in your lane's worktree, never on master:\n" + " cd .worktrees/fast && git switch -c fast/<task> master (wf there writes this TASKS.md)\n\n", out) + self.ok("next", "--lane", "slow", "--as", "sonnet", "--brief", env=son) + out = self.ok("next", "--lane", "slow", "--as", "sonnet", env=son) + self.assertIn(" git worktree add .worktrees/slow -b slow/<task> master (wf there writes this TASKS.md)\n", out) + code, out, err = self.wf("next", "--lane", "fast", "--as", "opus", project=False, cwd=self.wt, env=me) + self.assertIn("===== Multi-session =====\n2 live sessions here: you are in worktree .worktrees/fast " + "(branch fast/t-three); wf writes the main tree's TASKS.md\n", out) + + +if __name__ == "__main__": + unittest.main() + + +class StaleTest(ClaimCliTest): + def test_stale_list_and_clear(self): + self.ok("status", "t-three", "progress", "x", env=self.env("a", self.dead_pid())) + self.ok("status", "t-four", "progress", "y", env=self.env("b", os.getpid())) + self.assertEqual(self.ok("list", "--stale").splitlines()[0].split()[0], "t-three") + self.assertNotIn("t-four", self.ok("list", "--stale")) + self.assertIn("cleared t-three", self.ok("status", "--clear-stale")) + self.assertEqual(self.ok("list", "--progress").count("prog"), 1) + self.assertIn("t-four", self.ok("list", "--progress")) + self.assertFalse((self.root / ".wf" / "claims" / "t-three.json").exists()) + + def test_no_claim_listed(self): + self.ok("status", "t-three", "progress", "x") + (self.root / ".wf" / "claims" / "t-three.json").unlink(missing_ok=True) + self.assertIn("t-three", self.ok("list", "--stale")) diff --git a/tests/test_cli.py b/tests/test_cli.py new file mode 100644 index 0000000..43135c9 --- /dev/null +++ b/tests/test_cli.py @@ -0,0 +1,752 @@ +import datetime +import os +import subprocess +import sys +import tempfile +import unittest +from pathlib import Path + +HERE = Path(__file__).resolve().parent.parent +WF = HERE / "wf.py" +sys.path.insert(0, str(HERE)) + +TOML = ('format = 1\ntasks = "TASKS.md"\narchive = "tasks/archive.md"\ndocs = ["DESIGN.md", "docs/"]\n' + 'verify = ["make test"]\ndone = ["update CATALOG"]\nledgers = ".superpowers/sdd"\n') +TASKS = """\ +# Tasks — demo + +## Awaiting your decision + +- **a-key**: Key needed. Which one? + +## Pending + +- **t-one** [P1] (1h) (in progress: master): First thing. Do it. + - Steps: a + Ref: DESIGN.md#terrain, docs/plan.md + +- **t-two** [P2] (5h) (blocked: [[a-key]]): Second thing. + +- **t-three** [P2] (<1h): Third thing. Later. + - After: [[t-one]] + +## Needs human + +- **t-play** [P1] (<1h): Play-test feel. + - [ ] keyboard works + +## Deferred + +- **t-far** [P3] (100h): Far future. +""" +ARCHIVE = "# Archive (newest first)\n\n- 2026-09-01 **t-done** Done thing — ok\n- 2026-08-01 **t-older** Older thing — fine\n" +TODAY = datetime.date.today().isoformat() + + +class Cli(unittest.TestCase): + tasks_text = TASKS + toml = TOML + + def setUp(self): + self.tmp = tempfile.TemporaryDirectory() + self.root = Path(self.tmp.name).resolve() / "demo" + (self.root / "tasks").mkdir(parents=True) + (self.root / "docs").mkdir() + (self.root / "workflow.toml").write_text(self.toml) + (self.root / "TASKS.md").write_bytes(self.tasks_text.encode()) + (self.root / "tasks" / "archive.md").write_text(ARCHIVE) + (self.root / "DESIGN.md").write_text("# Design\n\n## terrain\n\nHeightmap.\n") + (self.root / "docs" / "plan.md").write_text("# The Plan\n\n**Goal:** Ship it.\n") + self.inbox = self.root.parent / "inbox.md" + + def tearDown(self): + self.tmp.cleanup() + + def wf(self, *args, stdin="", cwd=None, project=True, env=None, hints=False): + cmd = [sys.executable, str(WF), *(["--project", str(self.root)] if project else []), *args] + full = {**os.environ, "WF_INBOX": str(self.inbox), "WF_ROOT": str(self.root.parent), + "WF_TOOL_ROOT": str(self.root.parent / "no-tool"), "CLAUDE_CONFIG_DIR": str(self.root.parent / "no-claude"), **(env or {})} + r = subprocess.run(cmd, input=stdin, capture_output=True, text=True, cwd=cwd or self.root, env=full, timeout=30) + err = r.stderr if hints else "".join(l for l in r.stderr.splitlines(True) if not l.startswith("hint:")) + return r.returncode, r.stdout, err + + def ok(self, *args, **kw): + code, out, err = self.wf(*args, **kw) + self.assertEqual((code, err), (0, ""), out + err) + return out + + def fails(self, *args, code=1, **kw): + got, out, err = self.wf(*args, **kw) + self.assertEqual(got, code, out + err) + self.assertNotIn("Traceback", err) + return err + + def tasks(self): + return (self.root / "TASKS.md").read_text() + + def archive(self): + return (self.root / "tasks" / "archive.md").read_text() + + def item(self, id): + return self.ok("show", id) + + +class ReadTest(Cli): + def test_list_default_is_pending(self): + self.assertEqual(self.ok("list"), + "t-one P1 1h prog opus slow First thing\n" + "t-two P2 5h blkd opus fast Second thing\n" + "t-three P2 <1h - opus fast Third thing\n" + "pending 3 · human 1 · awaiting 1 · deferred 1\n") + + def test_list_all(self): + self.assertEqual(self.ok("list", "-s", "all"), + "== Awaiting your decision\n" + "a-key -- - - - - Key needed\n" + "== Pending\n" + "t-one P1 1h prog opus slow First thing\n" + "t-two P2 5h blkd opus fast Second thing\n" + "t-three P2 <1h - opus fast Third thing\n" + "== Needs human\n" + "t-play P1 <1h - opus fast Play-test feel\n" + "== Deferred\n" + "t-far P3 100h - opus fast Far future\n" + "pending 3 · human 1 · awaiting 1 · deferred 1\n") + + def first_column(self, *args): + return [l.split()[0] for l in self.ok("list", *args).splitlines()[:-1]] + + def test_list_filters(self): + self.assertEqual(self.first_column("-p", "1"), ["t-one"]) + self.assertEqual(self.first_column("--ready"), ["t-one"]) + self.assertEqual(self.first_column("--blocked"), ["t-two"]) + self.assertEqual(self.first_column("--progress"), ["t-one"]) + self.assertEqual(self.first_column("--ref", "DESIGN.md"), ["t-one"]) + self.assertEqual(self.first_column("--ref", "DESIGN.md#terrain"), ["t-one"]) + self.assertEqual(self.first_column("--ref", "DESIGN.md#other"), []) + self.assertEqual(self.first_column("-n", "2"), ["t-one", "t-two"]) + self.assertEqual(self.first_column("-s", "deferred"), ["t-far"]) + + def test_list_cuts_long_titles_to_100_columns(self): + self.ok("set", "t-far", "--title", "x" * 150) + line = self.ok("list", "-s", "deferred").splitlines()[0] + self.assertEqual(len(line), 100) + self.assertTrue(line.endswith("xxx…")) + + def test_show(self): + self.assertEqual(self.ok("show", "t-three", "a-key"), + "- **t-three** [P2] (<1h): Third thing. Later.\n - After: [[t-one]]\n\n" + "- **a-key**: Key needed. Which one?\n") + + def test_show_by_unique_prefix(self): + self.assertEqual(self.ok("show", "t-thr"), "- **t-three** [P2] (<1h): Third thing. Later.\n - After: [[t-one]]\n") + + def test_show_ambiguous_prefix(self): + self.assertEqual(self.fails("show", "t-t"), "wf: 't-t' matches t-three, t-two\n") + + def test_show_unknown(self): + self.assertRegex(self.fails("show", "t-zzz"), r"^wf: unknown id 't-zzz' \(nearest: .*\)\n$") + + def test_next(self): + self.assertEqual(self.ok("next", "--as", "opus"), + "===== Awaiting your decision (mention, don't block) =====\n" + "- a-key: Key needed. Which one?\n\n" + "===== Needs human (not picked) =====\n" + "- t-play: Play-test feel\n\n" + "===== Lanes =====\n" + "fast: 0 pickable · 2 waiting · no session\n" + "slow: 1 pickable · 0 waiting · no session → orchestrator or owner: start one (wf next --lane slow)\n\n" + "===== Next task =====\n" + "- **t-one** [P1] (1h) (in progress: master): First thing. Do it.\n" + " - Steps: a\n" + " Ref: DESIGN.md#terrain, docs/plan.md\n\n" + "===== DESIGN.md#terrain (line 3) =====\n## terrain\n\nHeightmap.\n\n" + "docs/plan.md: The Plan — Ship it.\n") + + def test_next_brief(self): + self.assertEqual(self.ok("next", "--as", "opus", "--brief"), + "- **t-one** [P1] (1h) (in progress: master): First thing. Do it.\n" + " - Steps: a\n Ref: DESIGN.md#terrain, docs/plan.md\n") + + def test_next_lists_what_it_skipped(self): + self.ok("move", "t-one", "deferred") + self.ok("prio", "t-far", "3") + code, out, err = self.wf("next", "--as", "opus") + self.assertEqual(code, 1) + self.assertIn("===== Skipped =====\n- t-two: blocked: a-key\n- t-three: after: t-one\n", out) + self.assertEqual(err, "wf: nothing pickable for all lanes (opus) in Pending\n") + + def test_next_shows_ledgers_and_inbox(self): + sdd = self.root / ".superpowers" / "sdd" / "p" + sdd.mkdir(parents=True) + (self.root / "docs" / "p.md").write_text("# P\n**Goal:** g\n### Task 1: A\n### Task 2: B\n### Task 3: C\n") + (sdd / "progress.md").write_text("# SDD ledger — plan: docs/p.md\nTask 1: complete (x)\nTask 2: Ruling: y\nTask 2: complete (z)\n") + self.inbox.write_text("- 2026-09-29 demo bug: x @abc\n- 2026-09-29 demo idea: y @abc\n") + out = self.ok("next", "--as", "opus") + self.assertIn("===== In-flight plans =====\n- docs/p.md: done 1, 2; resume at Task 3 (task-start PLAN 3) " + "(1 rulings; ledger .superpowers/sdd/p/progress.md)\n\n", out) + self.assertTrue(out.endswith("\nworkflow inbox: 2 reports (triage: workflow session)\n"), out) + + def test_ctx_item(self): + self.assertEqual(self.ok("ctx", "t-one"), + "- **t-one** [P1] (1h) (in progress: master): First thing. Do it.\n" + " - Steps: a\n" + " Ref: DESIGN.md#terrain, docs/plan.md\n\n" + "Section: Pending\n" + "Needed by: t-three\n" + "Verify (run before wf finish): make test\n\n" + "===== DESIGN.md#terrain (line 3) =====\n## terrain\n\nHeightmap.\n\n" + "docs/plan.md: The Plan — Ship it.\n") + + def test_ctx_dependencies_and_links(self): + out = self.ok("ctx", "t-three") + self.assertIn("Section: Pending\nAfter: t-one (open, Pending)\n", out) + self.assertEqual(self.ok("ctx", "a-key").split("\n\n", 1)[1], "Section: Awaiting your decision\nBlocks: t-two\n") + + def test_ctx_archived_id(self): + self.assertEqual(self.ok("ctx", "t-done"), "done: - 2026-09-01 **t-done** Done thing — ok\n") + + def test_ctx_anchor(self): + self.assertEqual(self.ok("ctx", "DESIGN.md#terrain"), + "===== DESIGN.md#terrain (line 3) =====\n## terrain\n\nHeightmap.\n\n" + "Tasks: t-one\n") + + def test_search(self): + self.assertEqual(self.ok("search", "thing", "later"), + "t-three Third thing. Later.\n" + "t-one First thing. Do it.\n" + "t-two Second thing.\n" + "tasks/archive.md:3 t-done Done thing — ok\n" + "tasks/archive.md:4 t-older Older thing — fine\n") + + def test_search_docs(self): + self.assertEqual(self.ok("search", "--docs", "height"), "DESIGN.md:3 terrain — Heightmap.\n") + + def test_search_nothing(self): + self.assertEqual(self.fails("search", "zzz"), "wf: no hits\n") + + def test_log(self): + self.assertEqual(self.ok("log", "-n", "1"), "- 2026-09-01 **t-done** Done thing — ok\n") + self.assertEqual(self.ok("log", "older"), "- 2026-08-01 **t-older** Older thing — fine\n") + + def test_check_clean(self): + self.assertEqual(self.ok("check"), "OK: 0 errors · 0 warnings\n") + + def test_check_errors(self): + (self.root / "TASKS.md").write_text(TASKS.replace("[[t-one]]", "[[t-gone]]")) + code, out, err = self.wf("check") + self.assertEqual((code, err), (1, "")) + self.assertEqual(out, "ERROR: TASKS.md:16: [[t-gone]]: no such id in TASKS or archive\n1 errors · 0 warnings\n") + + def test_projects(self): + other = self.root.parent / "group" / "other" + (other / "tasks").mkdir(parents=True) + (other / "workflow.toml").write_text('format = 1\ntasks = "TASKS.md"\narchive = "tasks/archive.md"\n') + (other / "TASKS.md").write_text("## Awaiting your decision\n\n## Pending\n\n- **t-x** [P9] (1h): X.\n\n## Needs human\n\n## Deferred\n") + (other / "tasks" / "archive.md").write_text("# Archive\n") + hidden = self.root.parent / ".gate" + hidden.mkdir() + (hidden / "workflow.toml").write_text(TOML) + tool = self.root.parent / "workflow" # a wf checkout: its templates/ is no project + (tool / "wflib").mkdir(parents=True) + (tool / "templates").mkdir() + (tool / "wf.py").write_text("") + (tool / "templates" / "workflow.toml").write_text(TOML) + self.assertEqual(self.ok("projects", project=False), + "demo pending 3 human 1 awaiting 1 errors 0 next: t-one First thing\n" + "group/other pending 1 human 0 awaiting 0 errors 1 next: t-x X\n") + + def test_outside_a_project(self): + err = self.fails("list", project=False, cwd=self.root.parent) + self.assertRegex(err, r"^wf: no workflow.toml in .* or above: not a wf project \(wf init makes one\)\n$") + + def test_broken_config(self): + (self.root / "workflow.toml").write_text(TOML + "verfy = []\n") + self.assertEqual(self.fails("list"), "wf: workflow.toml: unknown key 'verfy'\n") + + def test_finds_project_from_subfolder(self): + self.assertIn("t-one", self.ok("list", project=False, cwd=self.root / "docs")) + + def test_no_command_is_usage_error(self): + self.fails(code=2) + + +class AddTest(Cli): + def test_add_places_by_priority_and_prints_header(self): + out = self.ok("add", "Cave seams. Close the slit.", "-p", "1", "-e", "<1h") + self.assertEqual(out, "- **t-cave-seams** [P1] (<1h): Cave seams. Close the slit.\n") + self.assertEqual(self.first_ids(), ["t-one", "t-cave-seams", "t-two", "t-three"]) + + def first_ids(self, section="pending"): + return [l.split()[0] for l in self.ok("list", "-s", section).splitlines()[:-1]] + + def test_add_with_options(self): + self.ok("add", "Docs pass", "-p", "2", "-e", "5h", "--sessions", "owner", "--id", "t-docs", + "--after", "t-one,t-done", "--ref", "docs/plan.md,DESIGN.md#terrain") + self.assertEqual(self.item("t-docs"), + "- **t-docs** [P2] (5h): Docs pass.\n" + " Sessions: owner\n" + " - After: [[t-one]], [[t-done]]\n" + " Ref: docs/plan.md, DESIGN.md#terrain\n") + + def test_add_cloud(self): + self.ok("add", "Code fix", "-p", "1", "-e", "1h", "--done", "tests pass", "--model", "sonnet", "--cloud", "yes", + "--id", "t-cf") + self.assertEqual(self.item("t-cf"), "- **t-cf** [P1] (1h): Code fix.\n Done: tests pass\n" + " Model: sonnet\n Cloud: yes\n") + self.ok("add", "Gui fix", "-p", "1", "-e", "1h", "--cloud", "no", "--id", "t-gf") + self.assertIn(" Cloud: no\n", self.item("t-gf")) + + def test_add_body_from_stdin(self): + self.ok("add", "With body", "-p", "3", "-e", "1h", "-b", stdin="- Steps: a\n - sub\n- Done: b\n") + self.assertEqual(self.item("t-with-body"), + "- **t-with-body** [P3] (1h): With body.\n - Steps: a\n - sub\n - Done: b\n") + + def test_add_to_other_sections(self): + self.assertEqual(self.ok("add", "Which colour. Red or blue?", "-s", "awaiting"), + "- **a-which-colour** : Which colour. Red or blue?\n".replace(" :", ":")) + self.ok("add", "Feel check", "-p", "2", "-e", "<1h", "-s", "human") + self.assertEqual(self.first_ids("human"), ["t-play", "t-feel-check"]) + + def test_add_slice(self): + self.ok("add", "Map scaffold", "-e", "1h", "--parent", "t-far") + self.ok("add", "Combat rows", "-e", "1h", "--parent", "t-far") + self.assertEqual(self.ok("show", "t-far", "t-far-1", "t-far-2"), + "- **t-far** [P3] (100h): Far future.\n - Slices: [[t-far-1]], [[t-far-2]]\n\n" + "- **t-far-1** [P3] (1h): Map scaffold.\n\n" + "- **t-far-2** [P3] (1h): Combat rows.\n - After: [[t-far-1]]\n") + self.assertEqual(self.first_ids("deferred"), ["t-far", "t-far-1", "t-far-2"]) + + def test_add_slice_skips_deferred_previous_slice(self): + self.ok("add", "Map scaffold", "-e", "1h", "--parent", "t-far") + self.ok("add", "Combat rows", "-e", "1h", "--parent", "t-far", "-s", "pending") + self.ok("add", "Loot rows", "-e", "1h", "--parent", "t-far", "-s", "pending") + self.assertEqual(self.ok("show", "t-far-2", "t-far-3"), + "- **t-far-2** [P3] (1h): Combat rows.\n\n" + "- **t-far-3** [P3] (1h): Loot rows.\n - After: [[t-far-2]]\n") + + def test_add_named_slice_chains_after_previous(self): + self.ok("add", "Map scaffold", "-e", "1h", "--parent", "t-far", "--id", "t-map") + self.ok("add", "Combat rows", "-e", "1h", "--parent", "t-far", "--id", "t-combat") + self.assertEqual(self.ok("show", "t-far", "t-combat"), + "- **t-far** [P3] (100h): Far future.\n - Slices: [[t-map]], [[t-combat]]\n\n" + "- **t-combat** [P3] (1h): Combat rows.\n - After: [[t-map]]\n") + self.assertEqual(self.fails("done", "t-far", "-m", "x"), "wf: 't-far' has open slices: t-map, t-combat\n") + self.ok("done", "t-map", "-m", "x") + self.assertIn("last slice of t-far: finish it", self.ok("done", "t-combat", "-m", "x")) + + def test_add_leading_id_in_text(self): + self.assertEqual(self.ok("add", "a-colour: Which colour? Red or blue?", "-s", "awaiting"), + "- **a-colour**: Which colour? Red or blue?\n") + self.assertEqual(self.ok("add", "t-seams: Cave seams. Close them", "-p", "1", "-e", "1h"), + "- **t-seams** [P1] (1h): Cave seams. Close them.\n") + + def test_add_help_says_id_works_with_parent(self): + self.assertIn("--parent", self.ok("add", "-h").split("--id ID", 2)[2].split("--after")[0]) + + def test_add_block(self): + self.ok("add", "-", stdin="- **t-blk** [P0] (1h): Block. Goal.\n - Steps: a\n Done: x\n") + self.assertEqual(self.item("t-blk"), "- **t-blk** [P0] (1h): Block. Goal.\n - Steps: a\n Done: x\n") + self.assertEqual(self.first_ids()[0], "t-blk") + + def test_add_block_in_old_format_is_refused(self): + self.assertEqual(self.fails("add", "-", stdin="3. **[P2] Old** (Effort: 1h) — goal.\n"), + "wf: old numbered format: write '- **id** [Pn] (effort): Title. Goal.'\n") + + def test_add_needs_priority_and_effort(self): + self.assertIn("-p", self.fails("add", "No prio", "-e", "1h", code=2)) + self.assertIn("-e", self.fails("add", "No effort", "-p", "1", code=2)) + self.assertEqual(self.tasks(), TASKS) + + def test_add_title_without_letters(self): + self.assertEqual(self.fails("add", "???", "-p", "1", "-e", "1h"), + "wf: no id can be made from title '???': give one with --id\n") + + def test_add_same_title_twice(self): + self.ok("add", "Cave seams", "-p", "1", "-e", "1h") + self.assertEqual(self.ok("add", "Cave seams", "-p", "1", "-e", "1h"), "- **t-cave-seams-2** [P1] (1h): Cave seams.\n") + + def test_add_refused_when_result_fails_check(self): + self.assertEqual(self.fails("add", "Bad ref", "-p", "1", "-e", "1h", "--ref", "docs/none.md"), + "wf: refused: nothing written, the change adds problems:\n TASKS.md:14: t-bad-ref: Ref 'docs/none.md' does not exist\n") + self.assertEqual(self.tasks(), TASKS) + + def test_dry_run(self): + out = self.ok("add", "Cave seams", "-p", "1", "-e", "1h", "--dry-run") + self.assertIn("+- **t-cave-seams** [P1] (1h): Cave seams.\n", out) + self.assertIn("--- TASKS.md\n+++ TASKS.md (new)\n", out) + self.assertEqual(self.tasks(), TASKS) + + +class DoneTest(Cli): + def test_done_builtin_checklist_empty_cfg(self): + self.toml = TOML.replace('verify = ["make test"]\ndone = ["update CATALOG"]\n', '') + self.setUp() + out = self.ok("done", "t-one", "-m", "x") + self.assertEqual(out, "done: t-one → tasks/archive.md\nfast work now pickable: t-three (no session: tell the owner)\nchecklist:\n - re-learned anything (>3 greps to find)? one anchor line → that area's code map\n") + + def test_done(self): + out = self.ok("done", "t-one", "-m", "shipped\nwith care") + self.assertEqual(out, "done: t-one → tasks/archive.md\nfast work now pickable: t-three (no session: tell the owner)\nverify:\n make test\nchecklist:\n - re-learned anything (>3 greps to find)? one anchor line → that area's code map\n - update CATALOG\n") + self.assertNotIn("t-one**", self.tasks()) + self.assertEqual(self.archive().splitlines()[2], f"- {TODAY} **t-one** First thing — shipped with care") + self.assertEqual(self.ok("check"), "OK: 0 errors · 0 warnings\n") + self.assertEqual(self.ok("next", "--as", "opus", "--brief").splitlines()[0], "- **t-three** [P2] (<1h): Third thing. Later.") + + def test_done_needs_entry(self): + self.assertEqual(self.fails("done", "t-one", code=2), "wf: done: -m \"<entry>\" is required for a single task\n") + + def test_done_several_uses_goals(self): + self.ok("done", "t-one", "t-three") + self.assertEqual(self.archive().splitlines()[2:4], + [f"- {TODAY} **t-three** Third thing — Later.", f"- {TODAY} **t-one** First thing — Do it."]) + + def test_done_refuses_parent_with_open_slices(self): + self.ok("add", "Map scaffold", "-e", "1h", "--parent", "t-far") + self.assertEqual(self.fails("done", "t-far", "-m", "x"), "wf: 't-far' has open slices: t-far-1\n") + + def test_done_awaiting_item_unblocks(self): + self.assertEqual(self.ok("done", "a-key"), "removed: a-key\nunblocked: t-two\n") + self.assertEqual(self.item("t-two"), "- **t-two** [P2] (5h): Second thing.\n") + self.assertEqual(self.archive(), ARCHIVE) + + def test_done_unknown(self): + self.assertRegex(self.fails("done", "t-on", "-m", "x"), r"^wf: unknown id 't-on' \(nearest: t-one") + self.assertEqual(self.archive(), ARCHIVE) + + +class EditTest(Cli): + def order(self): + return [l.split()[0] for l in self.ok("list").splitlines()[:-1]] + + def test_prio(self): + self.assertEqual(self.ok("prio", "t-three", "0"), "- **t-three** [P0] (<1h): Third thing. Later.\n") + + def test_prio_refused_when_it_jumps_a_dependency(self): + # t-three is After: t-one, the insert rule keeps it behind t-one + self.ok("prio", "t-three", "0") + self.assertEqual(self.order(), ["t-one", "t-three", "t-two"]) + + def test_move_section(self): + self.assertEqual(self.ok("move", "t-two", "deferred"), "- **t-two** [P2] (5h) (blocked: [[a-key]]): Second thing.\n") + self.assertEqual(self.order(), ["t-one", "t-three"]) + + def test_move_relative(self): + self.ok("move", "t-three", "--before", "t-two") + self.assertEqual(self.order(), ["t-one", "t-three", "t-two"]) + self.assertEqual(self.fails("move", "t-two", "--before", "t-one"), + "wf: moving 't-two' there breaks priority order (wf prio, or --force)\n") + self.ok("move", "t-two", "--before", "t-one", "--force") + self.assertEqual(self.order(), ["t-two", "t-one", "t-three"]) + + def test_move_needs_one_target(self): + self.fails("move", "t-two", code=2) + self.fails("move", "t-two", "deferred", "--before", "t-one", code=2) + + def test_status(self): + self.assertEqual(self.ok("status", "t-three", "progress", "feature/x"), + "- **t-three** [P2] (<1h) (in progress: feature/x): Third thing. Later.\n") + self.assertEqual(self.ok("status", "t-three", "blocked", "a-key"), + "- **t-three** [P2] (<1h) (blocked: [[a-key]]): Third thing. Later.\n") + self.assertEqual(self.ok("status", "t-three", "clear"), "- **t-three** [P2] (<1h): Third thing. Later.\n") + self.assertEqual(self.fails("status", "t-three", "blocked", "a-none"), "wf: 'a-none' is not an open Awaiting item\n") + self.fails("status", "t-three", "progress", code=2) + + def test_set(self): + self.ok("set", "t-two", "--title", "Second: renamed", "--effort", "10h", "--sessions", "owner", + "--after", "t-one", "--ref", "docs/plan.md (goal)") + self.assertEqual(self.item("t-two"), + "- **t-two** [P2] (10h) (blocked: [[a-key]]): Second: renamed.\n" + " Sessions: owner\n - After: [[t-one]]\n Ref: docs/plan.md (goal)\n") + self.ok("set", "t-two", "--after", "", "--ref", "", "--sessions", "") + self.assertEqual(self.item("t-two"), "- **t-two** [P2] (10h) (blocked: [[a-key]]): Second: renamed.\n") + + def test_set_done(self): + self.ok("set", "t-two", "--done", "two works") + self.assertEqual(self.item("t-two"), "- **t-two** [P2] (5h) (blocked: [[a-key]]): Second thing.\n Done: two works\n") + self.ok("set", "t-two", "--done", "") + self.assertEqual(self.item("t-two"), "- **t-two** [P2] (5h) (blocked: [[a-key]]): Second thing.\n") + + def test_rename(self): + self.assertEqual(self.ok("rename", "a-key", "a-which-key"), "renamed: a-key → a-which-key (1 link)\n") + self.assertEqual(self.item("t-two"), "- **t-two** [P2] (5h) (blocked: [[a-which-key]]): Second thing.\n") + self.assertEqual(self.fails("rename", "t-one", "t-done"), "wf: id 't-done' already used in the archive (ids are never reused)\n") + + def test_set_refused_on_bad_ref(self): + self.assertIn("Ref 'missing.md' does not exist", self.fails("set", "t-two", "--ref", "missing.md")) + self.assertEqual(self.tasks(), TASKS) + + def test_set_without_fields(self): + self.fails("set", "t-two", code=2) + + def test_note(self): + self.ok("note", "t-one", "cause found") + self.assertEqual(self.item("t-one"), + "- **t-one** [P1] (1h) (in progress: master): First thing. Do it.\n" + " - Steps: a\n - cause found\n Ref: DESIGN.md#terrain, docs/plan.md\n") + + def test_body(self): + self.ok("body", "t-one", stdin="- Steps: b\n- Done: c\n") + self.assertEqual(self.item("t-one"), + "- **t-one** [P1] (1h) (in progress: master): First thing. Do it.\n" + " - Steps: b\n - Done: c\n Ref: DESIGN.md#terrain, docs/plan.md\n") + + def test_body_refusal_says_refused_and_writes_nothing(self): + before = self.tasks() + for _ in range(2): + err = self.fails("body", "t-one", stdin="- After: [[t-three]]\n") + self.assertTrue(err.startswith("wf: refused: nothing written"), err) + self.assertIn("placed before 't-three'", err) + self.assertEqual(self.tasks(), before) + + def test_body_positional_text_hints_stdin(self): + self.assertIn("stdin", self.fails("body", "t-one", "some", "text")) + self.assertIn("stdin", self.ok("body", "-h")) + + def test_tick(self): + self.assertEqual(self.ok("tick", "t-play", "keyboard"), "ticked: keyboard works\n") + self.assertIn(" - [x] keyboard works\n", self.tasks()) + self.assertEqual(self.fails("tick", "t-play", "1"), "wf: box already ticked: keyboard works\n") + + def test_write_needs_exact_id(self): + self.assertRegex(self.fails("prio", "t-thr", "1"), r"^wf: unknown id 't-thr' \(nearest: t-three") + + def test_untouched_text_survives_a_write(self): + self.ok("note", "t-far", "x") + self.assertEqual(self.tasks(), TASKS.replace("Far future.\n", "Far future.\n - x\n")) + + +class OldProblemsTest(Cli): + tasks_text = TASKS + "\n## Notes\n\nSee [[t-lost]].\n" + + def test_existing_problems_do_not_block_other_edits(self): + self.ok("note", "t-far", "x") + self.assertIn(" - x\n", self.tasks()) + + +class CrlfTest(Cli): + tasks_text = TASKS.replace("\n", "\r\n") + + def test_write_keeps_crlf(self): + self.ok("add", "Cave seams", "-p", "1", "-e", "1h") + raw = (self.root / "TASKS.md").read_bytes() + self.assertIn(b"- **t-cave-seams** [P1] (1h): Cave seams.\r\n", raw) + self.assertEqual(raw.count(b"\n"), raw.count(b"\r\n")) + self.assertTrue(raw.endswith(b"Far future.\r\n")) + + def test_read_commands_work(self): + self.assertEqual(self.ok("show", "a-key"), "- **a-key**: Key needed. Which one?\n") + + +class FormatGateTest(Cli): + toml = TOML.replace("format = 1", "format = 0") + + def test_write_refused(self): + self.assertEqual(self.fails("note", "t-one", "x"), + "wf: TASKS format 0, wf needs 1: run wf migrate --write (idle project, one commit)\n") + self.assertEqual(self.tasks(), TASKS) + + def test_read_works(self): + self.assertIn("t-one", self.ok("list")) + + +class NewerFormatTest(Cli): + toml = TOML.replace("format = 1", "format = 2") + + def test_refused(self): + self.assertEqual(self.fails("list"), "wf: project format 2 is newer than this wf (1): update /projects/public/workflow\n") + + +class ReportTest(Cli): + def test_report_line(self): + self.assertEqual(self.ok("report", "prio change\nneeds two commands", "--kind", "friction", "--cmd", "wf prio x 1"), + "reported (workflow inbox); carry on\n") + self.assertRegex(self.inbox.read_text(), + rf"^- {TODAY} demo friction: prio change needs two commands \(cmd: wf prio x 1\) @[0-9a-f]{{7}}\n$") + + def test_default_kind_and_outside_project(self): + self.ok("report", "something", project=False, cwd=self.root.parent) + self.assertRegex(self.inbox.read_text(), rf"^- {TODAY} {self.root.parent.name} bug: something @") + + def test_parallel_reports_stay_whole(self): + env = {**os.environ, "WF_INBOX": str(self.inbox)} + procs = [subprocess.Popen([sys.executable, str(WF), "--project", str(self.root), "report", f"report {n} " + "x" * 3000], + env=env, stdout=subprocess.DEVNULL) for n in range(8)] + for p in procs: + self.assertEqual(p.wait(timeout=30), 0) + lines = self.inbox.read_text().splitlines() + self.assertEqual(len(lines), 8) + for line in lines: + self.assertRegex(line, rf"^- {TODAY} demo bug: report \d x{{3000}} @[0-9a-f]{{7}}$") + + def test_empty_report(self): + self.fails("report", " ", code=2) + + +class InitTest(Cli): + def test_init(self): + new = self.root.parent / "fresh" + new.mkdir() + self.assertEqual(self.ok("init", project=False, cwd=new), + "wrote workflow.toml\nwrote TASKS.md\nwrote tasks/archive.md\nwrote CLAUDE.md\n") + self.assertTrue((new / "TASKS.md").read_text().startswith("# Tasks — fresh\n")) + self.assertTrue((new / "CLAUDE.md").read_text().startswith("# CLAUDE.md — fresh\n")) + self.assertEqual(self.ok("check", project=False, cwd=new), "OK: 0 errors · 0 warnings\n") + self.assertEqual(self.ok("add", "First", "-p", "1", "-e", "1h", project=False, cwd=new), "- **t-first** [P1] (1h): First.\n") + + def test_init_twice(self): + self.assertEqual(self.fails("init"), f"wf: {self.root}/workflow.toml exists already\n") + + def test_init_keeps_existing_tasks_file(self): + new = self.root.parent / "old" + new.mkdir() + (new / "TASKS.md").write_text("mine\n") + self.assertEqual(self.ok("init", project=False, cwd=new), + "wrote workflow.toml\nkept TASKS.md (exists; old format → wf migrate)\nwrote tasks/archive.md\nwrote CLAUDE.md\n") + self.assertEqual((new / "TASKS.md").read_text(), "mine\n") + + def test_init_keeps_existing_claude_md(self): + new = self.root.parent / "oldc" + new.mkdir() + (new / "CLAUDE.md").write_text("mine\n") + out = self.ok("init", project=False, cwd=new) + self.assertIn("kept CLAUDE.md (exists)\n", out) + self.assertEqual((new / "CLAUDE.md").read_text(), "mine\n") + + def test_init_inside_a_project_subfolder_makes_a_new_project_there(self): + self.assertEqual(self.ok("init", project=False, cwd=self.root / "docs").splitlines()[0], "wrote workflow.toml") + self.assertTrue((self.root / "docs" / "workflow.toml").is_file()) + + +class ProjectsRootTest(unittest.TestCase): + def setUp(self): + import wf + self.wf = wf + self.tmp = tempfile.TemporaryDirectory() + self.top = Path(self.tmp.name) + self.tool = self.top / "public" / "workflow" + self.tool.mkdir(parents=True) + + def tearDown(self): + self.tmp.cleanup() + + def project(self, rel): + (self.top / rel).mkdir(parents=True) + (self.top / rel / "workflow.toml").write_text("") + + def test_parent_with_projects_wins(self): + self.project("public/a") + self.project("games/b") + self.assertEqual(self.wf.default_root(self.tool), self.top / "public") + + def test_parent_without_projects_falls_back_to_grandparent(self): + self.project("games/b") + self.assertEqual(self.wf.default_root(self.tool), self.top) + + def test_no_projects_anywhere_keeps_parent(self): + self.assertEqual(self.wf.default_root(self.tool), self.top / "public") + + +class WriteSafetyTest(unittest.TestCase): + def setUp(self): + import wf + self.wf = wf + self.tmp = tempfile.TemporaryDirectory() + self.path = Path(self.tmp.name) / "TASKS.md" + self.path.write_text("one\n") + + def tearDown(self): + self.tmp.cleanup() + + def test_write(self): + stamp = self.wf.stamp(self.path) + self.wf.write_if_unchanged(self.path, "two\n", stamp) + self.assertEqual(self.path.read_text(), "two\n") + self.assertEqual(sorted(p.name for p in self.path.parent.iterdir()), ["TASKS.md"]) + + def test_stops_when_the_file_changed_since_read(self): + stamp = self.wf.stamp(self.path) + self.path.write_text("someone else wrote this\n") + with self.assertRaisesRegex(self.wf.Failure, "TASKS.md changed on disk since it was read: nothing written, run again"): + self.wf.write_if_unchanged(self.path, "two\n", stamp) + self.assertEqual(self.path.read_text(), "someone else wrote this\n") + + def test_keeps_file_mode(self): + self.path.chmod(0o640) + self.wf.write_if_unchanged(self.path, "two\n", self.wf.stamp(self.path)) + self.assertEqual(self.path.stat().st_mode & 0o777, 0o640) + + +if __name__ == "__main__": + unittest.main() + + +OLD_TASKS = """\ +# Tasks — demo + +## Pending + +1. **[P1] First thing** (Effort: 1h) — do it. + - Steps: a + Reference: [DESIGN.md#terrain](DESIGN.md#terrain) + +2. **[P2] Second thing** (Effort: 5h) — later. + - After: First thing + +## Needs human + +## Awaiting your decision + +- Which key to use. +""" +MIGRATED = """\ +# Tasks — demo + +## Pending + +- **t-first-thing** [P1] (1h): First thing. do it. + - Steps: a + Ref: DESIGN.md#terrain + +- **t-second-thing** [P2] (5h): Second thing. later. + - After: [[t-first-thing]] + +## Needs human + +## Awaiting your decision + +- **a-which-key-to-use**: Which key to use. + +## Deferred +""" + + +class MigrateCliTest(Cli): + tasks_text = OLD_TASKS + toml = TOML.replace("format = 1", "format = 0") + + def test_dry_run_prints_diff_and_map_and_writes_nothing(self): + out = self.ok("migrate") + self.assertIn("+- **t-first-thing** [P1] (1h): First thing. do it.\n", out) + self.assertIn("\nids:\n t-first-thing First thing\n t-second-thing Second thing\n a-which-key-to-use Which key to use\n", out) + self.assertTrue(out.endswith("check after migrate: 0 errors\ndry run: nothing written (wf migrate --write)\n"), out) + self.assertEqual(self.tasks(), OLD_TASKS) + self.assertIn("format = 0", (self.root / "workflow.toml").read_text()) + + def test_write(self): + out = self.ok("migrate", "--write") + self.assertTrue(out.endswith("check after migrate: 0 errors\nwrote TASKS.md, workflow.toml (format = 1)\n"), out) + self.assertEqual(self.tasks(), MIGRATED) + self.assertEqual((self.root / "workflow.toml").read_text(), TOML) + self.assertEqual(self.ok("check"), "OK: 0 errors · 0 warnings\n") + self.assertEqual(self.ok("next", "--as", "opus", "--brief").splitlines()[0], "- **t-first-thing** [P1] (1h): First thing. do it.") + + def test_second_run_changes_nothing(self): + self.ok("migrate", "--write") + self.assertEqual(self.ok("migrate", "--write"), "nothing to migrate\n") + self.assertEqual(self.tasks(), MIGRATED) + + def test_ids_avoid_archived_ones(self): + (self.root / "tasks" / "archive.md").write_text("# A\n\n- 2026-09-01 **t-first-thing** First thing — old\n") + self.ok("migrate", "--write") + self.assertIn("- **t-first-thing-2** [P1] (1h): First thing. do it.\n", self.tasks()) diff --git a/tests/test_cloud.py b/tests/test_cloud.py new file mode 100644 index 0000000..e922f9c --- /dev/null +++ b/tests/test_cloud.py @@ -0,0 +1,1174 @@ +import datetime as dt +import io +import json +import os +import shutil +import sys +import tempfile +import unittest +import subprocess +import textwrap +from contextlib import redirect_stderr, redirect_stdout +from pathlib import Path + +HERE = Path(__file__).resolve().parent.parent +sys.path.insert(0, str(HERE)) +import wf_cloud # noqa: E402 +from wflib import cloud as C, usage as U # noqa: E402 + +T0 = dt.datetime(2026, 10, 6, 12, 0, tzinfo=dt.timezone.utc) + + +def led_with(n_running, **kw): + led = {**C.new(), **kw} + for i in range(n_running): + C.add(led, f"t{i}", "p", f"s{i}", "claude-sonnet-5-5", T0) + return led + + +class Pure(unittest.TestCase): + def test_default_budget(self): + self.assertEqual(C.new()["budget"], 240.0) + + def test_balance(self): + # 240 - 10 - 4*2 = 222 + self.assertEqual(C.balance(led_with(2, spent=10.0)), 222.0) + self.assertEqual(C.balance(C.new()), 240.0) + + def test_refuse_low_balance(self): + # 20 - 17 - 0 = 3 < 4 + self.assertIn("balance", C.refusal(led_with(0, budget=20.0, spent=17.0))) + # exactly 4 left: ok + self.assertIsNone(C.refusal(led_with(0, budget=20.0, spent=16.0))) + + def test_refuse_max_parallel(self): + self.assertIn("max_parallel", C.refusal(led_with(3))) + self.assertIsNone(C.refusal(led_with(2))) + + def test_balance_counts_reserve_toward_refusal(self): + # 10 - 0 - 4*2 = 2 < 4 + self.assertIn("balance", C.refusal(led_with(2, budget=10.0))) + + def test_set_balance(self): + led = led_with(0, spent=50.0) + row = C.set_balance(led, 200.0, T0) + self.assertEqual(led["spent"], 40.0) + self.assertEqual((row["usd_source"], row["usd"]), ("owner", -10.0)) + + def test_charge_from_usage(self): + led = led_with(1) + u = U.Usage(turns=1, inp=1_000_000, out=1_000_000) # sonnet-5-5: 2 + 10 = 12 ; x1.15 = 13.8 + e = C.end(led, "s0", "done", u) + self.assertAlmostEqual(e["usd"], 13.8) + self.assertEqual(e["usd_source"], "self") + self.assertAlmostEqual(led["spent"], 13.8) + self.assertEqual(C.running(led), 0) + + def test_charge_missing_usage_is_reserve(self): + led = led_with(1) + e = C.end(led, "s0", "handback", None) + self.assertEqual((e["usd"], e["usd_source"]), (4.0, "est")) + + def test_end_twice_refused(self): + led = led_with(1) + C.end(led, "s0", "done", None) + with self.assertRaises(C.CloudError): + C.end(led, "s0", "done", None) + + def test_expire_lost_after_24h(self): + led = led_with(2) + led["entries"][1]["sent"] = (T0 + dt.timedelta(hours=20)).isoformat() + lost = C.expire(led, T0 + dt.timedelta(hours=25)) + self.assertEqual([e["sid"] for e in lost], ["s0"]) + self.assertEqual(led["entries"][0]["state"], "lost") + self.assertEqual(led["spent"], 4.0) + self.assertEqual(C.running(led), 1) + + def test_roundtrip_and_bad(self): + led = led_with(1) + self.assertEqual(C.loads(C.dumps(led)), led) + with self.assertRaises(C.CloudError): + C.loads("{nope") + + +class ArchivePure(unittest.TestCase): + CREDS = '{"claudeAiOauth": {"accessToken": "tok-1", "refreshToken": "r"}, "other": 1}' + + def test_request_url_and_headers(self): + url, h = C.archive_request("session_01AbCdEf", self.CREDS, "2.1.291") + self.assertEqual(url, "https://api.anthropic.com/v1/code/sessions/session_01AbCdEf/archive") + self.assertEqual(h, {"Authorization": "Bearer tok-1", "Content-Type": "application/json", + "anthropic-version": "2023-06-01", "User-Agent": "claude-code/2.1.291"}) + + def test_trusted_device_token_sent(self): + creds = '{"claudeAiOauth": {"accessToken": "tok-1"}, "trustedDeviceToken": "dev-9"}' + _, h = C.archive_request("session_01AbCdEf", creds, "2.1.291") + self.assertEqual(h["X-Trusted-Device-Token"], "dev-9") + + def test_no_token_or_bad_sid(self): + for creds in ("", "{}", '{"claudeAiOauth": {}}', "not json"): + with self.assertRaisesRegex(C.CloudError, "no claude.ai login token"): + C.archive_request("session_01AbCdEf", creds, "1") + for sid in ("pending:x", "-", "session_01/../x"): + with self.assertRaisesRegex(C.CloudError, "not a cloud session id"): + C.archive_request(sid, self.CREDS, "1") + + def test_problem(self): + self.assertIsNone(C.archive_problem(200, "")) + self.assertIsNone(C.archive_problem(409, "already")) + self.assertEqual(C.archive_problem(401, "x"), "HTTP 401 (login expired? run claude once)") + self.assertEqual(C.archive_problem(404, "no\nsuch " + "y" * 200), "HTTP 404: no such " + "y" * 102) + + def test_unarchived(self): + led = led_with(3) + C.end(led, "s0", "done") + C.end(led, "s1", "lost") + led["entries"][1]["archived"] = True + for e in led["entries"]: + e["sid"] = "session_" + e["sid"] + C.set_balance(led, 200, T0) + self.assertEqual(C.unarchived(led), ["session_s0"]) + + +class Cli(unittest.TestCase): + def run_wf(self, st, *argv): + out = io.StringIO() + with redirect_stdout(out): + code = wf_cloud.main(list(argv), state=st, now=T0) + return code, out.getvalue() + + def test_ledger_flow(self): + with tempfile.TemporaryDirectory() as d: + st = Path(d) + code, out = self.run_wf(st, "ledger") + self.assertEqual(code, 0) + self.assertIn("budget $240.00 spent $0.00 balance $240.00 running 0/3", out) + _, out = self.run_wf(st, "ledger", "--budget", "100", "--set-balance", "70") + self.assertIn("budget $100.00 spent $30.00 balance $70.00", out) + _, out = self.run_wf(st, "ledger") # persisted + self.assertIn("spent $30.00", out) + + +# Synthetic, same shape as a real `claude --cloud` run in a pty (escape codes, CRLF, title line). +SEND_OUT = ("\x1b[?2004l\x1b]3008;start=x;type=command\x1b\\\x1b7\x1b[r\x1b8\x1b[?25h\x1b[>4;2m\x1b[c" + "Created cloud session: Some title\r\n" + "View: https://claude.ai/code/session_01AbCdEfGhIjKlMnOpQrStUv?from=cli&m=0\r\n" + "Resume with: claude --teleport session_01AbCdEfGhIjKlMnOpQrStUv\r\n") + +FAKE = r"""#!/usr/bin/env python3 +import os, sys, time, tty +mode, st = os.environ.get("FAKE_MODE", ""), os.environ["FAKE_STATE"] +def say(s): sys.stdout.write(s); sys.stdout.flush() +def line(): + return sys.stdin.readline().strip() +n = int(open(st).read()) if os.path.exists(st) else 0 +open(st, "w").write(str(n + 1)) +if "trust" in mode and n == 0: + say("\x1b[2J Do\x1b[4Gyou trust the files in this folder?\r\n 1. No, exit\r\n 2. Yes, proceed\r\n") + if line() != "\x1b[B": + say("refused\r\n"); sys.exit(1) +if sys.argv[1] == "--cloud": + if os.environ.get("FAKE_ARGS"): + open(os.environ["FAKE_ARGS"], "w").write(os.getcwd() + "\n" + sys.argv[2]) + if "nosid" in mode: + say("Error: something\r\nline2\r\n"); sys.exit(1) + say(%r); sys.exit(0) +assert sys.argv[1] == "--teleport", sys.argv +say("\x1b[3G◯\x1b[5GChecking\x1b[14Gout\x1b[18Gbranch\r\n") +if "noresume" in mode or ("flaky" in mode and n == 0): + sys.exit(0) +say("●\x1b[3GSession\x1b[11Gresumed\r\n❯ ") +while True: + cmd = line() + if cmd.startswith("/export "): + if line() == "": # 2nd Enter (autocomplete) needed + open(cmd.split(" ", 1)[1], "w").write("● WF-RESULT done\n") + say("Conversation exported\r\n") + elif cmd == "/exit": + sys.exit(0) +""" % SEND_OUT + + +class Parse(unittest.TestCase): + def test_sid_from_send_output(self): + self.assertEqual(C.parse_sid(SEND_OUT), "session_01AbCdEfGhIjKlMnOpQrStUv") + + def test_sid_from_view_only(self): + self.assertEqual(C.parse_sid("View: https://claude.ai/code/session_01Zz9Zz9Zz9Zz9?from=cli\n"), + "session_01Zz9Zz9Zz9Zz9") + + def test_sid_absent(self): + self.assertIsNone(C.parse_sid("Error: not logged in\nclaude --teleport\n")) + + def test_strip_ansi_gaps(self): + # CSI n G (column) and CSI n C (forward) are word gaps in the TUI + self.assertEqual(C.strip_ansi("\x1b[39m●\x1b[3GSession\x1b[11Gresumed\r\n"), "● Session resumed\n") + self.assertEqual(C.strip_ansi("a\x1b[2Cb"), "a b") + + def test_last_lines(self): + self.assertEqual(C.last_lines("a\r\n\r\nb\nc\n", 2), ["b", "c"]) + + def test_trust_keeps_other_keys(self): + out = json.loads(C.trust('{"x": 1, "projects": {"/a": {"k": 2}}}', "/b")) + self.assertEqual(out, {"x": 1, "projects": {"/a": {"k": 2}, "/b": {"hasTrustDialogAccepted": True}}}) + out = json.loads(C.trust('{"projects": {"/a": {"k": 2}}}', "/a")) + self.assertEqual(out["projects"]["/a"], {"k": 2, "hasTrustDialogAccepted": True}) + self.assertEqual(json.loads(C.trust("", "/c")), {"projects": {"/c": {"hasTrustDialogAccepted": True}}}) + with self.assertRaises(C.CloudError): + C.trust("[1]", "/c") + + +class PtyDriver(unittest.TestCase): + """send/export against a fake claude on a real pty.""" + + def setUp(self): + self.d = Path(tempfile.mkdtemp()) + fake = self.d / "claude" + fake.write_text(FAKE) + fake.chmod(0o755) + self.repo = self.d / "repo" + self.repo.mkdir() + self.cj = self.d / "claude.json" + self.cj.write_text('{"keep": true}') + self.env = {"WF_CLAUDE_JSON": str(self.cj), "FAKE_STATE": str(self.d / "n")} + self.old = (wf_cloud.CLAUDE, wf_cloud.SETTLE, dict(os.environ)) + wf_cloud.CLAUDE, wf_cloud.SETTLE = str(fake), 0.3 + os.environ.update(self.env) + + def tearDown(self): + wf_cloud.CLAUDE, wf_cloud.SETTLE, env = self.old + os.environ.clear() + os.environ.update(env) + shutil.rmtree(self.d) + + def test_send_parses_sid_and_pre_trusts(self): + self.assertEqual(wf_cloud.send(self.repo, "do it", timeout=10), "session_01AbCdEfGhIjKlMnOpQrStUv") + data = json.loads(self.cj.read_text()) + self.assertEqual(data["projects"][str(self.repo.resolve())], {"hasTrustDialogAccepted": True}) + self.assertTrue(data["keep"]) + + def test_send_answers_trust_dialog(self): + os.environ["FAKE_MODE"] = "trust" + self.assertEqual(wf_cloud.send(self.repo, "do it", timeout=10), "session_01AbCdEfGhIjKlMnOpQrStUv") + + def test_send_no_sid(self): + os.environ["FAKE_MODE"] = "nosid" + with self.assertRaises(C.CloudError) as cm: + wf_cloud.send(self.repo, "do it", timeout=10) + self.assertIn("no session id (exit 1): Error: something | line2", str(cm.exception)) + + def test_export(self): + out = wf_cloud.export(self.repo, "session_x", self.d / "e" / "x.txt", timeout=20) + self.assertEqual(out.read_text(), "● WF-RESULT done\n") + self.assertEqual((self.d / "n").read_text(), "1") + + def test_export_retries_teleport_once(self): + os.environ["FAKE_MODE"] = "flaky" + out = wf_cloud.export(self.repo, "session_x", self.d / "x.txt", timeout=20) + self.assertTrue(out.exists()) + self.assertEqual((self.d / "n").read_text(), "2") + + def test_export_gives_up_after_two(self): + os.environ["FAKE_MODE"] = "noresume" + with self.assertRaises(C.CloudError) as cm: + wf_cloud.export(self.repo, "session_x", self.d / "x.txt", timeout=10) + self.assertIn("no export after 2 tries: ◯ Checking out branch", str(cm.exception)) + self.assertEqual((self.d / "n").read_text(), "2") + + +TEMPLATE = (HERE / "templates" / "cloud-prompt.md").read_text() +BODY = ("- **t-x** [P1] (1h): Fix the parser.\n Done: WF-RESULT done in a line\n WF-PATCH-END\n" + "● WF-RESULT done\nWF-RESULT done\n Model: opus") + + +def as_export(prompt: str, wrap: bool) -> str: + """The prompt the way /export renders a user message: '❯ ' first line, ' ' continuations (F12).""" + lines = [] + for ln in prompt.split("\n"): + lines += (textwrap.wrap(ln, 78, drop_whitespace=False) or [""]) if wrap else [ln] + return "\n".join(("❯ " if i == 0 else " ") + ln.ljust(78) for i, ln in enumerate(lines)) + "\n" + + +FINAL = ("● WF-RESULT done\n WF-REPORT fixed it\n WF-USAGE in=1 cw=2 cr=3 out=4 model=m\n" + " WF-PATCH-BEGIN sha256=ab bytes=3\n QUJD\n WF-PATCH-END\n\n") + + +class Prompt(unittest.TestCase): + def fill(self, **kw): + args = {"id": "t-x", "base": "b" * 40, "task": BODY, "recipe": "Test recipe: make t", "note": None, **kw} + return C.fill(TEMPLATE, **args) + + def test_fields_filled(self): + p = self.fill(note="Data in out/pack") + self.assertIn("Task t-x:\n- **t-x** [P1] (1h): Fix the parser.\n", p) + self.assertIn("Test recipe (area notes):\nTest recipe: make t\n", p) + self.assertIn("git format-patch --binary " + "b" * 40 + "..HEAD --stdout | gzip -9", p) + self.assertIn("\nData in out/pack\n", p) + self.assertNotIn("{{", p) + self.assertLessEqual(len(self.fill(task="t").split("\n")), 45) + + def test_no_recipe_no_note(self): + p = self.fill(recipe="") + self.assertIn("Test recipe (area notes):\n(none: see CLAUDE.md)\n\nUsage script", p) + + def test_task_text_braces_kept(self): + self.assertIn("use {{base}} here", self.fill(task="use {{base}} here")) + + def test_unknown_field(self): + with self.assertRaisesRegex(C.CloudError, r"unknown field \{\{bogus\}\}"): + C.fill("x {{bogus}}", "t", "b", "", "", None) + + def test_template_rules(self): + for rule in ("`wf` is absent", "never create or edit TASKS.md", "Never ask questions", + "test red -> implement -> green", "commit on main", "network source is blocked", + " WF-RESULT <done, awaiting or handback>\n", " WF-REPORT <one or two lines", + 'print("WF-USAGE in=%d cw=%d cr=%d out=%d model=%s"', 'echo "WF-PATCH-BEGIN sha256=$(', + "echo WF-PATCH-END", "base64 -w 76", "~/.claude/projects", "never -A", "__pycache__"): + self.assertIn(rule, TEMPLATE) + # live run 2026-10-06: '[WF-RESULT R]' placeholders made the session drop every key word + self.assertNotIn("[WF-", TEMPLATE) + self.assertIn("key word", TEMPLATE) + + def test_parser_never_matches_prompt(self): + p = self.fill() + for wrap in (False, True): + self.assertIsNone(C.final_message(as_export(p, wrap)), wrap) + + def test_parser_takes_final_message_after_prompt(self): + exp = as_export(self.fill(), True) + "\n● Working on it.\n Ran 3 shell commands\n\n" + FINAL + self.assertEqual(C.final_message(exp), ["WF-RESULT done", "WF-REPORT fixed it", + "WF-USAGE in=1 cw=2 cr=3 out=4 model=m", + "WF-PATCH-BEGIN sha256=ab bytes=3", "QUJD", "WF-PATCH-END"]) + + def test_parser_last_result_wins_and_stops_at_next_message(self): + exp = "● WF-RESULT handback\n WF-REPORT old\n\n● WF-RESULT awaiting\n WF-REPORT q?\n\n❯ more\n" + self.assertEqual(C.final_message(exp), ["WF-RESULT awaiting", "WF-REPORT q?"]) + + def test_include_problem(self): + for bad in ("/abs", "../x", "a/../../b", "", ".git", ".git/config"): + self.assertIsNotNone(C.include_problem(bad), bad) + for ok in ("out/pack", "data.bin", "out/a/b/"): + self.assertIsNone(C.include_problem(ok), ok) + + def test_size_refusal(self): + self.assertIsNone(C.size_refusal(90_000_000)) + self.assertEqual(C.size_refusal(91_400_000), "snapshot 91 MB > 90 MB; trim cloud_include") + + +from test_claims import FOUR, git # noqa: E402 +from test_setup import GitCli # noqa: E402 + +AREAS = "# Demo\n\n## Areas\n\n### app\n- Test recipe: RECIPE-MARK python3 -m unittest\n- Paths: src/app.py\n" + + + +def stub_archive(tc, d: Path, status=200): + """No network: wf_cloud.post records (url, headers) and answers `status`; fake credentials file.""" + creds = d / "credentials.json" + creds.write_text('{"claudeAiOauth": {"accessToken": "tok-1"}}') + tc.posts, tc.answer = [], status + saved = (wf_cloud.post, wf_cloud.cli_version, os.environ.get("WF_CLAUDE_CREDENTIALS")) + + def post(url, headers, timeout=10): + tc.posts.append((url, headers)) + if isinstance(tc.answer, Exception): + raise tc.answer + return tc.answer, "body" + + def restore(): + wf_cloud.post, wf_cloud.cli_version, env = saved + os.environ.pop("WF_CLAUDE_CREDENTIALS", None) + if env is not None: + os.environ["WF_CLAUDE_CREDENTIALS"] = env + wf_cloud.post, wf_cloud.cli_version = post, lambda: "9.9.9" + os.environ["WF_CLAUDE_CREDENTIALS"] = str(creds) + tc.addCleanup(restore) + + +class CloudProject(GitCli): + """A cloud-opted git project + fake claude; wf cloud in-process.""" + tasks_text = FOUR.replace("Four.", "Four: fix src/app.py.") + toml = GitCli.toml + 'cloud = true\ncloud_include = ["out/pack"]\ncloud_note = "NOTE-MARK data in out/pack"\n' + + def setUp(self): + super().setUp() + (self.root / "src").mkdir() + (self.root / "src" / "app.py").write_text("v1\n") + (self.root / "CLAUDE.md").write_text(AREAS) + (self.root / "out").mkdir() + (self.root / "out" / "tracked.txt").write_text("tracked under out/\n") + git(self.root, "add", "-A") + git(self.root, "add", "-f", "out/tracked.txt") + git(self.root, "commit", "-qm", "app") + (self.root / "src" / "app.py").write_text("dirty\n") # working tree: not in the snapshot + (self.root / "out" / "pack").mkdir(parents=True) + (self.root / "out" / "pack" / "data.bin").write_text("pack\n") + (self.root / "out" / "other.txt").write_text("not included\n") + (self.root / ".wf").mkdir(exist_ok=True) + (self.root / ".wf" / "x").write_text("state\n") + self.d = Path(tempfile.mkdtemp()) + fake = self.d / "claude" + fake.write_text(FAKE) + fake.chmod(0o755) + self.st = self.d / "state" + self.old = (wf_cloud.CLAUDE, wf_cloud.CAP, dict(os.environ)) + wf_cloud.CLAUDE = str(fake) + os.environ.update({"WF_CLAUDE_JSON": str(self.d / "claude.json"), "FAKE_STATE": str(self.d / "n"), + "FAKE_ARGS": str(self.d / "args"), "CLAUDE_CODE_MESSAGING_SOCKET": "", + "WF_INBOX": str(self.d / "inbox.md")}) + self.snap = self.root / "out" / "cloud" / "t-four" + self.rec = self.root / ".wf" / "cloud" / "t-four.json" + stub_archive(self, self.d) + + def tearDown(self): + wf_cloud.CLAUDE, wf_cloud.CAP, env = self.old + os.environ.clear() + os.environ.update(env) + shutil.rmtree(self.d) + super().tearDown() + + def send(self, *extra): + out, err = io.StringIO(), io.StringIO() + with redirect_stdout(out), redirect_stderr(err): + code = wf_cloud.main(["send", "t-four", "--project", str(self.root), *extra], state=self.st, now=T0) + return code, out.getvalue(), err.getvalue() + + def ledger(self): + return json.loads((self.st / "cloud.json").read_text()) if (self.st / "cloud.json").exists() else None + + def sh(self, cwd, *args): + return subprocess.run(["git", "-C", str(cwd), *args], capture_output=True, text=True).stdout.strip() + +class Send(CloudProject): + """wf cloud send against the fake claude.""" + + def test_send_snapshot_prompt_record_claim(self): + master = self.sh(self.root, "rev-parse", "master") + code, out, err = self.send() + self.assertEqual((code, err), (0, ""), out) + sid = "session_01AbCdEfGhIjKlMnOpQrStUv" + files = sorted(str(p.relative_to(self.snap)) for p in self.snap.rglob("*") + if p.is_file() and ".git" not in p.relative_to(self.snap).parts) + self.assertEqual(files, [".gitignore", "CLAUDE.md", "DESIGN.md", "docs/plan.md", "out/pack/data.bin", + "src/app.py", "workflow.toml"]) + self.assertEqual((self.snap / "src" / "app.py").read_text(), "v1\n") + self.assertEqual(self.sh(self.snap, "log", "--format=%s"), f"base {master}") + self.assertEqual(self.sh(self.snap, "ls-files").split("\n"), files) # all committed (-f) + self.assertEqual(self.sh(self.snap, "status", "--porcelain"), "") + self.assertEqual(self.sh(self.snap, "branch", "--show-current"), "main") + base = self.sh(self.snap, "rev-parse", "HEAD") + cwd, prompt = (self.d / "args").read_text().split("\n", 1) + self.assertEqual(Path(cwd), self.snap.resolve()) + self.assertIn("Task t-four:\n- **t-four** [P3] (1h): Four: fix src/app.py.\n", prompt) + self.assertIn("RECIPE-MARK", prompt) + self.assertIn("NOTE-MARK data in out/pack", prompt) + self.assertIn(f"format-patch --binary {base}..HEAD", prompt) + rec = json.loads(self.rec.read_text()) + self.assertEqual({k: rec[k] for k in ("id", "sid", "master", "base", "lane", "sent", "model")}, + {"id": "t-four", "sid": sid, "master": master, "base": base, "lane": "slow", + "sent": "2026-10-06T12:00:00+00:00", "model": "claude-opus-5-5"}) + self.assertIn(f"**t-four** [P3] (1h) (in progress: cloud:{sid}): Four", self.tasks()) + led = self.ledger() + self.assertEqual([(e["id"], e["sid"], e["state"]) for e in led["entries"]], [("t-four", sid, "running")]) + self.assertIn(f"sent t-four: {sid}", out) + code, _, err = self.send() # twice: refused + self.assertEqual(code, 1) + self.assertIn(f"t-four already sent ({sid})", err) + + def test_dry_run_sends_nothing(self): + code, out, err = self.send("--dry-run") + self.assertEqual((code, err), (0, "")) + self.assertRegex(out, r"^snapshot 0\.0 MB \(master [0-9a-f]{12}, base [0-9a-f]{12}\)\n\nYou are a remote") + self.assertIn("Task t-four:", out) + self.assertFalse(self.snap.exists()) + self.assertFalse(self.snap.parent.exists()) # out/cloud gone too, out/ kept + self.assertFalse(self.rec.exists()) + self.assertIsNone(self.ledger()) + self.assertFalse((self.d / "n").exists()) # claude never ran + self.assertNotIn("in progress", self.tasks()) + + def test_dry_run_keeps_live_snapshot(self): + self.assertEqual(self.send()[0], 0) + head = self.sh(self.snap, "rev-parse", "HEAD") + code, out, err = self.send("--dry-run") + self.assertEqual((code, err), (0, "")) + self.assertEqual(self.sh(self.snap, "rev-parse", "HEAD"), head) # live session's folder intact + self.assertEqual(sorted(p.name for p in self.snap.parent.iterdir()), ["t-four"]) # dry-run folder dropped + + def test_size_refused(self): + wf_cloud.CAP = 10 + code, _, err = self.send() + self.assertEqual(code, 1) + self.assertIn("MB > 0 MB; trim cloud_include", err) + self.assertFalse(self.snap.exists()) + self.assertIsNone(self.ledger()) + self.assertFalse((self.d / "n").exists()) + + def test_ledger_refused_exit_3(self): + with wf_cloud.locked(self.st) as led: + for i in range(3): + C.add(led, f"t{i}", "p", f"s{i}", "claude-opus-5-5", T0) + code, out, err = self.send() + self.assertEqual(code, 3) + self.assertEqual(err.count("\n"), 1) + self.assertIn("cloud refused: ", err) + self.assertIn("max_parallel", err) + self.assertEqual(len(self.ledger()["entries"]), 3) + self.assertFalse(self.snap.exists()) + self.assertFalse(self.rec.exists()) + self.assertFalse((self.d / "n").exists()) + + def test_send_failure_releases_reserve(self): + os.environ["FAKE_MODE"] = "nosid" + code, _, err = self.send() + self.assertEqual(code, 1) + self.assertIn("no session id", err) + self.assertEqual(self.ledger()["entries"], []) + self.assertFalse(self.snap.exists()) + self.assertFalse(self.rec.exists()) + + def test_not_opted_in(self): + (self.root / "workflow.toml").write_text(GitCli.toml) + code, _, err = self.send() + self.assertEqual(code, 1) + self.assertIn("not opted in", err) + + def test_include_outside_refused(self): + (self.root / "workflow.toml").write_text(GitCli.toml + 'cloud = true\ncloud_include = ["../x"]\n') + code, _, err = self.send("--dry-run") + self.assertEqual(code, 1) + self.assertIn("cloud_include '../x': must be a path inside the project", err) + + +if __name__ == "__main__": + unittest.main() + + +class PullPure(unittest.TestCase): + def result(self, raw: bytes, sha=None, n=None): + import base64, gzip, hashlib + gz = gzip.compress(raw) + b64 = base64.b64encode(gz).decode() + return ["WF-RESULT done", "WF-REPORT fixed the parser,", "tests green", + "WF-USAGE in=1000 cw=2000 cr=3000 out=4000 model=claude-opus-5-5", + f"WF-PATCH-BEGIN sha256={sha or hashlib.sha256(gz).hexdigest()} bytes={n or len(gz)}", + *[b64[i:i + 76] for i in range(0, len(b64), 76)], "WF-PATCH-END"] + + def test_parse_and_decode(self): + r = C.parse_result(self.result(b"diff --git a/x b/x\n")) + self.assertEqual((r.state, r.report, r.model), ("done", "fixed the parser, tests green", "claude-opus-5-5")) + self.assertEqual((r.usage.inp, r.usage.cw, r.usage.cr, r.usage.out), (1000, 2000, 3000, 4000)) + self.assertEqual(C.decode_patch(r), b"diff --git a/x b/x\n") + + def test_header_wrapped_anywhere(self): + lines = self.result(b"abc") + head = lines[4] + lines[4:5] = [head[:30], head[30:61], head[61:]] # sha split over three lines + self.assertEqual(C.decode_patch(C.parse_result(lines)), b"abc") + + def test_mismatches(self): + with self.assertRaisesRegex(C.CloudError, "sha256 mismatch"): + C.decode_patch(C.parse_result(self.result(b"abc", sha="0" * 64))) + with self.assertRaisesRegex(C.CloudError, "bytes .* != 7 announced"): + C.decode_patch(C.parse_result(self.result(b"abc", n=7))) + with self.assertRaisesRegex(C.CloudError, "bad WF-RESULT 'maybe'"): + C.parse_result(["WF-RESULT maybe"]) + with self.assertRaisesRegex(C.CloudError, "without WF-PATCH-END"): + C.parse_result(self.result(b"abc")[:-1]) + self.assertIsNone(C.parse_result(["WF-RESULT handback", "WF-REPORT no"]).usage) + + def keyless(self, raw: bytes) -> str: + """The keys-dropped shape seen live: '● done', report, usage, header split, base64, bare WF-PATCH-END.""" + import base64, gzip, hashlib + gz = gzip.compress(raw) + b64 = base64.b64encode(gz).decode() + msg = ["done", "Added sub and a test.", "usage in=4 cw=5 cr=6 out=7 model=claude-opus-5-5", + f"sha256={hashlib.sha256(gz).hexdigest()}", f"bytes={len(gz)}", + *[b64[i:i + 76] for i in range(0, len(b64), 76)], "WF-PATCH-END"] + return "\n".join(("● " if i == 0 else " ") + ln for i, ln in enumerate(msg)) + "\n\n" + + def test_keyless_final_message(self): + prompt = as_export("do it\nend with\n [WF-PATCH-END]\nWF-PATCH-END", wrap=False) + text = prompt + "● working\n\n" + self.keyless(b"diff --git a/x b/x\n") + "● Session resumed\n" + self.assertIsNone(C.final_message(text)) + lines = C.keyless_message(text) + self.assertEqual((lines[0], lines[-1]), ("done", "WF-PATCH-END")) + r = C.keyless_result(lines) + self.assertEqual((r.state, r.model, r.usage.inp, r.usage.out), ("handback", "claude-opus-5-5", 4, 7)) + self.assertEqual(C.decode_patch(r), b"diff --git a/x b/x\n") + self.assertIsNone(C.keyless_message(prompt)) # prompt only: running + self.assertIsNone(C.keyless_message(prompt + "● still working\n")) + self.assertIsNone(C.keyless_message(text + as_export("redo with the keys", wrap=False))) # redo sent + self.assertIsNotNone(C.keyless_message(text + "❯ \n")) # empty input line + r = C.keyless_result(["done", "no patch here", "WF-PATCH-END"]) + self.assertEqual((r.has_patch, r.usage), (False, None)) + + def test_patch_files_and_problem(self): + patch = ("diff --git a/src/a.py b/src/a.py\n--- a/src/a.py\n+++ b/src/a.py\n" + "diff --git a/old.txt b/new.txt\nrename from old.txt\nrename to new.txt\n") + self.assertEqual(C.patch_files(patch), ["src/a.py", "old.txt", "new.txt"]) + never = ["TASKS.md", "tasks/archive.md", ".wf", "out", "data/pack"] + self.assertIsNone(C.patch_problem(["src/a.py", "outish.txt"], "", never)) + self.assertIn("TASKS.md", C.patch_problem(["TASKS.md"], "", never)) + self.assertIn("data/pack", C.patch_problem(["data/pack/x"], "", never)) + self.assertIn("outside the tree", C.patch_problem(["../x"], "", never)) + self.assertIn("outside the tree", C.patch_problem([".git/hooks/x"], "", never)) + self.assertIn("outside the project folder", C.patch_problem(["other/x"], "proj", never)) + self.assertIsNone(C.patch_problem(["proj/src/a.py"], "proj", never)) + self.assertIn("out", C.patch_problem(["proj/out/x"], "proj", never)) + + +class Pull(CloudProject): + """wf cloud pull --export FIXTURE on a git project with a recorded send.""" + toml = None # set in setUp: no worktree_setup (it writes log.txt), a quick_gate + + def setUp(self): + from test_cli import TOML + self.toml = (TOML + 'cloud = true\ncloud_include = ["out/pack"]\n' + 'quick_gate = ["grep -q fixed src/app.py"]\n') + super().setUp() + os.environ.update({"GIT_AUTHOR_NAME": "t", "GIT_AUTHOR_EMAIL": "t@t", "GIT_COMMITTER_NAME": "t", + "GIT_COMMITTER_EMAIL": "t@t"}) + (self.root / "src" / "app.py").write_text("v1\n") + import wf + from wflib import config + cfg = config.load(self.root) + self.master, self.base, _ = wf_cloud.snapshot(cfg, self.snap) + self.sid = "session_01PullPullPullPull" + with wf_cloud.locked(self.st) as led: + C.add(led, "t-four", str(self.root), self.sid, C.MODEL, T0) + self.rec.parent.mkdir(parents=True, exist_ok=True) + self.rec.write_text(json.dumps({"id": "t-four", "sid": self.sid, "project": str(self.root), "lane": "slow", + "master": self.master, "base": self.base, "folder": str(self.snap), + "sent": T0.isoformat(), "model": C.MODEL, "bytes": 1})) + with redirect_stdout(io.StringIO()): + self.assertEqual(wf.main(["--project", str(self.root), "status", "t-four", "progress", + f"cloud:{self.sid}"]), 0) + self.exp = self.d / "export.txt" + + def change(self, files: dict[str, str]): + """Commit files in the snapshot (the cloud's work) -> the gzip patch bytes.""" + import gzip + for rel, text in files.items(): + (self.snap / rel).parent.mkdir(parents=True, exist_ok=True) + (self.snap / rel).write_text(text) + git(self.snap, "add", "-A", "-f") + git(self.snap, "-c", "user.name=c", "-c", "user.email=c@x", "commit", "-qm", "cloud work") + p = subprocess.run(["git", "-C", str(self.snap), "format-patch", "--binary", f"{self.base}..HEAD", "--stdout"], + capture_output=True, check=True).stdout + return gzip.compress(p) + + def write_export(self, state="done", gz=b"", report="fixed app", sha=None, wrap=False, final=True): + import base64, hashlib + b64 = base64.b64encode(gz).decode() + msg = [f"WF-RESULT {state}", f"WF-REPORT {report}", "WF-USAGE in=1000 cw=0 cr=0 out=1000 model=claude-opus-5-5", + f"WF-PATCH-BEGIN sha256={sha or hashlib.sha256(gz).hexdigest()} bytes={len(gz)}", + *[b64[i:i + 76] for i in range(0, len(b64), 76)], "WF-PATCH-END"] + if wrap: # the export soft-wraps at 60 columns: the header too + msg = [part for ln in msg for part in textwrap.wrap(ln, 60, break_long_words=True)] + text = as_export("prompt with ● WF-RESULT done inside\nWF-PATCH-BEGIN", wrap=False) + "\n" + if final: + text += "\n".join(("● " if i == 0 else " ") + ln for i, ln in enumerate(msg)) + "\n\n❯ \n" + self.exp.write_text(text) + + def pull(self, *extra, now=T0 + dt.timedelta(hours=1), export=True): + out, err = io.StringIO(), io.StringIO() + argv = ["pull", "t-four", "--project", str(self.root), "--no-push", + *(["--export", str(self.exp)] if export else []), *extra] + with redirect_stdout(out), redirect_stderr(err): + code = wf_cloud.main(argv, state=self.st, now=now) + return code, out.getvalue(), err.getvalue() + + def entry(self): + return self.ledger()["entries"][0] + + def assert_ended(self, state): + self.assertEqual(self.entry()["state"], state) + self.assertFalse(self.snap.exists()) + self.assertFalse(self.rec.exists()) + + def assert_handback(self, why, kept: bool): + self.assert_ended("handback") + self.assertIn(f"Recovery: cloud attempt {self.sid} — {why}", self.tasks()) + self.assertNotIn("in progress", self.tasks()) + self.assertEqual(self.sh(self.root, "rev-parse", "master"), self.master) + self.assertEqual(self.sh(self.root, "branch", "--list", "slow/*"), "") + self.assertEqual((self.root / "out" / "cloud" / "t-four.patch").exists(), kept) + + def test_done_archives_session(self): + self.write_export(gz=self.change({"src/app.py": "fixed\n"})) + code, out, err = self.pull() + self.assertEqual((code, err), (0, ""), out) + self.assertEqual(self.posts, [(f"https://api.anthropic.com/v1/code/sessions/{self.sid}/archive", + {"Authorization": "Bearer tok-1", "Content-Type": "application/json", + "anthropic-version": "2023-06-01", "User-Agent": "claude-code/9.9.9"})]) + self.assertIs(self.entry()["archived"], True) + self.assertNotIn("archive", out.replace("archive.md", "")) + + def test_second_pull_same_id_skips(self): + """Two pulls of one id at once (batch sidecar + orchestrator): the one without the record lock skips + with rc 0, no 'branch … exists (local WIP?)'; the holder's pull still ends the task.""" + import fcntl + self.write_export(gz=self.change({"src/app.py": "fixed\n"})) + fd = os.open(self.rec, os.O_RDONLY) + try: + fcntl.flock(fd, fcntl.LOCK_EX | fcntl.LOCK_NB) # the other pull, mid-flight + code, out, err = self.pull() + finally: + os.close(fd) + self.assertEqual((code, err), (0, ""), out) + self.assertEqual(out, "t-four: pulled by another wf cloud pull, skipped\n") + self.assertTrue(self.rec.exists()) + self.assertEqual(self.entry()["state"], "running") + self.assertEqual(self.sh(self.root, "branch", "--list", "slow/*"), "") + code, out, err = self.pull() + self.assertEqual((code, err), (0, ""), out) + self.assert_ended("done") + + def test_pull_claim_record_gone(self): + with wf_cloud.pull_claim(self.d / "gone.json") as mine: + self.assertFalse(mine) + with wf_cloud.pull_claim(self.rec) as mine: + self.assertTrue(mine) + self.rec.unlink() # winner ended it while we waited: re-checked on the next claim + with wf_cloud.pull_claim(self.rec) as mine: + self.assertFalse(mine) + + def archive_fails(self, answer, why): + self.answer = answer + self.write_export(state="handback", report="cannot") + code, out, err = self.pull() + self.assertEqual((code, err), (0, ""), out) + self.assert_ended("handback") + self.assertIn(f"archive {self.sid} failed: {why}; archive by hand (wf cloud archive --ended)", out) + self.assertNotIn("archived", self.entry()) + + def test_archive_http_error_warns_pull_still_ends(self): + self.archive_fails(403, "HTTP 403: body") + + def test_archive_network_error_warns_pull_still_ends(self): + self.archive_fails(C.CloudError("timed out"), "timed out") + + def test_running_not_archived(self): + self.write_export(final=False) + code, out, err = self.pull() + self.assertEqual(code, 4, out + err) + self.assertEqual(self.posts, []) + + def test_archive_command_ended(self): + self.write_export(state="handback", report="cannot") + self.answer = 500 + self.pull() + self.answer = 200 + out = io.StringIO() + with redirect_stdout(out): + code = wf_cloud.main(["archive", "--ended"], state=self.st, now=T0) + self.assertEqual((code, out.getvalue()), (0, f"archived {self.sid}\n")) + self.assertIs(self.entry()["archived"], True) + with redirect_stdout(out := io.StringIO()): + code = wf_cloud.main(["archive", "--ended"], state=self.st, now=T0) + self.assertEqual((code, out.getvalue()), (0, "no ended cloud session left to archive\n")) + + def test_archive_command_one_sid_failure_exit_1(self): + self.answer = 401 + err = io.StringIO() + with redirect_stdout(io.StringIO()), redirect_stderr(err): + code = wf_cloud.main(["archive", "session_01Other"], state=self.st, now=T0) + self.assertEqual(code, 1) + self.assertEqual(err.getvalue(), "wf: archive session_01Other failed: HTTP 401 (login expired? run claude once)\n") + + def test_done_applies_gates_merges(self): + self.write_export(gz=self.change({"src/app.py": "fixed\n", "src/new.py": "new\n"})) + code, out, err = self.pull() + self.assertEqual((code, err), (0, ""), out) + self.assertEqual(self.sh(self.root, "show", "master:src/app.py"), "fixed") + self.assertEqual(self.sh(self.root, "show", "master:src/new.py"), "new") + self.assertIn("cloud work", self.sh(self.root, "log", "--format=%s", "master")) + self.assertNotIn("t-four", self.tasks()) + self.assertIn(f"fixed app (cloud {self.sid})", self.archive()) + self.assertEqual(self.sh(self.root, "status", "--porcelain", "TASKS.md", "tasks"), "") # bookkeeping committed + self.assert_ended("done") + e = self.entry() + self.assertEqual(e["usd_source"], "self") + self.assertAlmostEqual(e["usd"], U.cost(C.MODEL, U.Usage(inp=1000, out=1000)) * C.OVERHEAD) + self.assertFalse((self.root / "out" / "cloud" / "t-four.patch").exists()) + self.assertRegex(out, r"report: commit [0-9a-f]{7,}\n$") + self.assertIn("$ grep -q fixed src/app.py", out) + + def test_wrapped_indented_base64_rejoined(self): + self.write_export(gz=self.change({"src/app.py": "fixed\n"}), wrap=True) + code, out, err = self.pull() + self.assertEqual((code, err), (0, ""), out) + self.assertEqual(self.sh(self.root, "show", "master:src/app.py"), "fixed") + + def test_sha_mismatch_hands_back(self): + self.write_export(gz=self.change({"src/app.py": "fixed\n"}), sha="0" * 64) + code, out, err = self.pull() + self.assertEqual(code, 0, err) + self.assert_handback("patch sha256 mismatch", kept=False) + + def test_gate_red_hands_back_keeps_patch(self): + self.write_export(gz=self.change({"src/app.py": "broken\n"})) + code, out, err = self.pull() + self.assertEqual(code, 0, err) + self.assert_handback("quick_gate 'grep -q fixed src/app.py' red (exit 1)", kept=True) + wt = self.root / ".worktrees" / "slow" + self.assertEqual(self.sh(wt, "status", "--porcelain"), "") + self.assertEqual(self.sh(wt, "branch", "--show-current"), "") + + def test_patch_touching_cloud_include_refused(self): + self.write_export(gz=self.change({"src/app.py": "fixed\n", "out/pack/data.bin": "x\n"})) + code, out, err = self.pull() + self.assertEqual(code, 0, err) + self.assert_handback("patch touches out/pack/data.bin", kept=True) + + def test_patch_touching_tasks_refused(self): + self.write_export(gz=self.change({"src/app.py": "fixed\n", "TASKS.md": "x\n"})) + code, out, err = self.pull() + self.assertEqual(code, 0, err) + self.assert_handback("patch touches TASKS.md", kept=True) + + def test_awaiting_adds_question_and_blocks(self): + self.write_export(state="awaiting", report="Which format: csv or json?") + code, out, err = self.pull() + self.assertEqual(code, 0, err) + self.assert_ended("awaiting") + m = __import__("re").search(r"\*\*(a-[^*]+)\*\*: Which format: csv or json\? \[\[t-four\]\]", self.tasks()) + self.assertTrue(m, self.tasks()) + self.assertIn(f"(blocked: [[{m[1]}]])", self.tasks()) + self.assertEqual(self.sh(self.root, "rev-parse", "master"), self.master) + + def test_cloud_handback(self): + self.write_export(state="handback", report="needs the LAN host") + code, out, err = self.pull() + self.assertEqual(code, 0, err) + self.assert_handback("cloud handback: needs the LAN host", kept=False) + + def test_no_marker_running_exit_4(self): + self.write_export(final=False) + code, out, err = self.pull() + self.assertEqual((code, err), (4, "")) + self.assertIn(f"t-four: running ({self.sid}, sent 1h00m ago)", out) + self.assertTrue(self.rec.exists() and self.snap.exists()) + self.assertEqual(self.entry()["state"], "running") + self.assertIn(f"in progress: cloud:{self.sid}", self.tasks()) + + def test_keyless_final_message_hands_back_keeps_patch(self): + import gzip + self.write_export(final=False) + msg = PullPure().keyless(gzip.decompress(self.change({"src/app.py": "fixed\n"}))) + self.exp.write_text(self.exp.read_text() + msg + "● Session resumed\n") + code, out, err = self.pull() + self.assertEqual(code, 0, err) + self.assert_handback("bad result: no WF-RESULT key; patch out/cloud/t-four.patch", kept=True) + self.assertIn(b"+fixed", (self.root / "out" / "cloud" / "t-four.patch").read_bytes()) + self.assertEqual(self.entry()["usd_source"], "self") + + def test_teleport_export_used_and_removed(self): + self.write_export(final=False) + seen = [] + + def fake_export(folder, sid, out, timeout=300): + seen.append((Path(folder), sid)) + Path(out).parent.mkdir(parents=True, exist_ok=True) + Path(out).write_text(self.exp.read_text()) + return Path(out) + + old, wf_cloud.export = wf_cloud.export, fake_export + try: + code, out, err = self.pull(export=False) + finally: + wf_cloud.export = old + self.assertEqual((code, seen), (4, [(self.snap, self.sid)])) + self.assertFalse((self.root / "out" / "cloud" / "t-four.export.txt").exists()) + + def test_over_24h_lost(self): + self.write_export(final=False) + code, out, err = self.pull(now=T0 + dt.timedelta(hours=25)) + self.assertEqual(code, 0, err) + self.assert_ended("lost") + self.assertEqual((self.entry()["usd"], self.entry()["usd_source"]), (4.0, "est")) + self.assertIn(f"Recovery: cloud attempt {self.sid} — lost: no WF-RESULT after 25h00m", self.tasks()) + self.assertNotIn("in progress", self.tasks()) + + def test_all_and_not_sent(self): + os.environ["FAKE_MODE"] = "noresume" # teleport never resumes: export fails + out, err = io.StringIO(), io.StringIO() + with redirect_stdout(out), redirect_stderr(err): + code = wf_cloud.main(["pull", "--all", "--project", str(self.root)], state=self.st, + now=T0 + dt.timedelta(hours=25)) + self.assertEqual(code, 0, err.getvalue()) # no export, > 24h: lost + self.assert_ended("lost") + code, _, err = self.pull() + self.assertEqual(code, 1) + self.assertIn("t-four is not out in the cloud", err) + + +# ------------------------------------------------------------------ fit (spec §4.5) + +FIT_DOC = """\ +# Tasks — demo + +## Awaiting your decision + +## Pending + +- **t-ok** [P1] (1h): Plain code. Goal. + Done: tests pass +- **t-sonnet** [P1] (1h): Plain code. Goal. + Done: tests pass + Model: sonnet +- **t-slice** [P1] (5h): Big. Goal. + Done: tests pass +- **t-owner** [P1] (1h): Plain. Goal. + Done: tests pass + Sessions: owner +- **t-solo** [P1] (1h): Plain. Goal. + Done: tests pass + Sessions: solo +- **t-nodone** [P1] (1h): Plain. Goal. +- **t-no** [P1] (1h): Plain. Goal. + Done: tests pass + Cloud: no +- **t-wf** [P1] (1h): Plain. Goal. + Steps: run wf set x --model opus then check + Done: tests pass +- **t-res** [P1] (1h): Plain. Goal. + Steps: wf res run the build + Done: tests pass +- **t-gui** [P1] (1h): Plain. Goal. + Steps: click through the GUI + Done: tests pass +- **t-lan** [P1] (1h): Plain. Goal. + Steps: copy to 10.0.0.20 + Done: tests pass +- **t-srv** [P1] (1h): Plain. Goal. + Steps: hit the live server + Done: tests pass +- **t-yes-wf** [P1] (1h): Plain. Goal. + Steps: run wf set x --model opus then check + Done: tests pass + Cloud: yes +- **t-yes-sonnet** [P1] (1h): Plain. Goal. + Done: tests pass + Model: sonnet + Cloud: yes +- **t-yes-owner** [P1] (1h): Plain. Goal. + Done: tests pass + Sessions: owner + Cloud: yes +- **t-yes-haiku** [P1] (<1h): Plain. Goal. + Done: tests pass + Model: haiku + Cloud: yes +- **t-no-opus** [P1] (1h): Plain. Goal. + Done: tests pass + Model: opus + Cloud: no +- **t-yes-slice** [P1] (5h): Big. Goal. + Done: tests pass + Cloud: yes + +## Needs human + +## Deferred +""" + + +class FitTest(unittest.TestCase): + def setUp(self): + from wflib import tasks as TK + self.TK = TK + self.doc = TK.parse(FIT_DOC) + self.cfg = type("Cfg", (), {"cloud": True, "slice_above": "1h"})() + + def fit(self, id): + return C.fit(self.doc.item(id), self.cfg) + + def test_fits(self): + self.assertIsNone(self.fit("t-ok")) + + def test_exclusions(self): + for id, why in [("t-sonnet", "Model sonnet"), ("t-slice", "slice"), ("t-owner", "not runner-ready"), + ("t-solo", "Sessions: solo"), ("t-nodone", "not runner-ready"), ("t-no", "Cloud: no"), + ("t-wf", "`wf` commands"), ("t-res", "wf res"), ("t-gui", "GUI"), ("t-lan", "LAN"), + ("t-srv", "live server")]: + with self.subTest(id): + self.assertIn(why, self.fit(id) or "") + + def test_not_opted_in(self): + self.cfg.cloud = False + self.assertIn("not opted in", self.fit("t-ok")) + + def test_cloud_yes_skips_regex_only_opus_only(self): + self.assertIsNone(self.fit("t-yes-wf")) + self.assertEqual(self.fit("t-yes-sonnet"), "Model sonnet stays local") + self.assertEqual(self.fit("t-yes-haiku"), "Model haiku stays local") + self.assertEqual(self.fit("t-no-opus"), "Cloud: no") + self.assertEqual(self.fit("t-sonnet"), "Model sonnet stays local") # unset: old rule + self.assertIn("not runner-ready", self.fit("t-yes-owner")) + self.assertIn("slice", self.fit("t-yes-slice")) + + def test_cloud_property_and_set(self): + self.assertEqual(self.doc.item("t-no").cloud, "no") + self.assertIsNone(self.doc.item("t-ok").cloud) + self.TK.set_fields(self.doc, "t-ok", cloud="yes") + self.assertEqual(self.doc.item("t-ok").body, [" Done: tests pass", " Cloud: yes"]) + self.TK.set_fields(self.doc, "t-ok", cloud="no", model="sonnet") + self.assertEqual(self.doc.item("t-ok").body, [" Done: tests pass", " Model: sonnet", " Cloud: no"]) + self.TK.set_fields(self.doc, "t-ok", cloud="") + self.assertEqual(self.doc.item("t-ok").body, [" Done: tests pass", " Model: sonnet"]) + with self.assertRaises(self.TK.TaskError): + self.TK.set_fields(self.doc, "t-ok", cloud="maybe") + + +class PickKeyTest(unittest.TestCase): + def test_order(self): + from wflib import tasks as TK + doc = TK.parse("""# Tasks + +## Pending + +- **t-p1-small** [P1] (<1h): A. G. + Done: x +- **t-p1-big** [P1] (1h): A. G. + Done: x +- **t-p2-yes** [P2] (<1h): A. G. + Done: x + Cloud: yes +- **t-p0** [P0] (<1h): A. G. + Done: x + +## Needs human + +## Deferred +""") + items = [doc.item(i) for i in ("t-p1-small", "t-p1-big", "t-p2-yes", "t-p0")] + got = [i.id for i in sorted(items, key=C.pick_key)] + # yes first; then prio; then 1h before <1h + self.assertEqual(got, ["t-p2-yes", "t-p0", "t-p1-big", "t-p1-small"]) + + +class CloudLineCheck(unittest.TestCase): + def test_check(self): + sys.path.insert(0, str(HERE / "tests")) + from test_check import Base + + class B(Base): + def runTest(self): + self.pending("- **t-a** [P1] (1h): A. G.\n Done: x\n Cloud: yes\n") + self.assertEqual(self.run_check(), ([], [])) + self.pending("- **t-a** [P1] (1h): A. G.\n Done: x\n Cloud: maybe\n") + self.assertTrue(any("Cloud 'maybe'" in e for e in self.errors())) + self.pending("- **t-a** [P1] (1h): A. G.\n Done: x\n Cloud: yes\n Cloud: no\n") + self.assertTrue(any("two Cloud lines" in e for e in self.errors())) + r = unittest.TextTestRunner(stream=io.StringIO()).run(B()) + self.assertTrue(r.wasSuccessful(), r.failures + r.errors) + + +ORCH_TASKS = """\ +# Tasks — demo + +## Awaiting your decision + +## Pending + +- **t-fast** [P1] (<1h): Fast sonnet task. + - Done: fast works + - Model: sonnet + +- **t-gui** [P1] (<1h): Check the GUI screenshot. + - Done: gui works + +- **t-hai** [P1] (<1h): Haiku forced to the cloud. + - Done: hai works + - Model: haiku + - Cloud: yes + +- **t-slow** [P2] (1h): Slow opus task. + - Done: slow works + +- **t-yes** [P3] (<1h): Opus task marked for the cloud. + - Done: yes works + - Cloud: yes + +## Needs human + +## Deferred +""" + + +class OrchCloud(CloudProject): + """wf orch pick cloud (subprocess, fake claude via WF_CLAUDE, ledger via XDG_STATE_HOME).""" + tasks_text = ORCH_TASKS + + def setUp(self): + super().setUp() + from test_merge import IDENT + self.env = {**IDENT, "WF_CLAUDE": wf_cloud.CLAUDE, "XDG_STATE_HOME": str(self.d / "xdg")} + + def orch(self, *args): + out = self.ok("orch", *args, env=self.env) + return out + + def put_ledger(self, **kw): + n = kw.pop("running", 0) + led = {**C.new(), **kw} + for i in range(n): + C.add(led, f"t-x{i}", "/x", f"session_x{i}", "opus", T0) + (self.d / "xdg" / "wf").mkdir(parents=True, exist_ok=True) + (self.d / "xdg" / "wf" / "cloud.json").write_text(C.dumps(led)) + + def status(self, id): + return next(l for l in (self.root / "TASKS.md").read_text().splitlines() if f"**{id}**" in l) + + def test_pick_cloud_yes_first_then_opus_then_none(self): + out = self.orch("pick", "cloud") + self.assertIn("pick: t-yes (lane cloud, model opus", out) # Cloud: yes beats P2 1h; haiku never + out = self.orch("pick", "cloud") + self.assertIn("pick: t-slow (lane cloud, model opus", out) + self.assertIn("in progress: cloud:session_01AbCdEfGhIjKlMnOpQrStUv", self.status("t-slow")) + rec = json.loads((self.root / ".wf" / "orch" / "t-slow.json").read_text()) + self.assertEqual((rec["lane"], rec["task_lane"], rec["branch"]), ("cloud", "slow", "slow/t-slow")) + self.assertTrue((self.root / ".wf" / "cloud" / "t-slow.json").exists()) + out = self.orch("pick", "cloud") + self.assertEqual(out.strip().splitlines()[0][:26], "stop lane cloud: none fit ") + led = json.loads((self.d / "xdg" / "wf" / "cloud.json").read_text()) + self.assertEqual([e["id"] for e in led["entries"]], ["t-yes", "t-slow"]) + + def test_local_lane_skips_cloud_claim(self): + self.orch("pick", "cloud", "--id", "t-slow") + out = self.orch("pick", "slow") + self.assertNotIn("t-slow", out.splitlines()[0]) + self.assertIn("pick: t-fast", out) + + def test_stop_ledger(self): + self.put_ledger(budget=3.0) + out = self.orch("pick", "cloud") + self.assertIn("stop lane cloud: ledger (balance $3.00 < reserve $4.00)", out) + self.assertNotIn("in progress", self.status("t-slow")) + + def test_stop_max_parallel(self): + self.put_ledger(max_parallel=2, running=2) + out = self.orch("pick", "cloud") + self.assertIn("stop lane cloud: max parallel (2 running >= max_parallel 2)", out) + self.assertFalse((self.root / ".wf" / "cloud" / "t-slow.json").exists()) + + def test_id_unfit_refused(self): + out = self.orch("pick", "cloud", "--id", "t-gui") + self.assertIn("stop lane cloud: none fit (t-gui: needs a GUI)", out) + + def test_not_opted_in(self): + (self.root / "workflow.toml").write_text(GitCli.toml) + out = self.orch("pick", "cloud") + self.assertIn("stop lane cloud: none fit (project not opted in", out) + + def test_post_handback_commits_and_stops(self): + self.orch("pick", "cloud", "--id", "t-slow") + self.ok("note", "t-slow", "Recovery: cloud attempt x — red") # what wf cloud pull leaves behind + self.ok("status", "t-slow", "clear") + out = self.orch("post", "t-slow", "cloud", "--result", "handback") + self.assertIn("post: t-slow handback", out) + self.assertIn("committed leftover", out) + self.assertIn("stop lane cloud: handback", out) + self.assertIn(" cloud opus t-slow handback ", (self.root / "out" / "wf-orch.log").read_text()) diff --git a/tests/test_config.py b/tests/test_config.py new file mode 100644 index 0000000..e51d2bc --- /dev/null +++ b/tests/test_config.py @@ -0,0 +1,134 @@ +import sys +import tempfile +import unittest +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) +from wflib import config, lanes as L +from wflib import config as C + + +class ConfigTest(unittest.TestCase): + def setUp(self): + self.tmp = tempfile.TemporaryDirectory() + self.root = Path(self.tmp.name).resolve() + + def tearDown(self): + self.tmp.cleanup() + + def write(self, text): + (self.root / "workflow.toml").write_text(text) + + def test_find_root_from_nested_dir(self): + self.write('format = 1\ntasks = "TASKS.md"\narchive = "a.md"\n') + deep = self.root / "a" / "b" + deep.mkdir(parents=True) + self.assertEqual(C.find_root(deep), self.root) + + def test_find_root_none(self): + with self.assertRaisesRegex(C.ConfigError, "no workflow.toml in .* or above"): + C.find_root(self.root) + + def test_minimal(self): + self.write('format = 1\ntasks = "TASKS.md"\narchive = "tasks/archive.md"\n') + cfg = C.load(self.root) + self.assertEqual((cfg.format, cfg.tasks, cfg.archive), (1, self.root / "TASKS.md", self.root / "tasks/archive.md")) + self.assertEqual((cfg.docs, cfg.verify, cfg.done, cfg.ledgers, cfg.anchors_index), ([], [], [], None, None)) + + def test_full(self): + self.write('format = 1\ntasks = "docs/TASKS.md"\narchive = "docs/DONE.md"\ndocs = ["DESIGN.md", "docs/"]\n' + 'verify = ["make test"]\ndone = ["update CATALOG"]\nledgers = ".superpowers/sdd"\n' + '[anchors]\nindex = "DESIGN.md"\nindex_section = "Subsystems"\nspecs = "docs/specs"\n') + cfg = C.load(self.root) + self.assertEqual(cfg.docs, [self.root / "DESIGN.md", self.root / "docs"]) + self.assertEqual((cfg.verify, cfg.done), (["make test"], ["update CATALOG"])) + self.assertEqual(cfg.ledgers, self.root / ".superpowers/sdd") + self.assertEqual((cfg.anchors_index, cfg.anchors_section, cfg.anchors_specs), + (self.root / "DESIGN.md", "Subsystems", self.root / "docs/specs")) + + def test_cloud_keys(self): + self.write('format = 1\ntasks = "T.md"\narchive = "a.md"\n') + cfg = C.load(self.root) + self.assertEqual((cfg.cloud, cfg.cloud_include, cfg.cloud_note), (False, [], None)) + self.write('format = 1\ntasks = "T.md"\narchive = "a.md"\ncloud = true\ncloud_include = ["out/p"]\n' + 'cloud_note = "data in out/p"\n') + cfg = C.load(self.root) + self.assertEqual((cfg.cloud, cfg.cloud_include, cfg.cloud_note), (True, ["out/p"], "data in out/p")) + self.write('format = 1\ntasks = "T.md"\narchive = "a.md"\ncloud = "yes"\n') + with self.assertRaisesRegex(C.ConfigError, "'cloud' must be true or false"): + C.load(self.root) + + def test_unknown_key(self): + self.write('format = 1\ntasks = "T.md"\narchive = "a.md"\nverfy = []\n') + with self.assertRaisesRegex(C.ConfigError, "workflow.toml: unknown key 'verfy'"): + C.load(self.root) + + def test_unknown_anchors_key(self): + self.write('format = 1\ntasks = "T.md"\narchive = "a.md"\n[anchors]\nindx = "D.md"\n') + with self.assertRaisesRegex(C.ConfigError, r"unknown key 'anchors.indx'"): + C.load(self.root) + + def test_missing_required(self): + self.write('format = 1\narchive = "a.md"\n') + with self.assertRaisesRegex(C.ConfigError, "workflow.toml: missing 'tasks'"): + C.load(self.root) + + def test_wrong_type(self): + self.write('format = 1\ntasks = "T.md"\narchive = "a.md"\nverify = "make"\n') + with self.assertRaisesRegex(C.ConfigError, "'verify' must be a list of strings"): + C.load(self.root) + + def test_bad_toml(self): + self.write('format = \n') + with self.assertRaisesRegex(C.ConfigError, "workflow.toml: "): + C.load(self.root) + + + + + +class LaneConfigTest(unittest.TestCase): + def load(self, extra: str): + d = Path(tempfile.mkdtemp()) + self.addCleanup(__import__("shutil").rmtree, d) + (d / "workflow.toml").write_text('format = 1\ntasks = "TASKS.md"\narchive = "a.md"\n' + extra) + return config.load(d) + + def test_default_lanes(self): + cfg = self.load("") + self.assertEqual(cfg.slice_above, "1h") + self.assertEqual(cfg.lanes, (L.Lane("fast", ("<1h",), "unblock", None, True), + L.Lane("slow", ("1h",), "priority", "fast", False))) + self.assertEqual(cfg.area_stale_commits, 20) + self.assertEqual(cfg.areas_file, cfg.root / "CLAUDE.md") + + def test_custom_lanes_replace_default(self): + cfg = self.load('slice_above = "5h"\n[lanes.quick]\nefforts = ["<1h", "1h"]\norder = "unblock"\n' + '[lanes.long]\nefforts = ["5h"]\nfallback = "quick"\n') + self.assertEqual(cfg.lanes, (L.Lane("quick", ("<1h", "1h"), "unblock", None, True), + L.Lane("long", ("5h",), "priority", "quick", False))) + + def test_lane_errors(self): + bad = { + '[lanes.a]\nefforts = ["<1h", "1h"]\n[lanes.b]\nefforts = ["1h"]\n': "effort '1h' in lanes a and b", + '[lanes.a]\nefforts = ["<1h"]\n': "effort '1h' (≤ slice_above) in no lane", + '[lanes.a]\nefforts = ["<1h"]\nslices = true\n[lanes.b]\nefforts = ["1h"]\nslices = true\n': + "slices = true on more than one lane", + '[lanes.a]\nefforts = ["<1h"]\nfallback = "b"\n[lanes.b]\nefforts = ["1h"]\nfallback = "a"\n': + "fallback cycle: a ↔ b", + '[lanes.a]\nefforts = ["<1h", "1h", "5h"]\n': "lane a: effort '5h' above slice_above '1h'", + '[lanes.a]\nefforts = ["<1h", "1h"]\norder = "fifo"\n': "lane a: order 'fifo' (want unblock, priority)", + '[lanes.all]\nefforts = ["<1h", "1h"]\n': "lane name 'all' is reserved", + '[lanes.A]\nefforts = ["<1h", "1h"]\n': "bad lane name 'A'", + '[lanes.a]\nefforts = ["<1h", "1h"]\nspeed = 1\n': "unknown key 'lanes.a.speed'", + 'slice_above = "2h"\n': "slice_above '2h' (want <1h, 1h, 5h, 10h, 100h)", + 'area_stale_commits = 0\n': "'area_stale_commits' must be a number ≥ 1", + } + for extra, msg in bad.items(): + with self.subTest(msg=msg), self.assertRaises(config.ConfigError) as cm: + self.load(extra) + self.assertIn(msg, str(cm.exception)) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_ctx_hint.py b/tests/test_ctx_hint.py new file mode 100644 index 0000000..7750284 --- /dev/null +++ b/tests/test_ctx_hint.py @@ -0,0 +1,98 @@ +import json +import sys +import unittest +from pathlib import Path + +HERE = Path(__file__).resolve().parent.parent +sys.path.insert(0, str(HERE)) +sys.path.insert(0, str(HERE / "tests")) + +import test_cli # noqa: E402 +from test_usage import entry # noqa: E402 +from wflib import usage # noqa: E402 + +SID = "99999999-2222-3333-4444-555555555555" +USER = json.dumps({"type": "user", "timestamp": "2026-10-05T09:00:00Z", "message": {"role": "user", "content": "hi"}}) + + +def transcript(*sizes): + """One request per size (inp 10, cw 1000, rest cache read), each streamed as 2 entries.""" + lines = [USER] + for n, size in enumerate(sizes): + e = entry(f"m{n}", f"r{n}", "claude-opus-5-5", f"2026-10-05T09:0{n}:01Z", inp=10, cw5=1000, cr=size - 1010) + lines += [e, e, USER] + return "\n".join(lines) + "\n" + + +class ContextTokens(unittest.TestCase): + def test_last_request_prompt_size(self): + self.assertEqual(usage.context_tokens(transcript(50_000, 152_000)), 152_000) + + def test_skips_synthetic_and_sidechain(self): + text = transcript(40_000) + synth = json.loads(entry("s", "rs", "<synthetic>", "2026-10-05T10:00:00Z", inp=1)) + side = json.loads(entry("x", "rx", "claude-opus-5-5", "2026-10-05T10:00:01Z", cr=999_000)) + side["isSidechain"] = True + text += json.dumps(synth) + "\n" + json.dumps(side) + "\n" + self.assertEqual(usage.context_tokens(text), 40_000) + + def test_partial_first_line_and_empty(self): + self.assertEqual(usage.context_tokens('ens": 5}}}\n' + transcript(30_000)), 30_000) + self.assertIsNone(usage.context_tokens(USER + "\n")) + + def test_hint_line(self): + self.assertEqual(usage.ctx_hint(152_000, 100_000), + "context ~152k tokens (> 100k): ask the owner to /clear, then continue " + "(subagent: ignore, this is the main session)") + self.assertIsNone(usage.ctx_hint(99_999, 100_000)) + self.assertIsNone(usage.ctx_hint(None, 100_000)) + self.assertIsNone(usage.ctx_hint(500_000, 0)) + + +class HintCli(test_cli.Cli): + def setUp(self): + super().setUp() + self.home = self.root.parent / "claude-home" + (self.home / "projects" / "-work-demo").mkdir(parents=True) + self.jsonl = self.home / "projects" / "-work-demo" / f"{SID}.jsonl" + + def env(self, sid=SID): + return {"CLAUDE_CONFIG_DIR": str(self.home), "CLAUDE_CODE_SESSION_ID": sid} + + HINT = "context ~152k tokens (> 100k)" + + def test_done_prints_hint_when_big(self): + self.jsonl.write_text(transcript(152_000)) + out = self.ok("done", "t-one", "-m", "ok", env=self.env()) + self.assertIn(self.HINT, out.splitlines()[-1]) + + def test_next_prints_hint_when_big(self): + self.jsonl.write_text(transcript(152_000)) + out = self.ok("next", "--as", "opus", env=self.env()) + self.assertIn(self.HINT, out.splitlines()[-1]) + + def test_silent_when_small_or_missing(self): + self.jsonl.write_text(transcript(60_000)) + self.assertNotIn("context ~", self.ok("next", "--as", "opus", env=self.env())) + self.assertNotIn("context ~", self.ok("next", "--as", "opus", env=self.env(sid=""))) + self.assertNotIn("context ~", self.ok("next", "--as", "opus", env=self.env(sid="nope"))) + + def test_threshold_from_toml(self): + (self.root / "workflow.toml").write_text(self.toml + "ctx_hint = 200000\n") + self.jsonl.write_text(transcript(152_000)) + self.assertNotIn("context ~", self.ok("next", "--as", "opus", env=self.env())) + (self.root / "workflow.toml").write_text(self.toml + "ctx_hint = 0\n") + self.jsonl.write_text(transcript(900_000)) + self.assertNotIn("context ~", self.ok("next", "--as", "opus", env=self.env())) + + def test_bad_threshold(self): + (self.root / "workflow.toml").write_text(self.toml + 'ctx_hint = "big"\n') + self.assertIn("'ctx_hint' must be a number", self.fails("list")) + + def test_reads_only_the_tail(self): + self.jsonl.write_text(transcript(152_000) + (USER + "\n") * 40_000 + transcript(160_000)) + self.assertIn("context ~160k tokens", self.ok("next", "--as", "opus", env=self.env())) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_finish.py b/tests/test_finish.py new file mode 100644 index 0000000..38fceb1 --- /dev/null +++ b/tests/test_finish.py @@ -0,0 +1,210 @@ +import subprocess +import unittest + +from test_cli import TOML, Cli +from test_claims import git +from test_merge import IDENT + + +class FinishTest(Cli): + def setUp(self): + super().setUp() + git(self.root, "init", "-q", "-b", "master") + (self.root / ".gitignore").write_text(".worktrees/\n.wf/\n") + git(self.root, "add", "-A") + git(self.root, "commit", "-qm", "init") + self.wt = self.root / ".worktrees" / "sonnet" + git(self.root, "worktree", "add", "-q", str(self.wt), "-b", "sonnet/t-three") + + def out(self, *args, cwd=None): + return subprocess.run(["git", *args], cwd=cwd or self.root, capture_output=True, text=True).stdout + + def finish(self, *args, cwd=None): + return self.wf("finish", *args, project=False, cwd=cwd or self.wt, env=IDENT) + + def write_code(self): + (self.wt / "code.txt").write_text("x\n") + + def test_report_line_tool_commit(self): + self.write_code() + code, out, err = self.finish("t-three", "-m", "ok", "--commit", "impl", "code.txt", "--no-push", + "--tool-commit", "abc1234") + self.assertEqual(code, 0, err) + self.assertRegex(out, r"report: commit [0-9a-f]+ tool abc1234\n$") + + def test_report_line_no_worktree(self): + (self.root / "code.txt").write_text("x\n") + code, out, err = self.wf("finish", "t-three", "-m", "ok", "--commit", "impl", "code.txt", "--no-push", + project=False, cwd=self.root, env=IDENT) + self.assertEqual(code, 0, err) + sha = self.out("rev-parse", "--short", "HEAD").strip() + self.assertTrue(out.endswith(f"report: commit {sha}\n"), out) + + def test_commit_done_merge_in_one(self): + self.write_code() + code, out, err = self.finish("t-three", "-m", "ok", "--commit", "impl", "code.txt", "--no-push") + self.assertEqual((code, err), (0, ""), out) + self.assertIn("done: t-three → tasks/archive.md\n", out) + self.assertIn("committed code.txt\n", out) + self.assertIn("merged sonnet/t-three into master\n", out) + sha = self.out("log", "--format=%h", "-n1", "--grep=^impl$", "master").strip() + self.assertTrue(out.endswith(f"report: commit {sha}\n"), out) + self.assertNotIn("wf merge", out) # no merge-step reminder: finish merged + self.assertEqual(self.out("log", "--format=%s", "master"), "t-three done\nimpl\ninit\n") + self.assertEqual(self.out("status", "--porcelain"), "") + self.assertEqual((self.root / "code.txt").read_text(), "x\n") + self.assertIn("t-three", (self.root / "tasks" / "archive.md").read_text()) + + def test_wip_commits_notes_clears(self): + self.write_code() + self.wf("status", "t-three", "progress", "br", project=False, cwd=self.wt, env=IDENT) + code, out, err = self.wf("wip", "t-three", "-m", "half done, next: tests", "--commit", "wip", "code.txt", + project=False, cwd=self.wt, env=IDENT) + self.assertEqual(code, 0, err) + self.assertIn("committed code.txt", out) + self.assertEqual(self.out("log", "--format=%s", "-n1", cwd=self.wt), "wip\n") + self.assertEqual(self.out("status", "--porcelain", cwd=self.wt, ).count("code.txt"), 0) + code, out, err = self.wf("show", "t-three", project=False, cwd=self.wt) + self.assertIn("half done, next: tests", out) + self.assertNotIn("in progress", out) + + def test_no_commit_bookkeeping_only(self): + git(self.wt, "switch", "-q", "--detach", "master") + code, out, err = self.finish("t-three", "-m", "ok", "--no-push") + self.assertEqual((code, err), (0, ""), out) + self.assertTrue(out.endswith("committed TASKS.md tasks/archive.md\n"), out) + self.assertEqual(self.out("log", "--format=%s", "master"), "bookkeeping\ninit\n") + + def test_commit_tasks_only_no_path_lists_once(self): + code, out, err = self.finish("t-three", "-m", "ok", "--commit", "bk", "--no-push") + self.assertEqual((code, err), (0, ""), out) + self.assertEqual(out.count("TASKS.md"), 1, out) + self.assertIn("committed TASKS.md tasks/archive.md\n", out) + self.assertEqual(self.out("log", "--format=%s", "master"), "t-three done\ninit\n") + + def test_commit_tasks_path_named_lists_once(self): + code, out, err = self.finish("t-three", "-m", "ok", "--commit", "bk", "TASKS.md", "--no-push", cwd=self.root) + self.assertEqual((code, err), (0, ""), out) + self.assertEqual(out.count("TASKS.md"), 1, out) + self.assertEqual(self.out("log", "--format=%s", "master"), "bk\ninit\n") + + def test_dirty_outside_paths_refused_before_done(self): + self.write_code() + (self.wt / "DESIGN.md").write_text("stray\n") + code, out, err = self.finish("t-three", "-m", "ok", "--commit", "impl", "code.txt", "--no-push") + self.assertEqual((code, err), (1, "wf: uncommitted changes outside the --commit paths: DESIGN.md\n")) + self.assertIn("t-three", (self.root / "TASKS.md").read_text()) + self.assertEqual(self.out("log", "--format=%s", "master"), "init\n") + + def test_gate_red_refused_before_done(self): + (self.root / "workflow.toml").write_text(TOML + 'quick_gate = ["exit 3"]\n') + git(self.root, "commit", "-qam", "gate") + git(self.wt, "rebase", "-q", "master") + code, out, err = self.finish("t-three", "-m", "ok", "--no-push") + self.assertEqual(code, 1, out + err) + self.assertIn("quick_gate 'exit 3' red", err) + self.assertIn("wf add -p 0", err) + self.assertIn("done+gate-red", err) + self.assertIn("t-three", (self.root / "TASKS.md").read_text()) + + def test_paths_need_commit_message(self): + code, out, err = self.finish("t-three", "-m", "ok", "code.txt") + self.assertEqual(code, 2, out + err) + self.assertIn("paths need --commit", err) + + def test_main_tree_commits_code_and_bookkeeping(self): + (self.root / "code.txt").write_text("y\n") + code, out, err = self.finish("t-three", "-m", "ok", "--commit", "impl", "code.txt", cwd=self.root) + self.assertEqual((code, err), (0, ""), out) + self.assertEqual(self.out("log", "--format=%s", "master"), "impl\ninit\n") + self.assertEqual(self.out("show", "--stat", "--format=", "master").split("|")[0].strip(), "TASKS.md") + self.assertEqual(self.out("status", "--porcelain"), "") + + def test_main_tree_pushes_home(self): + bare = self.root.parent / "home-finish.git" + subprocess.run(["git", "init", "-q", "--bare", str(bare)], check=True) + git(self.root, "remote", "add", "home", str(bare)) + (self.root / "code.txt").write_text("y\n") + code, out, err = self.finish("t-three", "-m", "ok", "--commit", "impl", "code.txt", cwd=self.root) + self.assertEqual((code, err), (0, ""), out) + self.assertIn("pushed home\n", out) + self.assertEqual(self.out("log", "--format=%s", "master", cwd=bare), "impl\ninit\n") + + def test_main_tree_no_push_flag(self): + bare = self.root.parent / "home-finish2.git" + subprocess.run(["git", "init", "-q", "--bare", str(bare)], check=True) + git(self.root, "remote", "add", "home", str(bare)) + (self.root / "code.txt").write_text("y\n") + code, out, err = self.finish("t-three", "-m", "ok", "--commit", "impl", "code.txt", "--no-push", cwd=self.root) + self.assertEqual((code, err), (0, ""), out) + self.assertNotIn("pushed home", out) + + # books stay with the task: no half-done finish, no main-tree done of a worktree's task + + def test_path_outside_repo_refused_before_done(self): + other = self.root.parent / "other-repo-file.md" + other.write_text("x\n") + code, out, err = self.finish("t-three", "-m", "ok", "--commit", "impl", str(other), "--no-push") + self.assertEqual(code, 1, out + err) + self.assertIn("is outside this repo", err) + self.assertIn("t-three", (self.root / "TASKS.md").read_text()) + self.assertEqual(self.out("status", "--porcelain"), "") + + def test_missing_path_refused_before_done(self): + code, out, err = self.finish("t-three", "-m", "ok", "--commit", "impl", "shared/CLAUDE.md", "--no-push") + self.assertEqual(code, 1, out + err) + self.assertIn("does not exist", err) + self.assertIn("t-three", (self.root / "TASKS.md").read_text()) + + def test_unchanged_paths_refused_before_done(self): + code, out, err = self.finish("t-three", "-m", "ok", "--commit", "impl", "workflow.toml", "--no-push") + self.assertEqual(code, 1, out + err) + self.assertIn("nothing to commit", err) + self.assertIn("t-three", (self.root / "TASKS.md").read_text()) + + def test_worktree_books_copy_refused(self): + (self.wt / "TASKS.md").write_text((self.wt / "TASKS.md").read_text() + "\n") + code, out, err = self.finish("t-three", "-m", "ok", "--commit", "bk", "TASKS.md", "--no-push") + self.assertEqual(code, 1, out + err) + self.assertIn("copy of the books", err) + self.assertIn("t-three", (self.root / "TASKS.md").read_text()) + + def test_already_done_resumes_commit_and_merge(self): + self.write_code() + code, out, err = self.wf("done", "t-three", "-m", "ok", project=False, cwd=self.wt, env=IDENT) + self.assertEqual(code, 0, err) + code, out, err = self.finish("t-three", "-m", "ok", "--commit", "impl", "code.txt", "--no-push") + self.assertEqual((code, err), (0, ""), out) + self.assertIn("t-three already done (tasks/archive.md): resuming", out) + self.assertEqual(self.out("log", "--format=%s", "master"), "t-three done\nimpl\ninit\n") + self.assertEqual(self.out("status", "--porcelain"), "") + self.assertEqual((self.root / "tasks" / "archive.md").read_text().count("**t-three**"), 1) + + def test_main_tree_done_of_worktree_task_refused(self): + self.wf("status", "t-three", "progress", "sonnet/t-three", project=False, cwd=self.wt, env=IDENT) + for cmd in (["done", "t-three", "-m", "ok"], ["finish", "t-three", "-m", "ok", "--no-push"]): + code, out, err = self.wf(*cmd, project=False, cwd=self.root, env=IDENT) + self.assertEqual(code, 1, out + err) + self.assertIn("in progress in worktree", err) + self.assertIn("branch sonnet/t-three", err) + self.assertIn("t-three", (self.root / "TASKS.md").read_text()) + # from the worktree it goes through, books committed on master by the merge + self.write_code() + code, out, err = self.finish("t-three", "-m", "ok", "--commit", "impl", "code.txt", "--no-push") + self.assertEqual((code, err), (0, ""), out) + self.assertEqual(self.out("log", "--format=%s", "master"), "t-three done\nimpl\ninit\n") + + def test_main_tree_done_after_status_clear(self): + self.wf("status", "t-three", "progress", "sonnet/t-three", project=False, cwd=self.wt, env=IDENT) + self.wf("status", "t-three", "clear", project=False, cwd=self.root, env=IDENT) + code, out, err = self.wf("done", "t-three", "-m", "ok", project=False, cwd=self.root, env=IDENT) + self.assertEqual(code, 0, out + err) + + def test_main_tree_done_branch_without_worktree_ok(self): + self.wf("status", "t-three", "progress", "fast/t-three", project=False, cwd=self.root, env=IDENT) + code, out, err = self.wf("done", "t-three", "-m", "ok", project=False, cwd=self.root, env=IDENT) + self.assertEqual(code, 0, out + err) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_gate.py b/tests/test_gate.py new file mode 100644 index 0000000..7a33c1f --- /dev/null +++ b/tests/test_gate.py @@ -0,0 +1,74 @@ +import sys +import tempfile +import unittest +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) +from wflib import config as C +from test_claims import git +from test_cli import TOML +from test_setup import GitCli + +GATE = TOML + 'quick_gate = ["echo \\"$WF_MAIN\\" > gate.txt", "echo second >> gate.txt"]\n' + + +class GateConfigTest(unittest.TestCase): + def setUp(self): + self.tmp = tempfile.TemporaryDirectory() + self.root = Path(self.tmp.name).resolve() + + def tearDown(self): + self.tmp.cleanup() + + def load(self, extra): + (self.root / "workflow.toml").write_text('format = 1\ntasks = "T.md"\narchive = "a.md"\n' + extra) + return C.load(self.root) + + def test_default_empty(self): + self.assertEqual(self.load("").quick_gate, []) + + def test_list(self): + self.assertEqual(self.load('quick_gate = ["make quick"]\n').quick_gate, ["make quick"]) + + def test_wrong_type(self): + with self.assertRaisesRegex(C.ConfigError, "'quick_gate' must be a list of strings"): + self.load('quick_gate = "make quick"\n') + + +class GateCliTest(GitCli): + toml = GATE + + def setUp(self): + super().setUp() + self.wt = self.root / ".worktrees" / "opus" + git(self.root, "worktree", "add", "-q", str(self.wt), "-b", "opus/t-three") + + def test_runs_in_worktree_with_wf_main(self): + code, out, err = self.wf("gate", project=False, cwd=self.wt / "docs") + self.assertEqual((code, err), (0, ""), out) + self.assertEqual((self.wt / "gate.txt").read_text(), f"{self.root}\nsecond\n") + self.assertFalse((self.root / "gate.txt").exists()) + self.assertIn("$ echo second >> gate.txt\n", out) + self.assertTrue(out.endswith("quick_gate: 2 commands green\n"), out) + + def test_runs_in_main_tree(self): + code, out, err = self.wf("gate", project=False, cwd=self.root) + self.assertEqual((code, err), (0, ""), out) + self.assertEqual((self.root / "gate.txt").read_text(), f"{self.root}\nsecond\n") + + def test_red_exit_1_rest_skipped(self): + (self.wt / "workflow.toml").write_text(TOML + 'quick_gate = ["exit 4", "echo c > c.txt"]\n') + code, out, err = self.wf("gate", project=False, cwd=self.wt) + self.assertEqual(code, 1, out + err) + self.assertTrue(err.startswith("wf: quick_gate 'exit 4' red (exit 4): fix it before wf done; later commands skipped\n"), err) + self.assertIn("done+gate-red", err) + self.assertFalse((self.wt / "c.txt").exists()) + + def test_none_configured(self): + (self.wt / "workflow.toml").write_text(TOML) + code, out, err = self.wf("gate", project=False, cwd=self.wt) + self.assertEqual((code, out, err), (0, "no quick_gate in workflow.toml: nothing to do\n", "")) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_human_done.py b/tests/test_human_done.py new file mode 100644 index 0000000..e6affee --- /dev/null +++ b/tests/test_human_done.py @@ -0,0 +1,40 @@ +import sys +import unittest +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) +from wflib import tasks as T + + +def ready(done): + doc = T.parse(f"# T\n\n## Pending\n\n- **t-x** [P2] (<1h): X.\n Done: {done}\n") + return doc.sections[0].items[0].runner_ready if hasattr(doc, "sections") else None + + +class HumanDoneTest(unittest.TestCase): + def test_owner_bound_subject(self): + self.assertTrue(ready("owner-bound task with Done not counted")) + + def test_owner_actions(self): + self.assertFalse(ready("owner confirms X")) + self.assertFalse(ready("owner approves X")) + self.assertFalse(ready("report to owner")) + self.assertFalse(ready("confirm it works")) + + def test_word_alone_is_fine(self): + self.assertTrue(ready("owner-facing docs updated; owner-private names scrubbed")) + self.assertTrue(ready("tests confirm the parser handles X")) + self.assertTrue(ready("the owner file is read")) + + def test_more_actions(self): + self.assertFalse(ready("owner says it is fine")) + self.assertFalse(ready("Owner tests it")) + self.assertFalse(ready("confirm with the owner")) + + def test_match_exposed(self): + doc = T.parse("# T\n\n## Pending\n\n- **t-x** [P2] (<1h): X.\n Done: owner approves X\n") + self.assertEqual(doc.sections[0].items[0].human_done_match, "owner approves") + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_lanes.py b/tests/test_lanes.py new file mode 100644 index 0000000..a9bcf93 --- /dev/null +++ b/tests/test_lanes.py @@ -0,0 +1,408 @@ +import json +import os +import sys +import unittest +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) +from wflib import lanes as L +from wflib import tasks as T +from test_cli import Cli + +ITEMS = """\ +## Awaiting your decision + +## Pending + +- **t-a** [P1] (1h): A. + +- **t-b** [P1] (1h): B. + - After: [[t-s]] + +- **t-s** [P1] (1h): S. + Model: sonnet + +- **t-s2** [P2] (1h): S2. + Model: sonnet + - After: [[t-x]] + +- **t-c** [P2] (1h): C. + - After: [[t-s2]] + +## Needs human + +## Deferred +""" +SOCK = "/run/user/1000/cc-socks/1.sock" + + +class LanesTest(unittest.TestCase): + def setUp(self): + self.doc = T.parse(ITEMS) + + def test_counts(self): + self.assertEqual(L.counts(self.doc, set(), L.DEFAULT_LANES, "1h"), + {"fast": (0, 0, 0), "slow": (2, 3, 0)}) + + def test_counts_held(self): + self.assertEqual(L.counts(self.doc, set(), L.DEFAULT_LANES, "1h", {"t-a": "x"}), + {"fast": (0, 0, 0), "slow": (1, 4, 0)}) + + def test_counts_slice_jobs(self): + self.assertEqual(L.counts(T.parse(PICK), {"t-old1"}, L.DEFAULT_LANES, "1h"), + {"fast": (3, 2, 1), "slow": (1, 1, 0)}) + + def test_lanes_block(self): + counts = {"fast": (3, 1, 1), "slow": (0, 2, 0)} + sessions = {"slow": {"socket": SOCK, "alive": True, "model": "opus"}} + self.assertEqual(L.lanes_block(counts, sessions, "fast"), [ + "fast (you): 3 pickable (1 slice) · 1 waiting", + f"slow: 0 pickable · 2 waiting · session uds:{SOCK} (alive, opus)", + ]) + + def test_lanes_block_no_session(self): + self.assertEqual(L.lanes_block({"fast": (2, 0, 2), "slow": (0, 0, 0)}, {}, None), [ + "fast: 2 pickable (2 slices) · 0 waiting · no session → orchestrator or owner: start one (wf next --lane fast)", + "slow: 0 pickable · 0 waiting · no session", + ]) + + def test_cross_waits_by_lane(self): + doc = T.parse(PICK) + self.assertEqual([(m.id, b.id) for m, b in L.cross_waits(doc, {"t-old1"}, "slow", L.DEFAULT_LANES, "1h")], + [("t-w1", "t-blocker")]) + self.assertEqual(L.cross_waits(doc, {"t-old1"}, "fast", L.DEFAULT_LANES, "1h"), []) + + def test_waiting_block(self): + doc = T.parse(PICK) + waits = L.cross_waits(doc, {"t-old1"}, "slow", L.DEFAULT_LANES, "1h") + self.assertEqual(L.waiting_block(waits, {"fast": {"socket": SOCK, "alive": True}}, L.DEFAULT_LANES, "1h"), + [f'- t-w1 waits on t-blocker (fast lane) → message uds:{SOCK}: ' + '"t-blocker blocks my t-w1, please take it"']) + self.assertEqual(L.waiting_block(waits, {}, L.DEFAULT_LANES, "1h"), + ["- t-w1 waits on t-blocker (fast lane) → no fast session: tell the owner"]) + + def test_notify_block_by_lane(self): + doc = T.parse(PICK) + freed = [doc.item("t-w1"), doc.item("t-w2")] + self.assertEqual(L.notify_block(freed, {"slow": {"socket": SOCK, "alive": True}}, L.DEFAULT_LANES, "1h"), [ + "fast work now pickable: t-w2 (no session: tell the owner)", + f"notify slow uds:{SOCK}: now pickable t-w1", + ]) + + def test_solo_done_block_any_session_name(self): + sessions = {"fast": {"socket": SOCK, "alive": True, "pid": 1}, "all": {"socket": "/b", "alive": True, "pid": 2}, + "slow": {"socket": "/c", "alive": False, "pid": 3}} + self.assertEqual(L.solo_done_block(["t-s"], sessions, "2"), + [f"notify fast uds:{SOCK}: solo t-s done, run wf next"]) + + +PICK = """\ +## Awaiting your decision + +## Pending + +- **t-big** [P2] (5h): Big unsliced. + +- **t-p0** [P0] (<1h): Small P0. + +- **t-blocker** [P3] (<1h): Small, two wait on it. + +- **t-w1** [P2] (1h): Waits. + - After: [[t-blocker]] + +- **t-w2** [P3] (<1h): Waits too. + - After: [[t-blocker]] + Model: sonnet + +- **t-mid** [P1] (1h): Mid. + Model: sonnet + +- **t-done-parent** [P1] (10h): Sliced, slices archived. + - Slices: [[t-old1]] + +## Needs human + +## Deferred +""" + + +class PickTest(unittest.TestCase): + def setUp(self): + self.doc = T.parse(PICK) + + def pick(self, lane=None, model=None, archived=frozenset({"t-old1"})): + item, _ = L.pick(self.doc, set(archived), L.DEFAULT_LANES, "1h", lane, model) + return item.id if item else None + + def test_lane_of(self): + got = {i.id: L.lane_of(i, L.DEFAULT_LANES, "1h") for i in self.doc.section("pending").items} + self.assertEqual(got, {"t-big": "fast", "t-p0": "fast", "t-blocker": "fast", "t-w1": "slow", + "t-w2": "fast", "t-mid": "slow", "t-done-parent": "fast"}) + + def test_no_effort_is_fast(self): + item = T.Item(id="t-x", prio=1) + self.assertEqual(L.lane_of(item, L.DEFAULT_LANES, "1h"), "fast") + + def test_fast_unblock_order(self): + # unblock: waited-on first (t-blocker: 2 waiters), then the slice job t-big (0 waiters) vs t-p0 by prio + self.assertEqual([i.id for i in L.ranked(self.doc, {"t-old1"}, L.DEFAULT_LANES, "1h", "fast")], + ["t-blocker", "t-p0", "t-big"]) + + def test_slow_priority_order_then_fallback(self): + self.assertEqual(self.pick("slow"), "t-mid") + self.assertEqual([i.id for i in L.ranked(self.doc, {"t-old1"}, L.DEFAULT_LANES, "1h", "slow")], + ["t-mid", "t-blocker", "t-p0", "t-big"]) + + def test_fallback_when_lane_empty(self): + doc = T.parse(PICK.replace("(1h): Mid.", "(<1h): Mid.")) # slow has only t-w1 (blocked) + item, _ = L.pick(doc, {"t-old1"}, L.DEFAULT_LANES, "1h", "slow") + self.assertEqual(item.id, "t-blocker") + + def test_all_lanes_priority_order(self): + self.assertEqual(self.pick(None), "t-p0") + + def test_model_ceiling(self): + # haiku session: no task has Model haiku; opus (default) and sonnet are above it + self.assertIsNone(self.pick("fast", "haiku")) + self.assertEqual(self.pick("slow", "sonnet"), "t-mid") + self.assertEqual(self.pick("fast", "sonnet"), None) # t-w2 is blocked; others are opus + self.assertEqual(self.pick("fast", "opus"), "t-blocker") + + def test_slice_job(self): + big = self.doc.item("t-big") + self.assertTrue(L.is_slice_job(big, "1h")) + self.assertFalse(L.is_slice_job(self.doc.item("t-p0"), "1h")) + + def test_sliced_parent_not_slice_job(self): + parent = self.doc.item("t-done-parent") + self.assertFalse(L.is_slice_job(parent, "1h")) + self.assertTrue(L.sliced_out(parent, "1h")) + # skipped lists items ranked before the pick: drop the others so the parent is first + doc = T.parse(PICK[:PICK.index("- **t-big**")] + PICK[PICK.index("- **t-done-parent**"):]) + _, skipped = L.pick(doc, {"t-old1"}, L.DEFAULT_LANES, "1h", "fast") + self.assertIn(("t-done-parent", "sliced: all slices done → wf done t-done-parent or add slices"), + [(i.id, why) for i, why in skipped]) + + +PICK_DONE = PICK.replace("Small P0.\n", "Small P0.\n Done: x\n").replace( + "two wait on it.\n", "two wait on it.\n Done: x\n").replace("Big unsliced.\n", "Big unsliced.\n Done: x\n") + + +class LanesCliTest(Cli): + tasks_text = "# Tasks — demo\n\n" + PICK_DONE + + def env(self, name="me"): + sock = self.root / f"{name}.sock" + sock.write_text("") + return {"CLAUDE_CODE_MESSAGING_SOCKET": str(sock), "CLAUDE_PID": str(os.getpid()), "CLAUDE_CODE_SESSION_ID": name} + + def session(self, name): + return json.loads((self.root / ".wf" / "sessions" / f"{name}.json").read_text()) + + def test_next_lane_registers_lane(self): + out = self.ok("next", "--lane", "slow", "--as", "opus", env=self.env()) + self.assertIn("===== Next task =====\n- **t-mid**", out) + s = self.session("slow") + self.assertEqual((s["lane"], s["model"], s["socket"]), ("slow", "opus", str(self.root / "me.sock"))) + self.assertIn("===== Lanes =====\nfast: 3 pickable (1 slice) · 2 waiting · no session → " + "orchestrator or owner: start one (wf next --lane fast)\n" + "slow (you): 1 pickable · 1 waiting\n\n" + "===== Waiting on other lanes =====\n" + "- t-w1 waits on t-blocker (fast lane) → no fast session: tell the owner\n", out) + + def test_next_without_lane_registers_all(self): + out = self.ok("next", "--as", "opus", env=self.env()) + self.assertIn("===== Next task =====\n- **t-p0**", out) + self.assertEqual(self.session("all")["lane"], "all") + self.assertNotIn("Waiting on other lanes", out) + + def test_slice_job_output(self): + (self.root / "TASKS.md").write_text("# Tasks — demo\n\n## Awaiting your decision\n\n## Pending\n\n" + "- **t-big** [P2] (5h): Big unsliced.\n\n## Needs human\n\n## Deferred\n") + out = self.ok("next", "--lane", "fast", "--as", "opus", "--brief") + self.assertEqual(out, "- **t-big** [P2] (5h): Big unsliced.\n\n" + "===== Slice job (no code) =====\n" + "effort 5h > slice_above 1h: split it, don't implement.\n" + ' wf add "<title>. <goal>" -e <1h|1h> --parent t-big --model haiku|sonnet|opus' + " (then wf body: Steps/Done/Ref)\n" + ' wf note t-big "sliced into …" · wf status t-big clear · not wf done' + " (the parent waits on its slices)\n") + + def test_unknown_lane(self): + self.assertEqual(self.fails("next", "--lane", "nope", code=2), "wf: unknown lane 'nope' (fast, slow)\n") + self.assertEqual(self.fails("list", "--lane", "nope", code=2), "wf: unknown lane 'nope' (fast, slow)\n") + + def test_nothing_pickable_text(self): + self.assertEqual(self.fails("next", "--lane", "fast", "--as", "haiku", "--brief"), + "wf: nothing pickable for fast (haiku) in Pending\n") + self.assertEqual(self.fails("next", "--as", "haiku", "--brief").splitlines()[-1], + "wf: nothing pickable for all lanes (haiku) in Pending") + + def test_list_runner_lane_pick_order(self): + out = self.ok("list", "--runner", "--lane", "fast") + self.assertEqual([l.split()[0] for l in out.splitlines()[:-1]], ["t-blocker", "t-p0", "t-big"]) + + def test_list_lane_filter_and_column(self): + out = self.ok("list", "--lane", "slow") + self.assertEqual(out.splitlines()[:2], ["t-w1 P2 1h - opus slow Waits", + "t-mid P1 1h - sonnet slow Mid"]) + + def test_lanes_command(self): + out = self.ok("lanes", "--lane", "fast", "--as", "sonnet", env=self.env()) + self.assertEqual(out, "fast (you): 3 pickable (1 slice) · 2 waiting\n" + "slow: 0 pickable · 1 waiting · 1 not runner-ready (no Done) · no session\n") + self.assertEqual((self.session("fast")["lane"], self.session("fast")["model"]), ("fast", "sonnet")) + self.ok("lanes", "--unregister", env=self.env()) + self.assertFalse((self.root / ".wf" / "sessions" / "fast.json").exists()) + + def test_done_notifies_by_lane(self): + sock = self.root / "s.sock" + sock.write_text("") + d = self.root / ".wf" / "sessions" + d.mkdir(parents=True) + (d / "slow.json").write_text(json.dumps({"lane": "slow", "model": "opus", "socket": str(sock), + "pid": os.getpid()})) + out = self.ok("done", "t-blocker", "-m", "ok", env={"CLAUDE_CODE_MESSAGING_SOCKET": ""}) + self.assertIn(f"notify slow uds:{sock}: now pickable t-w1\n", out) + self.assertNotIn("t-w2", out) # same lane as t-blocker: the doer sees it + + + + +WAIT_NONE = """\ +## Awaiting your decision + +## Pending + +- **t-x** [P1] (1h): X. + - After: [[t-nope]] + +## Needs human + +## Deferred +""" + + +class LanesWaitTest(Cli): + tasks_text = WAIT_NONE + env = {"WF_LANES_POLL": "0.1"} + + def test_pickable_exits_0_at_once(self): + (self.root / "TASKS.md").write_text(DONE_ITEMS) + out = self.ok("lanes", "--wait", "5", env=self.env) + self.assertIn("slow: 2 pickable", out) + + def test_timeout_exits_1(self): + self.assertEqual(self.fails("lanes", "--wait", "1", env=self.env), "") + + def test_added_mid_wait(self): + import threading + import time + + def add(): + time.sleep(0.6) + (self.root / "TASKS.md").write_text(DONE_ITEMS) + th = threading.Thread(target=add) + th.start() + out = self.ok("lanes", "--wait", "10", env=self.env) + th.join() + self.assertIn("slow: 2 pickable", out) + + def test_stop_file_exits_2_at_once(self): + (self.root / "out").mkdir() + (self.root / "out" / "wf-batch.stop").touch() + code, out, err = self.wf("lanes", "--wait", "30", env=self.env) + self.assertEqual((code, err), (2, "")) + self.assertIn("stop requested: out/wf-batch.stop", out) + + def test_no_done_not_pickable(self): + (self.root / "TASKS.md").write_text(NO_DONE) + code, out, err = self.wf("lanes", "--wait", "1", env=self.env) + self.assertEqual((code, err), (1, "")) + self.assertEqual(out, + "fast: 0 pickable · 0 waiting · 2 not runner-ready (no Done) · no session\n" + "slow: 0 pickable · 0 waiting · no session\n") + + def test_lanes_plain_shows_not_ready(self): + (self.root / "TASKS.md").write_text(NO_DONE) + self.assertIn("fast: 0 pickable · 0 waiting · 2 not runner-ready (no Done)", self.ok("lanes")) + + +NO_DONE = """\ +## Awaiting your decision + +## Pending + +- **t-a** [P1] (<1h): A. + +- **t-b** [P1] (<1h): B. + +## Needs human + +## Deferred +""" + +DONE_ITEMS = ITEMS.replace("Model: sonnet\n", "Model: sonnet\n Done: ok.\n").replace("(1h): A.", "(1h): A.\n Done: ok.") \ + .replace("(1h): B.", "(1h): B.\n Done: ok.") + + +class LibRunnerTest(unittest.TestCase): + def test_counts_runner_and_not_ready(self): + doc = T.parse(NO_DONE.replace("A.", "A.\n Done: ok.")) + a = (doc, set(), L.DEFAULT_LANES, "1h") + self.assertEqual(L.counts(*a), {"fast": (2, 0, 0), "slow": (0, 0, 0)}) + self.assertEqual(L.counts(*a, runner=True), {"fast": (1, 0, 0), "slow": (0, 0, 0)}) + self.assertEqual(L.not_ready(*a), {"fast": 1, "slow": 0}) + + def test_not_ready_skips_owner_bound_with_done(self): + doc = T.parse(NO_DONE.replace("A.", "A.\n Done: ok.\n Sessions: owner")) + a = (doc, set(), L.DEFAULT_LANES, "1h") + self.assertEqual(L.not_ready(*a), {"fast": 1, "slow": 0}) + self.assertEqual(len(L.prep_targets(*a)), 1) + + +PREP = """\ +## Pending + +- **t-ok** [P1] (<1h): Has Done. + Done: x works. + +- **t-low** [P3] (<1h): Low, no Done. + +- **t-wait** [P0] (1h): Waits, no Done. + - After: [[t-low]] + +- **t-hi** [P1] (1h): High, no Done. + +- **t-own** [P0] (<1h): Owner, no Done. + Sessions: owner + +- **t-prog** [P0] (<1h) (in progress: w): Running. + +- **t-blk** [P0] (<1h) (blocked: a-q): Blocked. + +- **t-par** [P0] (5h): Parent. + +- **t-par-1** [P2] (<1h): Slice. + Done: y. + +## Needs human + +- **t-h** [P0] (<1h): Human. +""" + + +class PrepTargetsTest(unittest.TestCase): + def ids(self, **kw): + return [i.id for i in L.prep_targets(T.parse(PREP), set(), L.DEFAULT_LANES, "1h", **kw)] + + def test_pickable_first_then_prio(self): + self.assertEqual(self.ids(), ["t-hi", "t-low", "t-wait"]) + + def test_lane_filter(self): + self.assertEqual(self.ids(only={"fast"}), ["t-low"]) + self.assertEqual(self.ids(only={"slow"}), ["t-hi", "t-wait"]) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_ledgers.py b/tests/test_ledgers.py new file mode 100644 index 0000000..1a1add1 --- /dev/null +++ b/tests/test_ledgers.py @@ -0,0 +1,26 @@ +import sys +import unittest +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) +from wflib import ledgers as L + +PLAN = "# P\n**Goal:** g\n### Task 1: A\nx\n### Task 2: B\n### Task 3: C\n" + + +class SummaryTest(unittest.TestCase): + def test_resumes_at_first_task_without_complete_line(self): + ledger = "# SDD ledger — plan: p.md\nTask 1: complete (x)\nTask 2: Ruling: y\nTask 2: complete (z)\n" + self.assertEqual(L.summary(PLAN, ledger), "done 1, 2; resume at Task 3 (task-start PLAN 3)") + + def test_ruling_line_is_not_completion(self): + self.assertEqual(L.summary(PLAN, "Task 1: Ruling: complete rewrite\n"), + "done none; resume at Task 1 (task-start PLAN 1)") + + def test_all_done_reports_last_final_review_line(self): + ledger = "Task 1: complete\nTask 2: complete\nTask 3: complete\nFinal review: running\nFinal review: clean\n" + self.assertEqual(L.summary(PLAN, ledger), "done 1, 2, 3; all tasks done; Final review: clean") + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_lock.py b/tests/test_lock.py new file mode 100644 index 0000000..15d5e88 --- /dev/null +++ b/tests/test_lock.py @@ -0,0 +1,30 @@ +import fcntl +import subprocess +import sys +import time +import unittest + +from test_cli import Cli, WF + + +class LockTest(Cli): + def test_write_waits_for_lock(self): + lock_path = self.root / ".wf" / "lock" + lock_path.parent.mkdir(exist_ok=True) + with open(lock_path, "w") as held: + fcntl.flock(held, fcntl.LOCK_EX) + p = subprocess.Popen([sys.executable, str(WF), "--project", str(self.root), "note", "t-three", "first"], + stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True) + time.sleep(0.5) + self.assertIsNone(p.poll(), "write ran while the lock was held") + self.ok("show", "t-three") # read commands never wait + out, err = p.communicate(timeout=10) + self.assertEqual((p.returncode, err), (0, "")) + self.ok("note", "t-three", "second") + body = self.item("t-three") + self.assertIn("first", body) + self.assertIn("second", body) + + def test_lock_file_is_git_ignored_folder(self): + self.ok("note", "t-three", "x") + self.assertEqual((self.root / ".wf" / ".gitignore").read_text(), "*\n") diff --git a/tests/test_merge.py b/tests/test_merge.py new file mode 100644 index 0000000..cb14a20 --- /dev/null +++ b/tests/test_merge.py @@ -0,0 +1,159 @@ +import fcntl +import subprocess +import sys +import time +import unittest + +from test_cli import Cli, WF +from test_claims import git + +IDENT = {"GIT_AUTHOR_NAME": "t", "GIT_AUTHOR_EMAIL": "t@t", "GIT_COMMITTER_NAME": "t", "GIT_COMMITTER_EMAIL": "t@t", + "CLAUDE_CODE_MESSAGING_SOCKET": ""} + + +class MergeTest(Cli): + def setUp(self): + super().setUp() + git(self.root, "init", "-q", "-b", "master") + (self.root / ".gitignore").write_text(".worktrees/\n.wf/\n") + git(self.root, "add", "-A") + git(self.root, "commit", "-qm", "init") + self.wt = self.root / ".worktrees" / "sonnet" + git(self.root, "worktree", "add", "-q", str(self.wt), "-b", "sonnet/t-three") + + def merge(self, *args): + return self.wf("merge", *args, project=False, cwd=self.wt, env=IDENT) + + def out(self, *args, cwd=None): + return subprocess.run(["git", *args], cwd=cwd or self.root, capture_output=True, text=True).stdout + + def commit_code(self): + (self.wt / "code.txt").write_text("x\n") + git(self.wt, "add", "code.txt") + git(self.wt, "commit", "-qm", "code") + + def test_merge_ff_commits_bookkeeping_and_detaches(self): + self.commit_code() + self.wf("done", "t-three", "-m", "ok", project=False, cwd=self.wt, env=IDENT) + code, out, err = self.merge("--no-push") + self.assertEqual((code, err), (0, ""), out) + self.assertEqual(out, "rebased onto master\nfast-forwarded master\n" + "committed TASKS.md tasks/archive.md\nmerged sonnet/t-three into master\n") + self.assertEqual(self.out("log", "--format=%s", "master"), "t-three done\ncode\ninit\n") + self.assertEqual(self.out("branch", "--list", "sonnet/t-three"), "") + self.assertEqual(self.out("status", "--porcelain"), "") + self.assertEqual((self.root / "code.txt").read_text(), "x\n") + + def test_merge_commits_notes_and_followups_after_done(self): + self.commit_code() + self.wf("done", "t-three", "-m", "ok", project=False, cwd=self.wt, env=IDENT) + r = self.wf("add", "-p", "2", "-e", "1h", "follow up", project=False, cwd=self.wt, env=IDENT); self.assertEqual(r[0], 0, r) + code, out, err = self.merge("--no-push") + self.assertEqual((code, err), (0, ""), out) + self.assertEqual(self.out("status", "--porcelain"), "") + self.assertIn("follow up", self.out("show", "master:TASKS.md")) + + def test_merge_rebases_onto_moved_master(self): + self.commit_code() + (self.root / "other.txt").write_text("o\n") + git(self.root, "add", "other.txt") + git(self.root, "commit", "-qm", "other") + code, out, err = self.merge("--no-push") + self.assertEqual((code, err), (0, ""), out) + self.assertEqual(self.out("log", "--format=%s", "master"), "code\nother\ninit\n") + + def test_merge_conflict_aborts_and_keeps_branch(self): + (self.root / "DESIGN.md").write_text("main side\n") + git(self.root, "commit", "-qam", "main edit") + (self.wt / "DESIGN.md").write_text("branch side\n") + git(self.wt, "commit", "-qam", "branch edit") + code, out, err = self.merge("--no-push") + self.assertEqual((code, err), (1, "wf: rebase onto master conflicts: git rebase master, resolve, verify, " + "then wf merge again\n")) + self.assertEqual((self.wt / "DESIGN.md").read_text(), "branch side\n") + self.assertEqual(self.out("status", "--porcelain", cwd=self.wt), "") + self.assertEqual(self.out("branch", "--show-current", cwd=self.wt), "sonnet/t-three\n") + + def test_merge_refuses_dirty_worktree(self): + (self.wt / "DESIGN.md").write_text("dirty\n") + code, out, err = self.merge("--no-push") + self.assertEqual((code, err), (1, "wf: worktree has uncommitted changes: commit them first\n")) + + def test_merge_outside_worktree_fails(self): + self.assertEqual(self.fails("merge"), "wf: merge runs inside a linked git worktree (lane worktree)\n") + + def test_merge_detached_fails(self): + git(self.wt, "switch", "-q", "--detach", "master") + code, out, err = self.merge("--no-push") + self.assertEqual((code, err), (1, "wf: worktree is on a detached HEAD: nothing to merge\n")) + + def test_merge_detached_commits_leftover_bookkeeping(self): + self.commit_code() + self.wf("done", "t-three", "-m", "ok", project=False, cwd=self.wt, env=IDENT) + self.merge("--no-push") + self.wf("add", "-p", "2", "-e", "1h", "late follow up", project=False, cwd=self.wt, env=IDENT) + code, out, err = self.merge("--no-push") + self.assertEqual((code, err, out), (0, "", "committed TASKS.md tasks/archive.md\n")) + self.assertEqual(self.out("status", "--porcelain"), "") + self.assertIn("late follow up", self.out("show", "master:TASKS.md")) + + def test_merge_detached_with_impl_commit_merges_it(self): + git(self.wt, "switch", "-q", "--detach", "master") + (self.root / "other.txt").write_text("o\n") + git(self.root, "add", "other.txt") + git(self.root, "commit", "-qm", "other") + self.commit_code() + r = self.wf("done", "t-three", "-m", "ok", project=False, cwd=self.wt, env=IDENT) + self.assertIn("detached HEAD has 1 commit not in master: wf merge merges it", r[1]) + code, out, err = self.merge("--no-push") + self.assertEqual((code, err), (0, ""), out) + self.assertEqual(out, "rebased onto master\nfast-forwarded master\n" + "committed TASKS.md tasks/archive.md\nmerged detached HEAD into master\n") + self.assertEqual(self.out("log", "--format=%s", "master"), "bookkeeping\ncode\nother\ninit\n") + self.assertEqual((self.root / "code.txt").read_text(), "x\n") + self.assertEqual(self.out("rev-parse", "HEAD", cwd=self.wt), self.out("rev-parse", "master")) + self.assertEqual(self.out("status", "--porcelain"), "") + + def test_merge_detached_refuses_dirty_worktree(self): + git(self.wt, "switch", "-q", "--detach", "master") + (self.wt / "code.txt").write_text("x\n") + code, out, err = self.merge("--no-push") + self.assertEqual((code, err), (1, "wf: worktree has uncommitted changes: commit them first\n")) + + def test_merge_detached_conflict_aborts_and_keeps_commit(self): + git(self.wt, "switch", "-q", "--detach", "master") + (self.root / "DESIGN.md").write_text("main side\n") + git(self.root, "commit", "-qam", "main edit") + (self.wt / "DESIGN.md").write_text("branch side\n") + git(self.wt, "commit", "-qam", "branch edit") + code, out, err = self.merge("--no-push") + self.assertEqual((code, err), (1, "wf: rebase onto master conflicts: git rebase master, resolve, verify, " + "then wf merge again\n")) + self.assertEqual(self.out("log", "-1", "--format=%s", cwd=self.wt), "branch edit\n") + + def test_merge_pushes_to_home(self): + bare = self.root.parent / "home.git" + subprocess.run(["git", "init", "-q", "--bare", str(bare)], check=True) + git(self.root, "remote", "add", "home", str(bare)) + self.commit_code() + code, out, err = self.merge() + self.assertEqual((code, err), (0, ""), out) + self.assertIn("pushed home\n", out) + self.assertEqual(self.out("log", "--format=%s", "master", cwd=bare), "code\ninit\n") + + def test_merge_waits_for_lock(self): + self.commit_code() + (self.root / ".wf").mkdir(exist_ok=True) + with open(self.root / ".wf" / "lock", "w") as held: + fcntl.flock(held, fcntl.LOCK_EX) + import os + p = subprocess.Popen([sys.executable, str(WF), "merge", "--no-push"], cwd=self.wt, text=True, + stdout=subprocess.PIPE, stderr=subprocess.PIPE, env={**os.environ, **IDENT}) + time.sleep(0.5) + self.assertIsNone(p.poll(), "merge ran while the lock was held") + out, err = p.communicate(timeout=20) + self.assertEqual((p.returncode, err), (0, "")) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_migrate.py b/tests/test_migrate.py new file mode 100644 index 0000000..bd02f92 --- /dev/null +++ b/tests/test_migrate.py @@ -0,0 +1,160 @@ +import sys +import unittest +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) +from wflib import migrate as M +from wflib import tasks as T + +OLD = """\ +# Tasks — demo + +Conventions: CLAUDE.md. + +## Awaiting your decision + +- **Smoke screenshots (need you present)**: windows open hidden, + so screenshots are black. +- Online mod updates (spec step 4) not built. + +## Pending + +1. **[P1] Terrain: cave seams** (Effort: <1h) — close the slit beside the lintel. + - Tests (red first): no edge left open. + - After: Terrain 6/6: cave rendering + sound + Reference: [DESIGN.md#terrain](DESIGN.md#terrain) + +2. N. **[P2] Reputation factions** (Effort: 5h, interactive) (in progress: master; slices 1–2/3 done (see plan)) — make factions friendly. + - Steps: use the table + - nested + - After: Terrain: cave seams + Reference: docs/notes/exe-map.md (ai.c rows), [spec](docs/spec.md#rep). + +3. **[P2] Spirit Powers 3/5: technique keys** (Effort: 1h) — keys 1/2/3. + +4. **[P3] Spirit Powers 4/5: passives. And more** (Effort: 1h) + - After: Spirit Powers 3/5: technique keys + +## Needs human + +Only what a script can't judge. + +1. **[P1] Play-test feel** (Effort: <1h) — last pass 2026-09-13. + - [ ] keyboard works + - [ ] motion smooth + +## Deferred (explicitly out of scope, tracked for later) + +**Final art**: later. +""" + +NEW = """\ +# Tasks — demo + +Conventions: CLAUDE.md. + +## Awaiting your decision + +- **a-smoke-screenshots**: Smoke screenshots (need you present). windows open hidden, + so screenshots are black. + +- **a-online-mod-updates-not-built**: Online mod updates (spec step 4) not built. + +## Pending + +- **t-terrain-cave-seams** [P1] (<1h): Terrain: cave seams. close the slit beside the lintel. + - Tests (red first): no edge left open. + Ref: DESIGN.md#terrain + +- **t-reputation-factions** [P2] (5h, interactive) (in progress: master; slices 1–2/3 done [see plan]): Reputation factions. make factions friendly. + - Steps: use the table + - nested + - After: [[t-terrain-cave-seams]] + Ref: docs/notes/exe-map.md (ai.c rows), docs/spec.md#rep + +- **t-spirit-powers-3** [P2] (1h): Spirit Powers 3/5: technique keys. keys 1/2/3. + +- **t-spirit-powers-4** [P3] (1h): Spirit Powers 4/5: passives, And more. + - After: [[t-spirit-powers-3]] + +## Needs human + +Only what a script can't judge. + +- **t-play-test-feel** [P1] (<1h): Play-test feel. last pass 2026-09-13. + - [ ] keyboard works + - [ ] motion smooth + +## Deferred + +(explicitly out of scope, tracked for later) + +**Final art**: later. +""" + + +class MigrateTest(unittest.TestCase): + def test_converts_all_oddities(self): + new, report = M.migrate(OLD) + self.assertEqual(new, NEW) + self.assertEqual(report.ids, { + "Smoke screenshots (need you present)": "a-smoke-screenshots", + "Online mod updates (spec step 4) not built": "a-online-mod-updates-not-built", + "Terrain: cave seams": "t-terrain-cave-seams", + "Reputation factions": "t-reputation-factions", + "Spirit Powers 3/5: technique keys": "t-spirit-powers-3", + "Spirit Powers 4/5: passives. And more": "t-spirit-powers-4", + "Play-test feel": "t-play-test-feel", + }) + self.assertEqual(report.notes, [ + "t-terrain-cave-seams: dropped 'After: Terrain 6/6: cave rendering + sound' (no open task with that title: done)"]) + + def test_result_parses_clean(self): + doc = T.parse(M.migrate(OLD)[0]) + self.assertEqual([i.error for i in doc.all_items()], [None] * 7) + self.assertEqual(T.render(doc), NEW) + + def test_idempotent(self): + new, report = M.migrate(NEW) + self.assertEqual(new, NEW) + self.assertEqual((report.ids, report.notes), ({}, [])) + + def test_missing_sections_are_added(self): + new, _ = M.migrate("# T\n\n## Pending\n\n1. **[P1] A** (Effort: 1h) — a.\n\n## Notes\n\nprose\n") + self.assertEqual(new, "# T\n\n## Pending\n\n- **t-a** [P1] (1h): A. a.\n\n## Notes\n\nprose\n\n" + "## Awaiting your decision\n\n## Needs human\n\n## Deferred\n") + + def test_same_title_twice_gets_distinct_ids(self): + new, _ = M.migrate("## Pending\n\n1. **[P1] A** (Effort: 1h) — x.\n\n2. **[P1] A** (Effort: 1h) — y.\n") + self.assertIn("- **t-a** [P1] (1h): A. x.", new) + self.assertIn("- **t-a-2** [P1] (1h): A. y.", new) + + def test_ids_avoid_the_archive(self): + new, _ = M.migrate("## Pending\n\n1. **[P1] A** (Effort: 1h) — x.\n", taken={"t-a"}) + self.assertIn("- **t-a-2** [P1] (1h): A. x.", new) + + def test_one_column_deeper_under_a_colon_line_is_a_child(self): + new, _ = M.migrate("## Pending\n\n1. **[P1] A** (Effort: 1h) — x.\n - Seed Qs:\n - first?\n - second?\n" + " - [ ] box\n - [ ] sibling typo\n") + self.assertIn("- **t-a** [P1] (1h): A. x.\n - Seed Qs:\n - first?\n - second?\n" + " - [ ] box\n - [ ] sibling typo\n", new) + + def test_reference_words_that_are_no_path_become_a_note(self): + new, _ = M.migrate("## Pending\n\n1. **[P1] A** (Effort: 1h) — x.\n" + " Reference: docs/a.md (fate rows), skill row notes, src/X.cs, DESIGN.md#terrain\n\n" + "2. **[P1] B** (Effort: 1h) — y.\n Reference: main spec §5\n") + self.assertIn(" Ref: docs/a.md (fate rows; skill row notes), src/X.cs, DESIGN.md#terrain\n", new) + self.assertIn("- **t-b** [P1] (1h): B. y.\n - Reference: main spec §5\n", new) + + def test_reference_note_keeps_the_body_indent(self): + new, _ = M.migrate("## Pending\n\n1. **[P1] A** (Effort: 1h) — x.\n - Steps: s\n Reference: main spec §5\n") + self.assertIn("- **t-a** [P1] (1h): A. x.\n - Steps: s\n - Reference: main spec §5\n", new) + + def test_unreadable_numbered_item_is_reported_and_kept(self): + new, report = M.migrate("## Pending\n\n1. something without the pattern\n body\n") + self.assertIn("1. something without the pattern\n body\n", new) + self.assertEqual(report.notes, ["line 3: numbered item not understood, left as it is: 1. something without the pattern"]) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_model.py b/tests/test_model.py new file mode 100644 index 0000000..b2e0878 --- /dev/null +++ b/tests/test_model.py @@ -0,0 +1,247 @@ +import json +import os +import subprocess +import sys +import unittest +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) +from wflib import tasks as T +from wflib import lanes as L +from test_cli import Cli +from test_check import Base, CLEAN +from wflib import check as K +from wflib import config as C + +LANES = """\ +# Tasks — demo + +## Awaiting your decision + +## Pending + +- **t-one** [P1] (1h): One. + - Steps: a + Model: sonnet + - After: [[t-done]] + +- **t-two** [P2] (1h): Two. + - Model: haiku + +- **t-three** [P2] (<1h): Three. + +## Needs human + +## Deferred +""" + + +class ModelLineTest(unittest.TestCase): + def setUp(self): + self.doc = T.parse(LANES) + + def test_parsed_from_body_default_opus(self): + self.assertEqual([i.model for i in self.doc.section("pending").items], ["sonnet", "haiku", "opus"]) + + def test_extra_words_ignored(self): + item = T.parse_block("- **t-x** [P1] (1h): X.\n - Model: sonnet ok\n") + self.assertEqual(item.model, "sonnet") + + def test_set_replaces_in_place(self): + T.set_fields(self.doc, "t-one", model="opus") + self.assertEqual(self.doc.item("t-one").body, [" - Steps: a", " Model: opus", " - After: [[t-done]]"]) + + def test_set_adds_before_after_and_ref(self): + doc = T.parse(CLEAN) + T.set_fields(doc, "t-one", model="haiku") + self.assertEqual(doc.item("t-one").body, + [" Model: haiku", " - After: [[t-done]]", " Ref: DESIGN.md#terrain, docs/specs/terrain.md"]) + + def test_set_empty_removes(self): + T.set_fields(self.doc, "t-two", model="") + self.assertEqual(self.doc.item("t-two").body, []) + self.assertEqual(self.doc.item("t-two").model, "opus") + + def test_set_bad_value(self): + with self.assertRaisesRegex(T.TaskError, r"model 'gpt' \(want haiku, sonnet, opus\)"): + T.set_fields(self.doc, "t-one", model="gpt") + + def test_note_goes_before_model_line(self): + T.add_note(self.doc, "t-two", "why") + self.assertEqual(self.doc.item("t-two").body, [" - why", " - Model: haiku"]) + + +def doc(items): + return T.parse("## Awaiting your decision\n\n## Pending\n\n" + items + "\n## Needs human\n\n## Deferred\n") + + +class PickOrderTest(unittest.TestCase): + def pick(self, items, model="opus", archived=()): + item, _ = L.pick(doc(items), set(archived), L.DEFAULT_LANES, "1h", None, model) + return item.id if item else None + + def test_priority_inherited_from_waiting_task(self): + self.assertEqual(self.pick("- **t-c** [P1] (1h): C.\n\n- **t-b** [P3] (1h): B.\n\n" + "- **t-a** [P0] (1h): A.\n - After: [[t-b]]\n"), "t-b") + + def test_inherited_transitively(self): + self.assertEqual(self.pick("- **t-c** [P1] (1h): C.\n\n- **t-x** [P3] (1h): X.\n\n" + "- **t-b** [P3] (1h): B.\n - After: [[t-x]]\n\n" + "- **t-a** [P0] (1h): A.\n - After: [[t-b]]\n"), "t-x") + + def test_unblocking_only_low_work_does_not_jump(self): + self.assertEqual(self.pick("- **t-c** [P1] (1h): C.\n\n- **t-b** [P3] (1h): B.\n\n" + "- **t-a** [P3] (1h): A.\n - After: [[t-b]]\n"), "t-c") + + def test_other_lane_waiting_first(self): + self.assertEqual(self.pick("- **t-f** [P2] (1h): F.\n\n- **t-k** [P2] (1h): K.\n\n" + "- **t-l** [P2] (1h): L.\n - After: [[t-k]]\n\n- **t-g** [P2] (1h): G.\n\n" + "- **t-h** [P2] (<1h): H.\n - After: [[t-g]]\n"), "t-g") + + def test_more_waiting_first(self): + self.assertEqual(self.pick("- **t-f** [P2] (1h): F.\n - After: [[t-z]]\n\n- **t-g** [P2] (1h): G.\n\n" + "- **t-i** [P2] (1h): I.\n - After: [[t-g]]\n\n" + "- **t-j** [P2] (1h): J.\n - After: [[t-i]]\n\n" + "- **t-z** [P2] (1h): Z.\n", archived=()), "t-g") + + def test_parent_waits_on_its_slices(self): + self.assertEqual(self.pick("- **t-q** [P1] (1h): Q.\n\n- **t-p** [P0] (5h): P.\n - Slices: [[t-p-1]]\n\n" + "- **t-p-1** [P2] (1h): P1.\n"), "t-p-1") + + def test_lane_filter(self): + items = ("- **t-a** [P0] (1h): A.\n\n- **t-b** [P1] (1h): B.\n Model: sonnet\n\n" + "- **t-c** [P2] (1h): C.\n Model: haiku\n") + self.assertEqual([self.pick(items, m) for m in ("opus", "sonnet", "haiku", None)], + ["t-a", "t-b", "t-c", "t-a"]) + self.assertIsNone(self.pick("- **t-a** [P0] (1h): A.\n", "haiku")) + + def test_skipped_are_ranked_before_pick(self): + d = doc("- **t-a** [P0] (1h) (blocked: [[a-k]]): A.\n\n- **t-s** [P0] (1h): S.\n Model: sonnet\n\n" + "- **t-b** [P1] (1h): B.\n\n- **t-c** [P2] (1h): C.\n - After: [[t-b]]\n") + item, skipped = L.pick(d, set(), L.DEFAULT_LANES, "1h", "slow", "opus") + self.assertEqual((item.id, [(i.id, why) for i, why in skipped]), ("t-s", [("t-a", "blocked: a-k")])) + item, skipped = L.pick(d, set(), L.DEFAULT_LANES, "1h", "slow", "haiku") + self.assertEqual((item, skipped), (None, [])) + + +class ModelCheckTest(Base): + def test_values(self): + self.pending("- **t-a** [P1] (1h): A.\n Model: gpt\n\n" + "- **t-b** [P1] (1h): B.\n - Model: sonnet ok\n\n" + "- **t-c** [P1] (1h): C.\n Model: haiku\n Model: opus\n\n" + "- **t-d** [P1] (1h): D.\n") + errors, warnings = K.check(C.load(self.root)) + self.assertEqual([(p.id, p.message) for p in errors], + [("t-a", "Model 'gpt' (want haiku, sonnet, opus)"), ("t-c", "two Model lines")]) + self.assertEqual([(p.id, p.message) for p in warnings], + [("t-b", "Model line 'sonnet ok': write 'Model: sonnet' (wf set --model)")]) + + +class ModelCliTest(Cli): + tasks_text = LANES + + def test_list_column_and_filter(self): + self.assertEqual(self.ok("list"), + "t-one P1 1h - sonnet slow One\n" + "t-two P2 1h - haiku slow Two\n" + "t-three P2 <1h - opus fast Three\n" + "pending 3 · human 0 · awaiting 0 · deferred 0\n") + self.assertEqual(self.ok("list", "--model", "opus"), + "t-three P2 <1h - opus fast Three\n" + "pending 3 · human 0 · awaiting 0 · deferred 0\n") + + def test_add_and_set(self): + self.assertEqual(self.ok("add", "Four. Goal.", "-p", "3", "-e", "1h", "--model", "haiku", "--ref", "DESIGN.md"), + "- **t-four** [P3] (1h): Four. Goal.\n") + self.assertEqual(self.item("t-four"), "- **t-four** [P3] (1h): Four. Goal.\n Model: haiku\n Ref: DESIGN.md\n") + self.ok("set", "t-four", "--model", "sonnet") + self.assertEqual(self.item("t-four"), "- **t-four** [P3] (1h): Four. Goal.\n Model: sonnet\n Ref: DESIGN.md\n") + self.ok("set", "t-four", "--model", "") + self.assertEqual(self.item("t-four"), "- **t-four** [P3] (1h): Four. Goal.\n Ref: DESIGN.md\n") + + def test_next_model_ceiling(self): + # --as M takes tasks with Model ≤ M; no --lane = all lanes, priority order + self.assertEqual(self.ok("next", "--as", "haiku", "--brief"), "- **t-two** [P2] (1h): Two.\n - Model: haiku\n") + self.assertEqual(self.ok("next", "--as", "opus", "--brief").splitlines()[0], "- **t-one** [P1] (1h): One.") + self.assertEqual(self.ok("next", "--as", "sonnet", "--brief").splitlines()[0], "- **t-one** [P1] (1h): One.") + self.assertEqual(self.ok("next", "--lane", "fast", "--as", "opus", "--brief"), "- **t-three** [P2] (<1h): Three.\n") + self.ok("set", "t-two", "--model", "opus") + self.assertEqual(self.fails("next", "--as", "haiku", "--brief").splitlines()[-1], + "wf: nothing pickable for all lanes (haiku) in Pending") + + def test_next_without_as_is_haiku_with_hint(self): + out = self.ok("next", "--brief") + self.assertEqual(out, "no --as: treated as haiku; pass --as haiku|sonnet|opus\n" + "- **t-two** [P2] (1h): Two.\n - Model: haiku\n") + + def session_env(self, sock): + sock.write_text("") + return {"CLAUDE_CODE_MESSAGING_SOCKET": str(sock), "CLAUDE_PID": str(os.getpid()), + "CLAUDE_CODE_SESSION_ID": "sess-1"} + + def dead_pid(self): + p = subprocess.Popen(["true"]) + p.wait() + return p.pid + + def test_next_registers_session(self): + env = self.session_env(self.root / "me.sock") + self.ok("next", "--as", "opus", "--brief", env=env) + reg = json.loads((self.root / ".wf" / "sessions" / "all.json").read_text()) + self.assertEqual({k: reg[k] for k in ("lane", "model", "socket", "pid", "session")}, + {"lane": "all", "model": "opus", "socket": str(self.root / "me.sock"), "pid": os.getpid(), "session": "sess-1"}) + self.assertEqual((self.root / ".wf" / ".gitignore").read_text(), "*\n") + + def test_no_register_without_as_or_env(self): + none = {"CLAUDE_CODE_MESSAGING_SOCKET": "", "CLAUDE_PID": "", "CLAUDE_CODE_SESSION_ID": ""} + self.ok("next", "--as", "opus", env=none) + self.ok("next", env=self.session_env(self.root / "me.sock")) + self.assertFalse((self.root / ".wf").exists()) + + def register(self, lane, model, sock, pid): + d = self.root / ".wf" / "sessions" + d.mkdir(parents=True, exist_ok=True) + (d / f"{lane}.json").write_text(json.dumps({"lane": lane, "model": model, "socket": str(sock), "pid": pid, + "session": "s", "at": "2026-10-04T10:00"})) + + def test_next_shows_lanes_and_waiting(self): + self.ok("set", "t-three", "--after", "t-one") + self.ok("set", "t-two", "--model", "opus") + sock = self.root / "son.sock" + sock.write_text("") + self.register("slow", "sonnet", sock, os.getpid()) + self.register("fast", "haiku", self.root / "gone.sock", self.dead_pid()) + none = {"CLAUDE_CODE_MESSAGING_SOCKET": ""} + code, out, err = self.wf("next", "--lane", "fast", "--as", "opus", env=none) # fast: no fallback lane + self.assertEqual((code, err), (1, "wf: nothing pickable for fast (opus) in Pending\n")) + self.assertIn("===== Lanes =====\n" + "fast (you): 0 pickable · 1 waiting\n" + f"slow: 2 pickable · 0 waiting · session uds:{sock} (alive, sonnet)\n\n" + "===== Waiting on other lanes =====\n" + f'- t-three waits on t-one (slow lane) → message uds:{sock}: "t-one blocks my t-three, please take it"\n', out) + self.assertEqual(self.ok("lanes", env=none), + "fast: 0 pickable · 1 waiting · no session\n" + f"slow: 0 pickable · 0 waiting · 2 not runner-ready (no Done) · session uds:{sock} (alive, sonnet)\n") + + def test_next_single_lane_prints_no_lanes_block(self): + (self.root / "workflow.toml").write_text(self.toml + '[lanes.one]\nefforts = ["<1h", "1h"]\n') + self.assertNotIn("Lanes", self.ok("next", "--as", "opus", env={"CLAUDE_CODE_MESSAGING_SOCKET": ""})) + + def test_done_notifies_other_lanes(self): + self.ok("add", "Four.", "-p", "3", "-e", "<1h", "--model", "haiku", "--after", "t-three") + self.ok("add", "Five.", "-p", "3", "-e", "1h", "--after", "t-three") + self.ok("add", "Six.", "-p", "3", "-e", "1h", "--model", "sonnet", "--after", "t-three") + sock = self.root / "h.sock" + sock.write_text("") + self.register("slow", "opus", sock, os.getpid()) + out = self.ok("done", "t-three", "-m", "ok") + self.assertIn(f"notify slow uds:{sock}: now pickable t-five, t-six\n", out) + self.assertNotIn("t-four", out) # same lane as t-three (fast) + + def test_bad_model_is_usage_error(self): + self.fails("add", "Four.", "-p", "3", "-e", "1h", "--model", "gpt", code=2) + self.fails("set", "t-one", "--model", "gpt", code=2) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_orch.py b/tests/test_orch.py new file mode 100644 index 0000000..83216b2 --- /dev/null +++ b/tests/test_orch.py @@ -0,0 +1,227 @@ +import json +import os +import subprocess +import unittest + +from test_cli import TOML, Cli +from test_claims import git +from test_merge import IDENT + +TASKS = """\ +# Tasks — demo + +## Awaiting your decision + +## Pending + +- **t-one** [P1] (<1h): One. + - Done: one works + - Model: sonnet + +- **t-two** [P2] (<1h): Two. + - Done: two works + +- **t-raw** [P2] (<1h): Raw, no Done line. + +- **t-big** [P2] (1h): Big. + - Done: big works + +## Needs human + +## Deferred +""" + + +class OrchTest(Cli): + tasks_text = TASKS + toml = TOML.replace('verify = ["make test"]\n', "") + + def setUp(self): + super().setUp() + git(self.root, "init", "-q", "-b", "master") + (self.root / ".gitignore").write_text(".worktrees/\n.wf/\nout/\n") + git(self.root, "add", "-A") + git(self.root, "commit", "-qm", "init") + + def orch(self, *args, code=0): + got, out, err = self.wf("orch", *args, env=IDENT) + self.assertEqual(got, code, out + err) + self.assertEqual(err, "") + return out + + def run_wf(self, cwd, *args): + got, out, err = self.wf(*args, project=False, cwd=cwd, env=IDENT) + self.assertEqual(got, 0, out + err) + return out + + def record(self, id): + return json.loads((self.root / ".wf" / "orch" / f"{id}.json").read_text()) + + def worker_done(self, id="t-one", lane="fast", merge=True): + """What a worker does: wf start, code, wf done/finish (finish merges, done alone leaves the branch).""" + wt = self.root / ".worktrees" / lane + self.run_wf(self.root, "start", id, "--worktree", str(wt), "--branch", f"{lane}/{id}") + (wt / "code.txt").write_text("x\n") + if merge: + self.run_wf(wt, "finish", id, "-m", "ok", "--commit", "impl", "code.txt", "--no-push") + else: + git(wt, "add", "code.txt") + git(wt, "commit", "-qm", "impl") + self.run_wf(wt, "done", id, "-m", "ok") + return wt + + def log(self): + return (self.root / "out" / "wf-orch.log").read_text() + + def test_pick_claims_and_prints_prompt(self): + out = self.orch("pick", "fast") + wt = self.root / ".worktrees" / "fast" + self.assertEqual(out, "pick: t-one (lane fast, model sonnet, effort <1h) · claimed (in progress: worker)\n" + "agent: subagent_type wf-worker, model sonnet, no isolation; prompt:\n" + "Task: t-one Lane: fast Model: sonnet\n" + f"Main tree: {self.root} Worktree: {wt} Branch: fast/t-one\n" + "Final message: the 4 report lines only.\n") + self.assertIn("- **t-one** [P1] (<1h) (in progress: worker): One.", self.tasks()) + self.assertEqual(self.record("t-one")["worktree"], str(wt)) + + def test_second_pick_skips_in_progress_and_busy_worktree(self): + self.orch("pick", "fast") + out = self.orch("pick", "fast") + self.assertIn("pick: t-two (lane fast, model opus", out) + self.assertIn(f"Worktree: {self.root / '.worktrees' / 'fast-2'} Branch: fast/t-two\n", out) + + def test_worktree_on_other_branch_not_reused(self): + git(self.root, "worktree", "add", "-q", str(self.root / ".worktrees" / "fast"), "-b", "fast/t-old") + self.assertIn(".worktrees/fast-2 Branch: fast/t-one", self.orch("pick", "fast")) + + def test_clean_detached_worktree_reused(self): + git(self.root, "worktree", "add", "-q", "--detach", str(self.root / ".worktrees" / "fast"), "master") + self.assertIn(".worktrees/fast Branch: fast/t-one", self.orch("pick", "fast")) + + def test_worktree_with_live_session_not_reused(self): + wt = self.root / ".worktrees" / "fast" + git(self.root, "worktree", "add", "-q", "--detach", str(wt), "master") + folder = self.root.parent / "no-claude" / "sessions" + folder.mkdir(parents=True) + (folder / "1.json").write_text(json.dumps({"pid": os.getpid(), "cwd": str(wt / "docs")})) + self.assertIn(".worktrees/fast-2 Branch: fast/t-one", self.orch("pick", "fast")) + + def test_stop_file_spawns_nothing(self): + (self.root / "out").mkdir() + (self.root / "out" / "wf-batch.stop").write_text("") + self.assertEqual(self.orch("pick", "fast"), "stop: out/wf-batch.stop exists: spawn nothing (let running workers finish)\n") + self.assertNotIn("in progress", self.tasks()) + + def test_none_pickable(self): + self.orch("pick", "fast") + self.orch("pick", "fast") + self.assertTrue(self.orch("pick", "fast").startswith("none: lane fast has no runner-ready task")) + + def test_explicit_id_and_recovery_line(self): + out = self.orch("pick", "slow", "--id", "t-big", "--recovery", "crashed") + self.assertIn("Branch: slow/t-big\nRecovery: crashed\nFinal message", out) + + def test_post_done_logs_and_picks_next(self): + self.orch("pick", "fast") + self.worker_done() + out = self.orch("post", "t-one", "fast", "--result", "done", "--commit", "abc1234", "--duration", "75") + self.assertTrue(out.startswith("post: t-one done\n\npick: t-two"), out) + self.assertRegex(self.log(), r"^\S+ fast sonnet t-one done abc1234 1m15s\n$") + self.assertFalse((self.root / ".wf" / "orch" / "t-one.json").exists()) + + def test_post_no_next_claims_nothing(self): + self.orch("pick", "fast") + self.worker_done() + out = self.orch("post", "t-one", "fast", "--result", "done", "--no-next", "--duration", "5") + self.assertTrue(out.startswith("post: t-one done"), out) + self.assertNotIn("pick:", out) + self.assertEqual([f.name for f in (self.root / ".wf" / "orch").glob("*.json")], []) + + def test_post_done_without_record_takes_model_from_history(self): + self.orch("pick", "fast") + self.worker_done() + (self.root / ".wf" / "orch" / "t-one.json").unlink() + self.orch("post", "t-one", "fast", "--result", "done", "--no-pick", "--duration", "5") + self.assertRegex(self.log(), r"^\S+ fast sonnet t-one done ") + + def test_post_done_merges_unmerged_branch(self): + self.orch("pick", "fast") + wt = self.worker_done(merge=False) + out = self.orch("post", "t-one", "fast", "--result", "done", "--no-pick", "--no-push") + self.assertEqual(out, "post: t-one done\n merged worktree HEAD: merged fast/t-one into master\n") + self.assertEqual(subprocess.run(["git", "-C", str(self.root), "show", "master:code.txt"], + capture_output=True, text=True).stdout, "x\n") + self.assertEqual(subprocess.run(["git", "-C", str(wt), "branch", "--show-current"], + capture_output=True, text=True).stdout, "") + + def test_post_done_without_archive_is_red(self): + self.orch("pick", "fast") + out = self.orch("post", "t-one", "fast", "--result", "done") + self.assertEqual(out, "post: t-one post-check-red\n no archive line for t-one\n" + "stop lane fast: post-check-red → tell the owner\n") + self.assertIn(" post-check-red ", self.log()) + + def backdate_pick(self, id="t-one", secs=125): + f = self.root / ".wf" / "orch" / f"{id}.json" + rec = json.loads(f.read_text()) + rec["at"] -= secs + f.write_text(json.dumps(rec) + "\n") + + def test_post_zero_duration_falls_back_to_pick_time(self): + self.orch("pick", "fast") + self.backdate_pick() + self.worker_done() + self.orch("post", "t-one", "fast", "--result", "done", "--no-pick", "--duration", "0") + self.assertRegex(self.log(), r" t-one done \S+ 2m0[5-9]s\n$") + + def test_post_red_keeps_pick_time_for_the_repost(self): + self.orch("pick", "fast") + self.backdate_pick() + self.orch("post", "t-one", "fast", "--result", "done") + self.assertTrue((self.root / ".wf" / "orch" / "t-one.json").exists()) + self.worker_done() + self.orch("post", "t-one", "fast", "--result", "done", "--no-pick") + self.assertRegex(self.log().splitlines()[-1], r" t-one done \S+ 2m0[5-9]s$") + self.assertFalse((self.root / ".wf" / "orch" / "t-one.json").exists()) + + def test_post_handback_commits_leftovers_and_stops(self): + self.orch("pick", "fast") + out = self.orch("post", "t-one", "fast", "--result", "handback unclear spec") + self.assertEqual(out, "post: t-one handback\n committed leftover TASKS.md tasks/archive.md\n" + "stop lane fast: handback → tell the owner\n") + self.assertEqual(subprocess.run(["git", "-C", str(self.root), "log", "-1", "--format=%s"], + capture_output=True, text=True).stdout, "t-one handback (orchestrator)\n") + + def test_post_done_on_slice_job_counts_as_sliced(self): + self.orch("pick", "slow", "--id", "t-big") + self.run_wf(self.root, "add", "--parent", "t-big", "-e", "<1h", "--done", "x", "Slice one") + out = self.orch("post", "t-big", "slow", "--result", "done", "--no-next", "--duration", "5") + self.assertTrue(out.startswith("post: t-big sliced"), out) + self.assertNotIn("post-check-red", self.log()) + self.assertIn(" t-big sliced ", self.log()) + + def test_post_model_raised_continues(self): + self.orch("pick", "fast") + self.wf("set", "t-one", "--model", "opus") + self.wf("status", "t-one", "clear") + out = self.orch("post", "t-one", "fast", "--result", "handback beyond sonnet") + self.assertTrue(out.startswith("post: t-one model-raised\n"), out) + self.assertIn("pick: t-one (lane fast, model opus", out) + + def test_post_no_report_suggests_recovery(self): + self.orch("pick", "fast") + self.assertIn('wf orch pick fast --id t-one --recovery "<why>"', self.orch("post", "t-one", "fast")) + + def test_post_agent_without_transcript_still_logs(self): + self.orch("pick", "fast") + out = self.orch("post", "t-one", "fast", "--result", "wip", "--agent", "abc") + self.assertIn(" cost: not logged (no transcript for agent abc)\n", out) + + def test_refused_in_linked_worktree(self): + git(self.root, "worktree", "add", "-q", "--detach", str(self.root / ".worktrees" / "x"), "master") + got, _, err = self.wf("orch", "pick", "fast", project=False, cwd=self.root / ".worktrees" / "x") + self.assertEqual((got, err), (1, "wf: orch runs in the main tree (the orchestrator's)\n")) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_publish_snapshot.py b/tests/test_publish_snapshot.py new file mode 100644 index 0000000..3243d4b --- /dev/null +++ b/tests/test_publish_snapshot.py @@ -0,0 +1,122 @@ +import os +import subprocess +import sys +import tempfile +import unittest +from pathlib import Path + +SCRIPT = Path(__file__).resolve().parent.parent / "scripts" / "publish_snapshot.py" +ENV = {**os.environ, "GIT_AUTHOR_NAME": "t", "GIT_AUTHOR_EMAIL": "t@t", "GIT_COMMITTER_NAME": "t", + "GIT_COMMITTER_EMAIL": "t@t", "GIT_CONFIG_GLOBAL": "/dev/null"} + + +def git(cwd, *args): + return subprocess.run(["git", *args], cwd=cwd, capture_output=True, text=True, env=ENV, check=True).stdout + + +class PublishSnapshotTest(unittest.TestCase): + def setUp(self): + self.tmp = tempfile.TemporaryDirectory() + t = Path(self.tmp.name) + self.src, self.dest, self.deny = t / "tool", t / "pub", t / "deny.txt" + self.src.mkdir() + git(self.src, "init", "-q", "-b", "master") + self.put({"wf.py": "print('hi')\n", "CHANGES.md": "# Changes\n\n- 2026-01-01 first.\n", + "wflib/a.py": "x = 1\n", "docs/d.md": "doc\n", + "inbox.md": "tracked inbox\n", "out/log": "x\n", "sub/__pycache__/c.pyc": "bin\n"}) + (self.src / "wf.py").chmod(0o755) + self.commit("one") + self.deny.write_text("# private\nSecretProj\n\n") + + def tearDown(self): + self.tmp.cleanup() + + def put(self, files): + for p, text in files.items(): + f = self.src / p + f.parent.mkdir(parents=True, exist_ok=True) + f.write_text(text) + + def commit(self, msg): + git(self.src, "add", "-A") + git(self.src, "commit", "-qm", msg) + + def run_it(self, *extra): + return subprocess.run([sys.executable, str(SCRIPT), str(self.dest), "--src", str(self.src), + "--denylist", str(self.deny), *extra], capture_output=True, text=True, env=ENV) + + def files(self): + return sorted(git(self.dest, "ls-files").split()) + + def test_first_run_one_commit_excludes(self): + r = self.run_it() + self.assertEqual(r.returncode, 0, r.stderr) + self.assertEqual(self.files(), ["CHANGES.md", "docs/d.md", "wf.py", "wflib/a.py"]) + self.assertEqual(git(self.dest, "log", "--format=%s").splitlines(), ["initial public snapshot"]) + self.assertTrue(os.access(self.dest / "wf.py", os.X_OK)) + self.assertEqual(git(self.dest, "remote"), "") + + def test_denylist_hit_content_and_path_writes_nothing(self): + self.put({"docs/d.md": "doc\nsee secretproj here\n", "SecretProj.md": "x\n"}) + self.commit("leak") + r = self.run_it() + self.assertEqual(r.returncode, 1) + self.assertIn("docs/d.md:2: SecretProj", r.stderr) + self.assertIn("SecretProj.md: SecretProj (path)", r.stderr) + self.assertFalse(self.dest.exists()) + + def test_uncommitted_files_not_published(self): + (self.src / "wip.py").write_text("SecretProj\n") + r = self.run_it() + self.assertEqual(r.returncode, 0, r.stderr) + self.assertNotIn("wip.py", self.files()) + + def test_missing_or_empty_denylist_refuses(self): + self.deny.unlink() + r = self.run_it() + self.assertEqual(r.returncode, 1) + self.assertIn("no denylist", r.stderr) + self.deny.write_text("# only comments\n") + self.assertEqual(self.run_it().returncode, 1) + self.assertFalse(self.dest.exists()) + + def test_second_run_commit_message_new_changes_lines(self): + self.run_it() + self.put({"CHANGES.md": "# Changes\n\n- 2026-01-03 third.\n- 2026-01-02 second.\n- 2026-01-01 first.\n", + "wflib/b.py": "y = 2\n"}) + (self.src / "docs/d.md").unlink() + self.commit("two") + r = self.run_it() + self.assertEqual(r.returncode, 0, r.stderr) + self.assertEqual(git(self.dest, "log", "-1", "--format=%B").strip(), + "public snapshot\n\n- 2026-01-03 third.\n- 2026-01-02 second.") + self.assertEqual(self.files(), ["CHANGES.md", "wf.py", "wflib/a.py", "wflib/b.py"]) + self.assertEqual(len(git(self.dest, "log", "--format=%h").split()), 2) + + def test_no_change_no_commit(self): + self.run_it() + r = self.run_it() + self.assertEqual(r.returncode, 0, r.stderr) + self.assertIn("nothing new", r.stdout) + self.assertEqual(len(git(self.dest, "log", "--format=%h").split()), 1) + + def test_ref_option_and_never_touches_remote(self): + self.run_it() + git(self.dest, "remote", "add", "home", "/nonexistent") + self.put({"wflib/a.py": "x = 2\n"}) + self.commit("two") + self.assertEqual(self.run_it("--ref", "HEAD~1").stdout.strip(), "nothing new: no commit") + r = self.run_it() + self.assertIn("committed", r.stdout) + self.assertEqual(git(self.dest, "remote").split(), ["home"]) + + def test_nonempty_non_git_dest_refused(self): + self.dest.mkdir() + (self.dest / "keep.txt").write_text("x") + r = self.run_it() + self.assertEqual(r.returncode, 1) + self.assertIn("not a git repo", r.stderr) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_refs.py b/tests/test_refs.py new file mode 100644 index 0000000..4242a03 --- /dev/null +++ b/tests/test_refs.py @@ -0,0 +1,109 @@ +import sys +import tempfile +import unittest +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) +from wflib import refs as R + +DOC = """\ +# Design + +Intro. + +## Subsystems + +### Character Model + +Bones and poses. + +#### Detail + +Deep. + +<a id="terrain"></a> +### Terrain & caves! + +Heightmap. + +## Other + +<a id="loose"></a> +Loose paragraph anchor. +""" + + +class LinksTest(unittest.TestCase): + def test_links(self): + self.assertEqual(R.links("see [[t-a]] and [[a-b|label]] and [[doc#part]]"), ["t-a", "a-b", "doc"]) + + def test_code_is_skipped(self): + self.assertEqual(R.links("`[[t-x]]` then [[t-y]]\n```\n[[t-z]]\n```\n"), ["t-y"]) + + +class SectionsTest(unittest.TestCase): + def test_slugify(self): + self.assertEqual(R.slugify("Character Model"), "character-model") + self.assertEqual(R.slugify("Terrain & caves!"), "terrain-caves") + + def test_sections(self): + got = [(s.level, s.heading, sorted(s.anchors), s.start, s.end) for s in R.sections(DOC)] + self.assertEqual(got, [ + (1, "Design", ["design"], 0, 23), + (2, "Subsystems", ["subsystems"], 4, 19), + (3, "Character Model", ["character-model"], 6, 14), + (4, "Detail", ["detail"], 10, 14), + (3, "Terrain & caves!", ["terrain", "terrain-caves"], 15, 19), + (2, "Other", ["other"], 19, 23), + (0, "", ["loose"], 21, 23), + ]) + + def test_anchors(self): + self.assertEqual(R.anchors(DOC), {"design", "subsystems", "character-model", "detail", + "terrain", "terrain-caves", "other", "loose"}) + + def test_headings_in_code_fences_are_not_sections(self): + self.assertEqual(R.anchors("# A\n```\n# not a heading\n```\n"), {"a"}) + + +class ResolveTest(unittest.TestCase): + def setUp(self): + self.tmp = tempfile.TemporaryDirectory() + self.root = Path(self.tmp.name) + (self.root / "docs").mkdir() + (self.root / "DESIGN.md").write_text(DOC) + long = "# Long\n\n## Part\n" + "".join(f"line {n}\n" for n in range(1, 100)) + (self.root / "docs" / "x.md").write_text(long) + (self.root / "docs" / "plan.md").write_text("# The Plan\n\n**Goal:** Ship it.\n\nMore.\n") + + def tearDown(self): + self.tmp.cleanup() + + def test_anchor_section(self): + self.assertEqual(R.resolve(self.root, "DESIGN.md", "terrain"), + "===== DESIGN.md#terrain (line 16) =====\n### Terrain & caves!\n\nHeightmap.") + + def test_long_section_is_cut(self): + out = R.resolve(self.root, "docs/x.md", "part").split("\n") + self.assertEqual(out[0], "===== docs/x.md#part (line 3) =====") + self.assertEqual(out[1], "## Part") + self.assertEqual(out[80], "line 79") + self.assertEqual(out[81], "… (20 more lines: docs/x.md:83)") + self.assertEqual(len(out), 82) + + def test_missing_anchor(self): + self.assertEqual(R.resolve(self.root, "DESIGN.md", "nope"), "(no anchor 'nope' in DESIGN.md)") + + def test_missing_file(self): + self.assertEqual(R.resolve(self.root, "docs/none.md", None), "(missing: docs/none.md)") + + def test_path_only_gives_heading_and_goal(self): + self.assertEqual(R.resolve(self.root, "docs/plan.md", None), "docs/plan.md: The Plan — Ship it.") + self.assertEqual(R.resolve(self.root, "DESIGN.md", None), "DESIGN.md: Design") + + def test_directory(self): + self.assertEqual(R.resolve(self.root, "docs", None), "docs/ (directory)") + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_res.py b/tests/test_res.py new file mode 100644 index 0000000..6c92549 --- /dev/null +++ b/tests/test_res.py @@ -0,0 +1,751 @@ +import datetime as dt +import json +import sys +import unittest +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) +from wflib import res as R + +UTC = dt.timezone.utc + + +def T(h, m=0, day=1): + return dt.datetime(2026, 10, day, h, m, tzinfo=UTC) + + +NOW = T(14, 10) +GIB = 1024 ** 3 + + +def running(id="r-4", project="proj-a", title="dotnet e2e", mem=10.0, cpus=2, est=40, started=T(14, 0)): + return R.Entry(id=id, project=project, owner=1, title=title, mem_gb=mem, cpus=cpus, est_min=est, + state="running", unit=f"wf-{id}.service", started=started, + expires=started + dt.timedelta(minutes=2 * est)) + + +def note(id="r-8", owner=77, mem=3.0, cpus=1, expires=T(14, 30)): + return R.Entry(id=id, project="proj-b", owner=owner, title="dotnet test", mem_gb=mem, cpus=cpus, + est_min=20, state="note", started=T(14, 0), expires=expires) + + +def facts(available=20.0, units=None, live=(), results=None, slice_gb=0.0, now=NOW): + return R.Facts(now=now, total_gb=30.0, available_gb=available, nproc=16, units=units or {}, + live=set(live), results=results or {}, slice_gb=slice_gb) + + +class Parse(unittest.TestCase): + def test_size(self): + self.assertEqual(R.parse_size("10G"), 10.0) + self.assertEqual(R.parse_size("512M"), 0.5) + self.assertEqual(R.parse_size("1.5g"), 1.5) + with self.assertRaisesRegex(R.ResError, "size '10'"): + R.parse_size("10") + + def test_duration(self): + self.assertEqual(R.parse_duration("40m"), 40) + self.assertEqual(R.parse_duration("4h"), 240) + self.assertEqual(R.parse_duration("1h30m"), 90) + self.assertEqual(R.parse_duration("90"), 90) + for bad in ("0m", "x", "", "h"): + with self.assertRaises(R.ResError): + R.parse_duration(bad) + + def test_meminfo(self): + text = "MemTotal: 31457280 kB\nMemFree: 1 kB\nMemAvailable: 12582912 kB\n" + self.assertEqual(R.meminfo(text), (30.0, 12.0)) + with self.assertRaises(R.ResError): + R.meminfo("MemFree: 1 kB\n") + + def test_show_units(self): + text = ("Id=agents.slice\nActiveState=active\nMemoryCurrent=0\n\n" + "Id=wf-r-1.service\nActiveState=inactive\nMemoryCurrent=[not set]\n") + self.assertEqual(R.show_units(text), { + "agents.slice": {"Id": "agents.slice", "ActiveState": "active", "MemoryCurrent": "0"}, + "wf-r-1.service": {"Id": "wf-r-1.service", "ActiveState": "inactive", "MemoryCurrent": "[not set]"}}) + + def test_gb_or_none(self): + self.assertEqual(R.gb_or_none("1073741824"), 1.0) + self.assertIsNone(R.gb_or_none("[not set]")) + self.assertIsNone(R.gb_or_none("infinity")) + + def test_formats(self): + self.assertEqual(R.fmt_gb(2.25), "2.2 GB") + self.assertEqual(R.fmt_bytes(1288490188), "1.2 GB") + self.assertEqual(R.fmt_bytes(300 * 1024 ** 2), "300 MB") + self.assertEqual(R.fmt_dur(240), "4h") + self.assertEqual(R.fmt_dur(90), "1h30m") + self.assertEqual(R.fmt_dur(40), "40m") + self.assertEqual(R.hhmm(T(9, 5)), "09:05") + + +class ConfigTest(unittest.TestCase): + def test_defaults_and_override(self): + cfg = R.load_config("user_reserve_gb = 8\n") + self.assertEqual((cfg.user_reserve_gb, cfg.user_reserve_cpus, cfg.game_reserve_gb, cfg.game_reserve_cpus, + cfg.game_hours, cfg.small_headroom_gb, cfg.scratch_hours), (8.0, 4, 12.0, 8, 4.0, 2.0, 2.0)) + + def test_unknown_key(self): + self.assertEqual(R.load_config("session_mem_gb = 12").session_mem_gb, 12.0) + self.assertEqual(R.Config().session_mem_gb, 6.0) + with self.assertRaisesRegex(R.ResError, "unknown key\\(s\\) foo"): + R.load_config("foo = 1\n") + + def test_negative(self): + with self.assertRaisesRegex(R.ResError, "game_hours must be a number ≥ 0"): + R.load_config("game_hours = -1\n") + + +class LedgerJson(unittest.TestCase): + def test_roundtrip(self): + led = R.Ledger(next=9, game_until=T(18), entries=[running(), note()]) + text = R.dumps(led) + self.assertIn('"started": "2026-10-01T14:00:00+00:00"', text) + self.assertEqual(R.loads(text), led) + + def test_empty(self): + self.assertEqual(R.loads(""), R.Ledger()) + + def test_corrupt(self): + for bad in ("{", '{"entries": [{"id": "r-1"}]}', "[]"): + with self.assertRaisesRegex(R.ResError, "corrupt ledger"): + R.loads(bad) + + def test_unknown_entry_keys_ignored(self): + d = json.loads(R.dumps(R.Ledger(next=9, entries=[running()]))) + d["entries"][0]["from_a_newer_version"] = 1 + self.assertEqual(R.loads(json.dumps(d)), R.Ledger(next=9, entries=[running()])) + + def test_unknown_keys_survive_rewrite(self): + d = json.loads(R.dumps(R.Ledger(next=9, entries=[running()]))) + d["entries"][0]["field_x"] = {"a": 1} + d["top_x"] = 5 + out = json.loads(R.dumps(R.loads(json.dumps(d)))) + self.assertEqual(out["entries"][0]["field_x"], {"a": 1}) + self.assertEqual(out["top_x"], 5) + + def test_new_id_and_get(self): + led = R.Ledger(next=3) + self.assertEqual(led.new_id(), "r-3") + self.assertEqual(led.next, 4) + with self.assertRaisesRegex(R.ResError, "no entry 'r-9'"): + led.get("r-9") + + +class Prune(unittest.TestCase): + def test_prune(self): + r1, r2 = running("r-1"), running("r-2") + n3, n5 = note("r-3", owner=77), note("r-5", owner=78, expires=T(14, 5)) + d6 = R.Entry(id="r-6", project="p", owner=1, title="old", mem_gb=1, cpus=1, est_min=1, state="done", + ended=NOW - dt.timedelta(hours=25)) + d7 = R.Entry(id="r-7", project="p", owner=1, title="new", mem_gb=1, cpus=1, est_min=1, state="done", + ended=NOW - dt.timedelta(hours=23)) + led = R.Ledger(game_until=T(14, 9), entries=[r1, r2, n3, n5, d6, d7]) + f = facts(units={"wf-r-1.service": R.Unit(False)}, live={78}, results={"r-1": (0, 9.4, "")}) + lines = R.prune(led, f) + self.assertEqual(lines, ["r-1 exited rc=0", "r-2 exited rc=?", "r-3 freed (owner gone)", + "r-5 freed (expired)", "game off (expired)"]) + self.assertEqual([e.id for e in led.entries], ["r-1", "r-2", "r-3", "r-5", "r-7"]) + self.assertEqual((r1.state, r1.rc, r1.peak_gb, r1.why, r1.ended), ("done", 0, 9.4, "exited", NOW)) + self.assertEqual((n3.why, n5.why), ("owner gone", "expired")) + self.assertIsNone(led.game_until) + + def test_prune_missing_results(self): + r = running("r-1") + R.prune(R.Ledger(entries=[r]), facts(units={"wf-r-1.service": R.Unit(False)})) + self.assertEqual((r.state, r.rc, r.peak_gb), ("done", None, None)) + + def test_overdue_running_is_kept(self): + r = running(started=T(10)) + R.prune(R.Ledger(entries=[r]), facts(units={"wf-r-4.service": R.Unit(True, 4.0)})) + self.assertEqual(r.state, "running") + + +class Capacity(unittest.TestCase): + def setUp(self): + self.cfg = R.Config() + + def test_budget(self): + led = R.Ledger(entries=[running(), note()]) + f = facts(units={"wf-r-4.service": R.Unit(True, 4.0)}, live={77}) + # 20 − 6 reserve − 2 headroom − (10−4) unused − 3 note = 3; 16 − 4 − (2+1) = 9 + self.assertEqual(R.budget(self.cfg, led, f), (3.0, 9)) + + def test_budget_gaming(self): + led = R.Ledger(game_until=T(15), entries=[running(), note()]) + f = facts(units={"wf-r-4.service": R.Unit(True, 4.0)}, live={77}) + # 20 − 12 − 2 − 6 − 3 = −3; 16 − 8 − 3 = 5 + self.assertEqual(R.budget(self.cfg, led, f), (-3.0, 5)) + + def test_held_over_reservation_is_zero(self): + self.assertEqual(R.held_gb(running(), facts(units={"wf-r-4.service": R.Unit(True, 12.0)})), 0.0) + self.assertEqual(R.frees_gb(running(), facts(units={"wf-r-4.service": R.Unit(True, 12.0)})), 12.0) + + def test_fits(self): + self.assertTrue(R.fits(3.0, 9, (3.0, 9))) + self.assertFalse(R.fits(3.1, 1, (3.0, 9))) + self.assertFalse(R.fits(1.0, 10, (3.0, 9))) + + def test_never_fits(self): + f = facts() + self.assertEqual(R.never_fits(self.cfg, f, 25.0, 1), "25.0 GB can never fit (max 22.0 GB for agents)") + self.assertEqual(R.never_fits(self.cfg, f, 1.0, 13), "13 cpus can never fit (max 12 for agents)") + self.assertIsNone(R.never_fits(self.cfg, f, 22.0, 12)) + + def test_queued_total(self): + q = running("r-9") + q.state = "queued" + self.assertEqual(R.queued_total(R.Ledger(entries=[q, running()])), (10.0, 2)) + + +class BusyLine(unittest.TestCase): + def setUp(self): + self.cfg = R.Config() + self.led = R.Ledger(entries=[running()]) + self.f = facts(units={"wf-r-4.service": R.Unit(True, 4.0)}) # budget 6.0 GB, 10 cpus + + def test_memory(self): + self.assertEqual(R.busy_line(self.cfg, self.led, self.f, 8.0, 1), + 'busy: 10.0 GB held by proj-a "dotnet e2e" (r-4) until ~14:40; 6.0 GB free for agents; ' + 'retry after ~14:40 or work on something else') + + def test_cpus(self): + self.assertEqual(R.busy_line(self.cfg, self.led, self.f, 1.0, 12), + 'busy: 2 cpus held by proj-a "dotnet e2e" (r-4) until ~14:40; 10 cpus free for agents; ' + 'retry after ~14:40 or work on something else') + + def test_no_holder(self): + self.assertEqual(R.busy_line(self.cfg, R.Ledger(), facts(available=7.0), 1.0, 1), + "busy: only 0.0 GB free for agents and no agent job holds any; other programs use the rest; " + "retry later or work on something else") + + def test_busy_overdue_holder(self): + led = R.Ledger(entries=[running(started=T(10))]) + self.assertEqual(R.busy_line(self.cfg, led, self.f, 8.0, 1), + 'busy: 10.0 GB held by proj-a "dotnet e2e" (r-4) until overdue; 6.0 GB free for agents; ' + 'retry later or work on something else') + + def test_two_holders_earliest_first(self): + a = running("r-1", project="a", title="A", mem=4.0, cpus=1, est=60) # ends 15:00 + b = running("r-2", project="b", title="B", mem=4.0, cpus=1, est=30) # ends 14:30 + f = facts(available=14.0, units={}) # 14 − 6 − 2 − 4 − 4 = −2.0 budget + self.assertEqual(R.busy_line(self.cfg, R.Ledger(entries=[a, b]), f, 5.0, 1), + 'busy: 4.0 GB held by b "B" (r-2) until ~14:30, 4.0 GB held by a "A" (r-1) until ~15:00; ' + '0.0 GB free for agents; retry after ~15:00 or work on something else') + + +class Lines(unittest.TestCase): + def test_run_argv(self): + e = running("r-3", mem=10.0) + e.cwd, e.log, e.cmd = "/projects/x", "/s/logs/r-3.log", ["make", "it big", "--fast"] + self.assertEqual(R.run_argv(e, "/s/logs"), [ + "systemd-run", "--user", "--quiet", "--collect", "--slice=agents-jobs.slice", "--unit=wf-r-3.service", + "--working-directory=/projects/x", "--setenv=WF_RES_ID=r-3", + "-p", "MemoryMax=10737418240", "-p", "MemoryHigh=9663676416", "-p", "MemorySwapMax=0", "-p", "Nice=10", + "-p", "StandardOutput=append:/s/logs/r-3.log", "-p", "StandardError=append:/s/logs/r-3.log", + "/bin/sh", "-c", + '"$@"; rc=$?; cat /sys/fs/cgroup$(cut -d: -f3 /proc/self/cgroup)/memory.peak > /s/logs/r-3.peak ' + '2>/dev/null; echo $rc > /s/logs/r-3.rc', + "sh", "make", "it big", "--fast"]) + + def test_run_argv_env(self): + e = running("r-3", mem=1.0) + e.cwd, e.log, e.cmd, e.env = "/p", "/s/r-3.log", ["x"], {"B": "2", "A": "1 2"} + argv = R.run_argv(e, "/s") + self.assertEqual(argv[argv.index("--working-directory=/p") + 1:][:2], ["--setenv=A=1 2", "--setenv=B=2"]) + + def test_env_diff(self): + self.assertEqual(R.env_diff({"A": "1", "B": "2", "C": "3", "PWD": "/", "OLDPWD": "/", "SHLVL": "1", "_": "/x", + "MY_TOKEN": "t", "API_SECRET": "s", "PGPASSWORD": "p", "SSH_AUTH_SOCK": "/s"}, + {"A": "1", "B": "9", "SSH_AUTH_SOCK": "/s"}), + {"B": "2", "C": "3"}) + + def test_manager_env_parse(self): + self.assertEqual(R.parse_show_environment("A=1\nB=x=y\n\nbad\n"), {"A": "1", "B": "x=y"}) + + def test_entry_lines(self): + f = facts(units={"wf-r-4.service": R.Unit(True, 4.0)}, live={77}) + q = running("r-9", project="p", title="Q", mem=1.0, cpus=1) + q.state = "queued" + d = R.Entry(id="r-2", project="p", owner=1, title="D", mem_gb=1, cpus=1, est_min=1, state="done", + ended=T(13, 50), rc=None, peak_gb=None, why="exited") + led = R.Ledger(entries=[running(), note(), q, d]) + self.assertEqual([R.entry_line(e, led, f) for e in led.entries], [ + 'r-4 proj-a "dotnet e2e" running 10.0 GB used 4.0 GB 2 cpu since 14:00 ETA ~14:40', + 'r-8 proj-b "dotnet test" note 3.0 GB 1 cpu until ~14:30', + 'r-9 p "Q" queued #1 1.0 GB 1 cpu', + 'r-2 p "D" done (exited) rc=? peak ? at 13:50']) + + def test_status_lines(self): + f = facts(units={"wf-r-4.service": R.Unit(True, 4.0)}, slice_gb=5.5) + self.assertEqual(R.status_lines(R.Config(), R.Ledger(entries=[running()]), f), [ + 'r-4 proj-a "dotnet e2e" running 10.0 GB used 4.0 GB 2 cpu since 14:00 ETA ~14:40', + "agents may use 6.0 GB, 10 cpus now; reserve 6.0 GB/4 cpus", + "unreserved agent memory 1.5 GB"]) + f.slice_cache_gb = 0.6 + self.assertEqual(R.status_lines(R.Config(), R.Ledger(entries=[running()]), f)[-1], + "unreserved agent memory 1.5 GB (0.6 GB of it file cache, reclaimable)") + f.slice_cache_gb = 3.0 # cache partly inside the jobs' 4.0 GB → capped at the unreserved figure + self.assertEqual(R.status_lines(R.Config(), R.Ledger(entries=[running()]), f)[-1], + "unreserved agent memory 1.5 GB (1.5 GB of it file cache, reclaimable)") + + def test_cache_gb(self): + # agents.slice memory.stat excerpt from this machine: file 5602209792, shmem (tmpfs, not reclaimable) 3861041152 + self.assertEqual(round(R.cache_gb("anon 11335516160\nfile 5602209792\nkernel 1\nshmem 3861041152\n"), 3), 1.622) + self.assertEqual(R.cache_gb(""), 0.0) + + def test_status_lines_empty_overdue(self): + self.assertEqual(R.status_lines(R.Config(), R.Ledger(), facts(available=7.0)), [ + "no reservations", "agents may use 0.0 GB, 12 cpus now; reserve 6.0 GB/4 cpus"]) + r = running(started=T(10)) + self.assertIn("ETA overdue (still running, not killed)", R.entry_line(r, R.Ledger(entries=[r]), facts())) + + def test_status_json(self): + import json + d = json.loads(R.status_json(R.Config(), R.Ledger(entries=[running()]), + facts(units={"wf-r-4.service": R.Unit(True, 4.0)}))) + self.assertEqual((d["budget_gb"], d["budget_cpus"], d["gaming"], d["entries"][0]["id"]), + (6.0, 10, False, "r-4")) + + def test_done_line(self): + e = running("r-3", started=T(14, 0)) + e.state, e.ended, e.rc, e.peak_gb = "done", T(14, 37), 0, 9.4 + self.assertEqual(R.done_line(e), "r-3 done rc=0 peak 9.4 GB in 37 min") + e.rc, e.peak_gb = None, None + self.assertEqual(R.done_line(e), "r-3 done rc=? peak ? in 37 min") + e.why = "killed: timeout" + self.assertEqual(R.done_line(e), "r-3 done rc=? peak ? in 37 min (killed: timeout)") + + def test_prune_killed(self): + r = running("r-1") + f = facts(units={"wf-r-1.service": R.Unit(False)}, results={"r-1": (None, 3.9, "killed: timeout")}) + self.assertEqual(R.prune(R.Ledger(entries=[r]), f), ["r-1 killed: timeout"]) + self.assertEqual((r.state, r.rc, r.peak_gb, r.why), ("done", None, 3.9, "killed: timeout")) + + +# journalctl --user -u wf-r-12.service -o cat, verbatim from a real systemd-oomd kill (systemd 258) +OOMD = """Started wf-r-12.service - [systemd-run] /bin/sh -c "x" sh dotnet test. +wf-r-12.service: systemd-oomd killed some process(es) in this unit. +wf-r-12.service: Main process exited, code=killed, status=9/KILL +wf-r-12.service: Failed with result 'oom-kill'. +wf-r-12.service: Consumed 11min 2.837s CPU time, 3.9G memory peak. +""" + + +class JournalReason(unittest.TestCase): + def test_oomd(self): + self.assertEqual(R.journal_reason(OOMD, 4.0), + ("killed: oom-kill by systemd-oomd, limit 4.0 GB; raise --mem", 3.9)) + + def test_kernel_oom(self): + text = ("u: A process of this unit has been killed by the OOM killer.\n" + "u: Main process exited, code=killed, status=9/KILL\n" + "u: Failed with result 'oom-kill'.\nu: Consumed 2s CPU time, 512M memory peak.\n") + self.assertEqual(R.journal_reason(text, 0.5), + ("killed: oom-kill at MemoryMax, limit 0.5 GB; raise --mem", 0.5)) + + def test_signal(self): + text = "u: Main process exited, code=killed, status=15/TERM\nu: Failed with result 'signal'.\n" + self.assertEqual(R.journal_reason(text, 1.0), ("killed: signal 15/TERM", None)) + + def test_timeout_and_exit(self): + self.assertEqual(R.journal_reason("u: Failed with result 'timeout'.\n", 1.0), ("killed: timeout", None)) + text = "u: Main process exited, code=exited, status=3/NOTIMPLEMENTED\nu: Failed with result 'exit-code'.\n" + self.assertEqual(R.journal_reason(text, 1.0), ("failed: exit 3/NOTIMPLEMENTED", None)) + + def test_nothing(self): + self.assertEqual(R.journal_reason("", 1.0), ("", None)) + self.assertEqual(R.journal_reason("u: Deactivated successfully.\nu: Consumed 1s CPU time, 1.5G memory peak.\n", 1.0), + ("", 1.5)) + +def queued(id, mem, cpus=1, at=T(13, 0)): + e = R.Entry(id=id, project="p", owner=1, title=id, mem_gb=mem, cpus=cpus, est_min=10, state="queued", + cmd=["x"], queued=at) + return e + + +class Queue(unittest.TestCase): + def test_fifo_blocked_head(self): + # budget 20 − 6 − 2 = 12 GB + led = R.Ledger(entries=[queued("r-1", 5, at=T(13, 0)), queued("r-2", 9, at=T(13, 1)), + queued("r-3", 1, at=T(13, 2))]) + self.assertEqual([e.id for e in R.to_start(R.Config(), led, facts())], ["r-1"]) + + def test_all_fit(self): + led = R.Ledger(entries=[queued("r-2", 4, at=T(13, 1)), queued("r-1", 4, at=T(13, 0))]) + self.assertEqual([e.id for e in R.to_start(R.Config(), led, facts())], ["r-1", "r-2"]) + + def test_estimate(self): + # running r-4 holds 10 (used 4), ends 14:40; budget 6; queue r-1 (5) ahead of r-2 (4): need 9 − 6 = 3 + led = R.Ledger(entries=[running(), queued("r-1", 5), queued("r-2", 4, at=T(13, 5))]) + f = facts(units={"wf-r-4.service": R.Unit(True, 4.0)}) + self.assertEqual(R.queue_estimate(R.Config(), led, f, led.get("r-2")), T(14, 40)) + + def test_estimate_unknown(self): + led = R.Ledger(entries=[queued("r-1", 20)]) + self.assertIsNone(R.queue_estimate(R.Config(), led, facts(), led.get("r-1"))) + +class Game(unittest.TestCase): + def setUp(self): + self.cfg = R.Config() + + def test_slice_props(self): + self.assertEqual(R.slice_props(self.cfg, R.Ledger(), facts()), + {"CPUWeight": "20", "IOWeight": "20", "MemoryHigh": str(24 * GIB)}) + on = R.Ledger(game_until=T(18)) + self.assertEqual(R.slice_props(self.cfg, on, facts(slice_gb=10.0))["MemoryHigh"], str(18 * GIB)) + self.assertEqual(R.slice_props(self.cfg, on, facts(slice_gb=20.0)), + {"CPUWeight": "5", "IOWeight": "5", "MemoryHigh": str(20 * GIB)}) + + def test_shortfall(self): + a = running("r-4", project="proj-a", title="dotnet e2e", mem=2.5, est=55, started=T(14, 10)) # ~15:05 + b = running("r-7", project="proj-b", title="r5 rebuild", mem=9.0, est=80, started=T(14, 10)) # ~15:30 + f = facts(available=11.5, units={"wf-r-4.service": R.Unit(True, 2.5), "wf-r-7.service": R.Unit(True, 9.0)}) + led = R.Ledger(game_until=T(18), entries=[a, b]) + # free for you 11.5 − 0 unused = 11.5 → short 0.5: r-4 alone frees 2.5 + self.assertEqual(R.shortfall_lines(self.cfg, led, f), [ + 'short 0.5 GB of 12.0 GB: r-4 proj-a "dotnet e2e" 2.5 GB ~15:05', + "full reserve free ~15:05 (est.); free now: wf res release r-4 --stop"]) + f.available_gb = 8.9 # short 3.1: needs both + self.assertEqual(R.shortfall_lines(self.cfg, led, f), [ + 'short 3.1 GB of 12.0 GB: r-4 proj-a "dotnet e2e" 2.5 GB ~15:05, r-7 proj-b "r5 rebuild" 9.0 GB ~15:30', + "full reserve free ~15:30 (est.); free now: wf res release r-7 --stop"]) + f.available_gb = 13.0 + self.assertEqual(R.shortfall_lines(self.cfg, led, f), []) + + def test_shortfall_not_agents(self): + self.assertEqual(R.shortfall_lines(self.cfg, R.Ledger(game_until=T(18)), facts(available=5.0)), [ + "short 7.0 GB of 12.0 GB: other programs, not agent jobs", + "agent jobs alone cannot free it; close other programs"]) + + def test_game_on_lines(self): + led = R.Ledger(game_until=T(18, 10)) + self.assertEqual(R.game_on_lines(self.cfg, led, facts(available=20.0), 240), + ["game on until 18:10 (4h); CPU/IO now yours"]) + + def test_status_shows_game(self): + lines = R.status_lines(self.cfg, R.Ledger(game_until=T(18, 10)), facts(available=5.0)) + self.assertEqual(lines[1:], ["agents may use 0.0 GB, 8 cpus now; reserve 12.0 GB/8 cpus", + "game on until 18:10 (4h left)", + "short 7.0 GB of 12.0 GB: other programs, not agent jobs", + "agent jobs alone cannot free it; close other programs"]) + +class Clean(unittest.TestCase): + def test_scratch_victims(self): + now = 10_000_000.0 + h = 3600 + tree = { + "/s/old-file.png": (now - 3 * h, {}), + "/s/-projects-a": (now - 3 * h, {"/s/-projects-a/s1": now - 3 * h}), + "/s/-projects-b": (now - 60, {"/s/-projects-b/s1": now - 5 * h, "/s/-projects-b/s2": now - 60}), + "/s/new.txt": (now - 60, {}), + } + self.assertEqual(R.scratch_victims(tree, now, 2.0), + ["/s/-projects-a", "/s/-projects-b/s1", "/s/old-file.png"]) + # live sessions keep their dir however idle; a stale top holding one is pruned per child + self.assertEqual(R.scratch_victims(tree, now, 2.0, keep={"s1"}), ["/s/old-file.png"]) + tree["/s/-projects-a"] = (now - 3 * h, {"/s/-projects-a/s1": now - 3 * h, "/s/-projects-a/s9": now - 3 * h}) + self.assertEqual(R.scratch_victims(tree, now, 2.0, keep={"s1"}), ["/s/-projects-a/s9", "/s/old-file.png"]) + + def test_live_session_id(self): + text = '{"pid": 817151, "sessionId": "4ad140ab-69ca", "procStart": "31379851", "kind": "interactive"}' + self.assertEqual(R.live_session_id(text, "31379851"), "4ad140ab-69ca") + self.assertIsNone(R.live_session_id(text, "999")) # pid reused by another process + self.assertEqual(R.live_session_id('{"sessionId": "x"}', "5"), "x") # no procStart recorded → trust pid + self.assertIsNone(R.live_session_id("{oops", "5")) + self.assertIsNone(R.live_session_id('{"pid": 1}', "5")) + + def test_cleanup_rule(self): + self.assertEqual(R.cleanup_rule("out/prof"), ("out/prof", None)) + self.assertEqual(R.cleanup_rule("out/history-logs/*.log:30d"), ("out/history-logs/*.log", 30)) + for bad in ("/etc", "../x", "out/../../x"): + with self.assertRaisesRegex(R.ResError, "cleanup pattern"): + R.cleanup_rule(bad) + + def test_proc_name(self): + # background sessions run the versioned binary: comm is the version, argv0 the path or `claude` + self.assertEqual(R.proc_name("2.1.283", "claude"), "claude") + self.assertEqual(R.proc_name("2.1.283", "claude bg-pty-host --bg-pty-host /tmp/x.sock"), "claude") # rewritten title + self.assertEqual(R.proc_name("2.1.283", "/h/u/.local/share/claude/versions/2.1.283"), "claude") + self.assertEqual(R.proc_name("claude", "/h/u/.local/bin/claude"), "claude") + self.assertEqual(R.proc_name("2.1.284", "ugrep"), "2.1.284") # tool re-exec of the binary: not a session + self.assertEqual(R.proc_name("bash", "/bin/bash"), "bash") + self.assertEqual(R.proc_name("x", ""), "x") # kernel thread / unreadable cmdline + + def test_adopt_groups(self): + procs = [ + (100, 1, "bash", "/u/app.slice/tab1.scope"), + (101, 100, "claude", "/u/app.slice/tab1.scope"), # root, outside + (102, 101, "bash", "/u/app.slice/tab1.scope"), # its shell + (103, 102, "make", "/u/app.slice/tab1.scope"), + (104, 101, "claude", "/u/app.slice/tab1.scope"), # child claude: not a root, but a descendant + (105, 101, "job", "/u/agents.slice/agents-jobs.slice/wf-r-1.service"), # already inside: skipped + (200, 1, "claude", "/u/agents.slice/run-9.scope"), # root already inside + (300, 1, "vim", "/u/app.slice/tab2.scope"), + ] + self.assertEqual(R.adopt_groups(procs), {101: [101, 102, 103, 104]}) + + def test_warning_lines(self): + self.assertEqual(R.warning_lines(30.0, 7.6, 2), [ + "warning: /tmp (RAM) holds 7.6 GB; wf res clean", + "warning: 2 claude sessions outside agents.slice (wf res adopt)"]) + self.assertEqual(R.warning_lines(30.0, 7.4, 0), []) + +class Units(unittest.TestCase): + def test_adopt_argv(self): + self.assertEqual(R.adopt_argv(101, [101, 102]), [ + "busctl", "--user", "call", "org.freedesktop.systemd1", "/org/freedesktop/systemd1", + "org.freedesktop.systemd1.Manager", "StartTransientUnit", "ssa(sv)a(sa(sv))", + "wf-claude-101.scope", "fail", "2", "PIDs", "au", "2", "101", "102", "Slice", "s", "agents.slice", "0"]) + + def test_texts(self): + self.assertEqual(R.service_text("/usr/bin/python3", "/projects/public/workflow/wf.py"), + "[Unit]\nDescription=wf res tick\n\n[Service]\nType=oneshot\n" + "ExecStart=/usr/bin/python3 /projects/public/workflow/wf.py res tick\n") + self.assertEqual(R.timer_text(), + "[Unit]\nDescription=wf res tick every minute\n\n[Timer]\nOnBootSec=1min\n" + "OnUnitActiveSec=1min\n\n[Install]\nWantedBy=timers.target\n") + self.assertEqual(R.shell_init_line(), + "alias claude='systemd-run --user --scope --quiet --slice=agents.slice claude'") + + +if __name__ == "__main__": + unittest.main() + + +class Throttle(unittest.TestCase): + def test_psi(self): + text = "some avg10=41.50 avg60=33.20 avg300=12.00 total=99\nfull avg10=30.00 avg60=25.00 avg300=9.00 total=88\n" + self.assertEqual(R.psi_some_avg60(text), 33.2) + self.assertIsNone(R.psi_some_avg60("")) + + def test_lines(self): + hot = running("r-1", mem=4.0) # 3.5 of 4.0 GB (≥ 0.85×), stall 33% → warn + cache = running("r-2", mem=4.0) # full of page cache, no stall → quiet + small = running("r-3", mem=10.0) # stalled but far below its limit (machine pressure) → quiet + f = facts(units={"wf-r-1.service": R.Unit(True, 3.5, 33.2), "wf-r-2.service": R.Unit(True, 3.9, 0.0), + "wf-r-3.service": R.Unit(True, 2.0, 50.0)}) + self.assertEqual(R.throttle_lines(R.Ledger(entries=[hot, cache, small]), f), [ + "r-1 throttled at its memory limit (3.5 of 4.0 GB, stalled 33% of the last minute): " + "likely too small; `wf res release r-1 --stop` and re-run with a bigger --mem"]) + + +class Owner(unittest.TestCase): + REC = [{"lane": "fast", "pid": 11, "socket": "/s/a"}, {"lane": "slow", "pid": 22, "socket": "/s/b"}] + + def test_from_env_branch_and_session_record(self): + env = {"CLAUDE_PID": "22", "CLAUDE_CODE_MESSAGING_SOCKET": "/s/b", "WF_RES_ID": "r-7"} + self.assertEqual(R.owner_by(env, "slow/t-res-owner", self.REC), + {"name": "slow session", "task": "t-res-owner", "batch": "r-7", "address": "uds:/s/b"}) + + def test_explicit_name_wins_then_env(self): + env = {"CLAUDE_PID": "22", "WF_SESSION_NAME": "pilot", "WF_TASK": "t-x"} + self.assertEqual(R.owner_by(env, "slow/t-y", self.REC, "worker-3"), {"name": "worker-3", "task": "t-x"}) + self.assertEqual(R.owner_by(env, "", self.REC), {"name": "pilot", "task": "t-x"}) + + def test_nothing_known(self): + self.assertEqual(R.owner_by({"CLAUDE_PID": "99"}, "master", self.REC), {}) + self.assertEqual(R.owner_by({}, "feature-x", [{"pid": 1}]), {}) + + def test_status_and_throttle_name_owner(self): + e = running("r-1", mem=4.0) + e.by = {"name": "slow session", "task": "t-a", "batch": "r-7", "address": "uds:/s/b"} + f = facts(units={"wf-r-1.service": R.Unit(True, 3.5, 33.2)}) + led = R.Ledger(entries=[e]) + self.assertTrue(R.status_lines(R.Config(), led, f)[0].endswith( + " [by slow session t-a batch r-7, message uds:/s/b]")) + self.assertTrue(R.throttle_lines(led, f)[0].endswith( + "bigger --mem; started by slow session t-a batch r-7, message uds:/s/b")) + + def test_ledger_roundtrip_and_old_entries(self): + e = running("r-1") + e.by = {"task": "t-a"} + self.assertEqual(R.loads(R.dumps(R.Ledger(entries=[e]))).get("r-1").by, {"task": "t-a"}) + d = json.loads(R.dumps(R.Ledger(entries=[running("r-2")]))) + for k in ("by", "lock"): # entries written before these fields + del d["entries"][0][k] + old = R.loads(json.dumps(d)).get("r-2") + self.assertEqual((old.by, old.lock), ({}, "")) + + +class BareUnits(unittest.TestCase): + def test_bare_units(self): + blocks = R.show_units( + "Id=wh-t27.service\nTransient=yes\nSlice=app.slice\nMemoryCurrent=2147483648\n\n" + "Id=run-u42.service\nTransient=yes\nSlice=app.slice\nMemoryCurrent=[not set]\n\n" + "Id=app-org.kde.konsole@65c0.service\nTransient=yes\nSlice=app.slice\nMemoryCurrent=1\n\n" + "Id=dbus-:1.1-org.kde.kwalletd6@0.service\nTransient=yes\nSlice=app.slice\nMemoryCurrent=1\n\n" + "Id=wf-r-3.service\nTransient=yes\nSlice=agents-jobs.slice\nMemoryCurrent=1\n\n" + "Id=pipewire.service\nTransient=no\nSlice=session.slice\nMemoryCurrent=1\n") + self.assertEqual(R.bare_units(blocks), [("run-u42.service", None), ("wh-t27.service", 2.0)]) + + def test_warning(self): + self.assertEqual(R.warning_lines(30.0, 0.0, 0, [("wh-t27.service", 2.0), ("run-u42.service", None)]), [ + "warning: 2 jobs outside wf res (bare systemd-run): wh-t27.service 2.0 GB, run-u42.service ? GB; " + "start jobs with wf res run, also from project scripts"]) + + +class Hook(unittest.TestCase): + BASH = {"tool_name": "Bash", "tool_input": {"command": "dotnet run --project Desktop", "timeout": 5000}} + + def test_gaming_hides_display(self): + out = R.hook_output(self.BASH, T(16, 0), NOW) + self.assertEqual(out, {"hookSpecificOutput": { + "hookEventName": "PreToolUse", + "updatedInput": {"command": "unset DISPLAY WAYLAND_DISPLAY; dotnet run --project Desktop", "timeout": 5000}, + "additionalContext": "wf res game mode until 16:00: this command has no display, so GUI windows fail. " + "Run headless/offscreen, or do other work until game off."}}) + + def test_quiet_when_not_gaming_or_not_bash(self): + self.assertIsNone(R.hook_output(self.BASH, None, NOW)) + self.assertIsNone(R.hook_output(self.BASH, T(14, 0), NOW)) # expired + self.assertIsNone(R.hook_output({"tool_name": "Read", "tool_input": {"file_path": "x"}}, T(16, 0), NOW)) + self.assertIsNone(R.hook_output({"tool_name": "Bash", "tool_input": {}}, T(16, 0), NOW)) + + +class Litter(unittest.TestCase): + def test_victims(self): + now, h = 10_000_000.0, 3600 + entries = [("/t/aB3xYz", "emptydir", now - 3 * h), # old empty dir → go + ("/t/MSBuildTemp1000", "emptydir", now - 5 * h), # → go + ("/t/fresh", "emptydir", now - 60), # too new + ("/t/full", "dir", now - 9 * h), # has content → never + ("/t/clr-debug-pipe-111-222-in", "other", now - 60), # pid 111 dead → go + ("/t/dotnet-diagnostic-333-444-socket", "other", now), # pid 333 alive + ("/t/notes.txt", "file", now - 9 * h)] # file → never + alive = {333}.__contains__ + self.assertEqual(R.litter_victims(entries, now, 2.0, alive), + ["/t/MSBuildTemp1000", "/t/aB3xYz", "/t/clr-debug-pipe-111-222-in"]) +class SessionCap(unittest.TestCase): + BLOCKS = { + "wf-claude-1.scope": {"Slice": "agents.slice", "MemoryHigh": "infinity", "MemoryCurrent": str(GIB)}, + "wf-claude-2.scope": {"Slice": "agents.slice", "MemoryHigh": str(10 * GIB), "MemoryCurrent": str(GIB)}, + "run-u7.scope": {"Slice": "agents.slice", "MemoryHigh": str(4 * GIB), "MemoryCurrent": str(GIB)}, + "podman-pause.scope": {"Slice": "user.slice", "MemoryHigh": "infinity", "MemoryCurrent": str(GIB)}, + } + + def test_caps_to_set(self): + self.assertEqual(R.session_caps(self.BLOCKS, 10.0), + [("run-u7.scope", str(10 * GIB)), ("wf-claude-1.scope", str(10 * GIB))]) + self.assertEqual(R.session_caps(self.BLOCKS, 0.0), # 0 = off → lift caps + [("run-u7.scope", "infinity"), ("wf-claude-2.scope", "infinity")]) + + def test_warn_at_cap(self): + blocks = {"wf-claude-5.scope": {"Slice": "agents.slice", "MemoryCurrent": str(int(9.6 * GIB))}, + "wf-claude-6.scope": {"Slice": "agents.slice", "MemoryCurrent": str(int(9.8 * GIB))}, + "wf-claude-7.scope": {"Slice": "agents.slice", "MemoryCurrent": str(int(2 * GIB))}} + stall = {"wf-claude-5.scope": 45.0, "wf-claude-6.scope": 1.0, "wf-claude-7.scope": 90.0} + self.assertEqual(R.session_cap_lines(blocks, stall, 10.0), [ + "warning: claude session 5 at its memory cap (9.6 of 10.0 GB, stalled 45% of the last minute): " + "run big work with wf res run"]) + self.assertEqual(R.session_cap_lines(blocks, stall, 0.0), []) + + +def run(title="gate 30735e0", peak=1.0, minutes=10.0, project="proj", mem=12.0, est=60, rc=0, id="r-1"): + return {"id": id, "project": project, "title": title, "mem_gb": mem, "est_min": est, "rc": rc, + "peak_gb": peak, "min": minutes} + + +TEN = [run(id=f"r-{i}", title=f"gate {i:07x}a", peak=float(i), minutes=6.0 * i) for i in range(1, 11)] + + +class History(unittest.TestCase): + def test_title_kind(self): + for title, kind in (("gate 30735e0", "gate"), ("gate 2c91cac7 (fix optimizer test)", "gate"), + ("mem-profile 16x5M (t-optimizer-x)", "mem-profile 16x5M"), ("wf-batch", "wf-batch"), + ("gate 97e6044b direct", "gate 97e6044b direct"), ("build deadbeef", "build deadbeef"), + ("gate a3cbde1 0cf9a79", "gate"), ("abc1234", "abc1234"), (" x ", "x")): + self.assertEqual(R.title_kind(title), kind, title) + + def test_pct_nearest_rank(self): + xs = [5.0, 1.0, 3.0, 2.0, 4.0] + self.assertEqual([R.pct(xs, p) for p in (0.5, 0.9, 0.95, 0.2)], [3.0, 5.0, 5.0, 1.0]) + + def test_history_record(self): + e = running(project="p", title="gate x", mem=10.0, est=40) + e.state, e.ended, e.rc, e.peak_gb = "done", T(14, 13, ), 0, 9.1 + e.ended = e.started + dt.timedelta(minutes=13, seconds=30) + self.assertEqual(R.history_record(e), {"id": "r-4", "project": "p", "title": "gate x", "mem_gb": 10.0, + "est_min": 40, "rc": 0, "peak_gb": 9.1, "min": 13.5}) + for change in ({"rc": 1}, {"peak_gb": None}, {"state": "running"}, {"ended": None}): + bad = R.Entry(**{**{f.name: getattr(e, f.name) for f in R.fields(R.Entry)}, **change}) + self.assertIsNone(R.history_record(bad), change) + + def test_suggest_p95_p90(self): + # peaks 1..10 → p95 = 10 → ×1.15 = 11.5; durations 6..60 → p90 = 54 → ×1.5 = 81 + self.assertEqual(R.suggest(TEN, "proj", "gate"), (11.5, 81, 10)) + + def test_suggest_floors_and_rounding(self): + tiny = [run(id=f"r-{i}", peak=0.01, minutes=1.0) for i in range(3)] + self.assertEqual(R.suggest(tiny, "proj", "gate"), (0.2, 5, 3)) + odd = [run(id=f"r-{i}", peak=1.01, minutes=7.1) for i in range(3)] + self.assertEqual(R.suggest(odd, "proj", "gate"), (1.2, 11, 3)) # 1.1615 → 1.2 up; 10.65 → 11 up + + def test_suggest_needs_three_matching(self): + two = TEN[:2] + [run(id="r-a", project="other"), run(id="r-b", title="build"), + run(id="r-c", rc=1), run(id="r-d", peak=None)] + self.assertIsNone(R.suggest(two, "proj", "gate")) + self.assertIsNone(R.suggest([], "proj", "gate")) + + def test_hint_line(self): + sug = (11.5, 81, 10) + line = "hint: history says ~11.5 GB / 81 min (10 runs)" + self.assertEqual(R.hint_line(23.1, 60, sug), line) + self.assertEqual(R.hint_line(10.0, 163, sug), line) + self.assertIsNone(R.hint_line(23.0, 162, sug)) + self.assertIsNone(R.hint_line(99.0, 999, None)) + + def test_merge_runs(self): + led = R.Ledger(entries=[running(id="r-4", project="proj", title="gate 1234567"), running(id="r-5")]) + led.entries[0].state, led.entries[0].rc, led.entries[0].peak_gb = "done", 0, 2.0 + led.entries[0].ended = led.entries[0].started + dt.timedelta(minutes=3) + merged = R.all_runs([run(id="r-1"), run(id="r-4", peak=7.0)], led) + self.assertEqual([(r["id"], r["peak_gb"]) for r in merged], [("r-1", 1.0), ("r-4", 7.0)]) + merged = R.all_runs([run(id="r-1")], led) + self.assertEqual([(r["id"], r["peak_gb"], r["min"]) for r in merged], [("r-1", 1.0, 10.0), ("r-4", 2.0, 3.0)]) + + def test_hist_lines(self): + runs = TEN + [run(id="r-x", title="wf-batch", peak=3.0, minutes=20.0, mem=6.0, est=400), + run(id="r-y", project="other", title="build")] + self.assertEqual(R.hist_lines(runs, "proj"), [ + "proj · gate · n 10 · req 12.0 GB · peak 5.0/10.0 GB · est 1h · dur 30m/54m · suggest 11.5 GB 1h21m", + "proj · wf-batch · n 1 · req 6.0 GB · peak 3.0/3.0 GB · est 6h40m · dur 20m/20m · suggest - (< 3 runs)"]) + self.assertEqual(len(R.hist_lines(runs, None)), 3) + self.assertEqual(R.hist_lines([], None), ["no finished runs with a peak yet"]) + + +ORCH_LOG = """\ +2026-10-06T10:00:00 slow opus t-a done abc1234 10m00s +2026-10-06T10:10:00 fast haiku t-b handback - 20m30s +2026-10-06T10:20:00 slow opus t-c done+gate-red abc1234 1h05m +2026-10-06T10:30:00 fast haiku t-d post-check-red abc1234 50m00s (branch x still there) +2026-10-06T10:40:00 fast haiku t-e done abc1234 0m00s +2026-10-06T10:50:00 slow opus t-f done abc1234 - +2026-10-06T11:00:00 cloud opus t-g done abc1234 3h00m +2026-10-06T11:10:00 slow sonnet t-h awaiting - 40m00s +21:30 pilot batch 1 rc=0 done=t-a stopped= added= +ALERT wf-pilot proj: awaiting a-1 +2026-10-06T11:20:00 slow opus t-i done abc1234 30m00s +""" + + +class TaskFit(unittest.TestCase): + def test_durations_from_log(self): + # done/handback rows of local lanes with a real duration only (minutes) + self.assertEqual(R.orch_durations(ORCH_LOG), [10.0, 20.5, 65.0, 30.0]) + + def test_p90(self): + # 4 rows: nearest rank ceil(0.9*4)=4 → 65 min + self.assertEqual(R.task_p90(ORCH_LOG), (65, 4)) + self.assertEqual(R.task_p90(""), (30, 0)) + two = "2026-10-06T10:00:00 slow opus t-a done abc 5m00s\n" * 2 + self.assertEqual(R.task_p90(two), (30, 2)) # < 3 runs → default 30 min + odd = "2026-10-06T10:00:00 slow opus t-a done abc 4m10s\n" * 3 + self.assertEqual(R.task_p90(odd), (5, 3)) # rounded up to whole minutes + + def test_batch_fit(self): + self.assertEqual(R.batch_fit(4, 150, 30), 4) + self.assertEqual(R.batch_fit(4, 90, 30), 3) + self.assertEqual(R.batch_fit(4, 59, 30), 1) + self.assertEqual(R.batch_fit(4, 29, 30), 0) + self.assertEqual(R.batch_fit(4, 0, 30), 0) diff --git a/tests/test_res_io.py b/tests/test_res_io.py new file mode 100644 index 0000000..d91046a --- /dev/null +++ b/tests/test_res_io.py @@ -0,0 +1,815 @@ +import datetime as dt +import io +import json +import os +import subprocess +import sys +import tempfile +import unittest +from contextlib import redirect_stderr, redirect_stdout +from pathlib import Path + +HERE = Path(__file__).resolve().parent.parent +sys.path.insert(0, str(HERE)) +import wf_res # noqa: E402 +from wflib import res as R # noqa: E402 + +UTC = dt.timezone.utc +GIB_KB = 1024 * 1024 + + +class Fake: + """Records argv; answers systemctl show from self.units {name: (ActiveState, MemoryCurrent bytes|None)}.""" + + def __init__(self): + self.calls, self.units, self.fail, self.journal, self.cgroups, self.bare = [], {}, {}, {}, {}, {} + self.scopes = {} # name → {prop: value} for agent session scopes + self.manager_env = "PATH=/usr/bin\nHOME=/h/u\n" + self.git = {} # git subcommand (branch | rev-parse) → stdout + + def __call__(self, argv): + self.calls.append(list(argv)) + key = argv[0] if argv[0] != "systemctl" else argv[2] + if key in self.fail: + return subprocess.CompletedProcess(argv, 1, "", self.fail[key] + "\n") + out = "" + if argv[:3] == ["systemctl", "--user", "show"]: + blocks = [] + for name in argv[5:]: + if name in self.scopes: + blocks.append("\n".join([f"Id={name}"] + [f"{k}={v}" for k, v in self.scopes[name].items()])) + continue + if name in self.bare: + blocks.append(f"Id={name}\nTransient=yes\nSlice=app.slice\nMemoryCurrent={self.bare[name]}") + continue + state, cur = self.units.get(name, ("inactive", None)) + blocks.append(f"Id={name}\nActiveState={state}\nMemoryCurrent={'[not set]' if cur is None else cur}" + f"\nControlGroup={self.cgroups.get(name, '')}") + out = "\n\n".join(blocks) + "\n" + elif argv[:3] == ["systemctl", "--user", "list-units"]: + names = self.scopes if "--type=scope" in argv else self.bare + out = "".join(f"{n} loaded active running x\n" for n in names) + elif argv[:3] == ["systemctl", "--user", "set-property"]: + if argv[4] in self.scopes: + self.scopes[argv[4]].update(a.split("=", 1) for a in argv[5:]) + elif argv[:3] == ["systemctl", "--user", "show-environment"]: + out = self.manager_env + elif argv[0] == "git": + out = self.git.get(argv[3], "") + elif argv[0] == "journalctl": + out = self.journal.get(argv[argv.index("-u") + 1], "") + elif argv[0] == "systemd-run": + unit = next(a for a in argv if a.startswith("--unit="))[len("--unit="):] + self.units[unit] = ("active", 0) + elif argv[:3] == ["systemctl", "--user", "stop"]: + self.units[argv[3]] = ("inactive", None) + return subprocess.CompletedProcess(argv, 0, out, "") + + def ran(self, word): + return [c for c in self.calls if word in c[:7]] + + +class Box: + """Temp dirs + fake machine; .env is a wf_res.Env.""" + + def __init__(self, tmp: Path, available_gb=20, total_gb=30, now=dt.datetime(2026, 10, 1, 14, 0, tzinfo=UTC)): + self.tmp, self.fake, self.clock, self.alive = tmp, Fake(), [now], {4242} + self.available_gb, self.total_gb = available_gb, total_gb + (tmp / "proj").mkdir(exist_ok=True) + self.env = wf_res.Env( + state=tmp / "state", config=tmp / "cfg" / "resources.toml", units=tmp / "units", + scratch=tmp / "scratch", run=self.fake, + meminfo=lambda: f"MemTotal: {int(self.total_gb * GIB_KB)} kB\nMemAvailable: {int(self.available_gb * GIB_KB)} kB\n", + nproc=16, now=lambda: self.clock[0], pid_alive=lambda p: p in self.alive, owner=lambda: 4242, + root=lambda: tmp / "proj", cwd=lambda: str(tmp / "proj"), procs=lambda: [], + sleep=self.tick, tmp_used_gb=lambda: 0.0, cgread=lambda cg, name: self.psi.get(cg, "") if name == "memory.pressure" + else self.stat.get(cg, "")) + self.psi, self.stat = {}, {} + self.env.live_sessions = lambda: set() + self.caller = {"PATH": "/usr/bin", "HOME": "/h/u"} + self.env.environ = lambda: dict(self.caller) + self.env.tmp_dirs = [] + + def tick(self, seconds): + self.clock[0] += dt.timedelta(seconds=seconds) + + def wf(self, *argv): + out, err = io.StringIO(), io.StringIO() + with redirect_stdout(out), redirect_stderr(err): + code = wf_res.main(list(argv), self.env) + return code, out.getvalue(), err.getvalue() + + def ledger(self): + return R.loads((self.env.state / "resources.json").read_text()) + + +class IOBase(unittest.TestCase): + def setUp(self): + self._tmp = tempfile.TemporaryDirectory() + self.box = Box(Path(self._tmp.name)) + + def tearDown(self): + self._tmp.cleanup() + + +class Run(IOBase): + def test_run_starts_unit(self): + code, out, err = self.box.wf("run", "--mem", "4G", "--for", "40m", "--title", "build", "--", "make", "-j8") + self.assertEqual((code, err), (0, "")) + log = self.box.env.state / "logs" / "r-1.log" + self.assertEqual(out, f"r-1 started; log {log}; ETA ~14:40 (estimate: not killed when over)\n") + argv = self.box.fake.ran("systemd-run")[0] + self.assertEqual(argv[-3:], ["sh", "make", "-j8"]) + self.assertIn("--unit=wf-r-1.service", argv) + e = self.box.ledger().get("r-1") + self.assertEqual((e.state, e.project, e.owner, e.cwd), ("running", "proj", 4242, str(self.box.tmp / "proj"))) + self.assertTrue((self.box.env.units / "agents.slice").exists()) + + def test_run_keeps_caller_env(self): + self.box.caller.update({"APP_DIR": "/data/arc", "PATH": "/venv/bin:/usr/bin", "PWD": "/x", "SHLVL": "2", + "CLAUDE_CODE_MESSAGING_TOKEN": "t", "GH_TOKEN": "t", "DB_PASSWORD": "p"}) + self.box.wf("run", "--mem", "1G", "--for", "1m", "--title", "x", "--", "x") + argv = self.box.fake.ran("systemd-run")[0] + self.assertEqual([a for a in argv if a.startswith("--setenv=")], + ["--setenv=APP_DIR=/data/arc", "--setenv=PATH=/venv/bin:/usr/bin", "--setenv=WF_RES_ID=r-1"]) + self.assertEqual(self.box.ledger().get("r-1").env, {"APP_DIR": "/data/arc", "PATH": "/venv/bin:/usr/bin"}) + + def test_run_records_owner(self): + main = self.box.tmp / "main" + (main / ".wf" / "sessions").mkdir(parents=True) + (main / ".wf" / "sessions" / "slow.json").write_text('{"lane": "slow", "pid": 77, "socket": "/s/o"}\n') + self.box.fake.git = {"branch": "slow/t-big\n", "rev-parse": f"{main}/.git\n"} + self.box.caller.update({"CLAUDE_PID": "77", "CLAUDE_CODE_MESSAGING_SOCKET": "/s/o", "WF_RES_ID": "r-9"}) + self.box.wf("run", "--mem", "1G", "--for", "1m", "--title", "x", "--", "x") + self.assertEqual(self.box.ledger().get("r-1").by, + {"name": "slow session", "task": "t-big", "batch": "r-9", "address": "uds:/s/o"}) + argv = self.box.fake.ran("systemd-run")[0] + self.assertEqual([a for a in argv if a.startswith("--setenv=WF_RES_ID")], ["--setenv=WF_RES_ID=r-1"]) + self.assertIn('"x" running', self.box.wf("status")[1]) + self.assertIn("[by slow session t-big batch r-9, message uds:/s/o]", self.box.wf("status")[1]) + self.box.wf("note", "--mem", "1G", "--for", "5m", "--by", "pilot", "edit") + self.assertEqual(self.box.ledger().get("r-2").by["name"], "pilot") + + def test_queued_run_starts_later_with_its_env(self): + self.box.wf("run", "--mem", "10G", "--for", "40m", "--title", "big", "--", "x") + self.box.caller["FOO"] = "1" + self.box.wf("run", "--mem", "8G", "--for", "10m", "--title", "q", "--queue", "--", "y") + self.box.caller.pop("FOO") + self.box.wf("release", "r-1", "--stop") + self.box.wf("status") + starts = self.box.fake.ran("systemd-run") + self.assertEqual([[a for a in s if a.startswith("--setenv=")] for s in starts], [["--setenv=WF_RES_ID=r-1"], ["--setenv=FOO=1", "--setenv=WF_RES_ID=r-2"]]) + + def test_busy_exit_3(self): + self.box.wf("run", "--mem", "10G", "--for", "40m", "--title", "big", "--", "x") + code, out, _ = self.box.wf("run", "--mem", "8G", "--for", "10m", "--title", "two", "--", "y") + # 20 − 6 − 2 − 10 unused = 2.0 free + self.assertEqual(code, 3) + self.assertEqual(out, 'busy: 10.0 GB held by proj "big" (r-1) until ~14:40; 2.0 GB free for agents; ' + 'retry after ~14:40 or work on something else; 12.0 GB really free beyond the reserve: ' + '--force starts it past the ledger (only if the holders will not use what they reserved)\n') + self.assertEqual([e.id for e in self.box.ledger().entries], ["r-1"]) + + def test_force_runs_past_unused_claim(self): + self.box.wf("run", "--mem", "10G", "--for", "40m", "--title", "big", "--", "x") + code, out, _ = self.box.wf("run", "--mem", "8G", "--for", "10m", "--title", "two", "--force", "--", "y") + # really free: 20 − 6 reserve − 2 headroom = 12 ≥ 8 (r-1's unused 10 GB ignored) + self.assertEqual(code, 0) + self.assertTrue(out.startswith("r-2 started"), out) + self.assertEqual([(e.id, e.state) for e in self.box.ledger().entries], [("r-1", "running"), ("r-2", "running")]) + + def test_force_refused_when_memory_really_used(self): + self.box.available_gb = 12 # really free beyond reserve: 12 − 6 − 2 = 4 + self.box.wf("run", "--mem", "3G", "--for", "40m", "--title", "a", "--", "x") + code, out, _ = self.box.wf("run", "--mem", "8G", "--for", "10m", "--title", "two", "--", "y") + self.assertEqual(code, 3) + self.assertNotIn("--force", out) # hint only when --force would fit + code, out, _ = self.box.wf("run", "--mem", "8G", "--for", "10m", "--title", "two", "--force", "--", "y") + self.assertEqual((code, out), (3, "busy even with --force: only 4.0 GB really free beyond the reserve; " + "retry later or work on something else\n")) + self.assertEqual([e.id for e in self.box.ledger().entries], ["r-1"]) + + def test_never_fits(self): + code, _, err = self.box.wf("run", "--mem", "25G", "--for", "1h", "--title", "x", "--", "x") + self.assertEqual((code, err), (1, "wf: 25.0 GB can never fit (max 22.0 GB for agents)\n")) + + def test_systemd_run_failure(self): + self.box.fake.fail["systemd-run"] = "Failed to start transient service unit: boom" + code, _, err = self.box.wf("run", "--mem", "1G", "--for", "1m", "--title", "x", "--", "x") + self.assertEqual((code, err), (1, "wf: systemd-run: Failed to start transient service unit: boom\n")) + self.assertEqual(self.box.ledger().entries, []) + + def test_no_command(self): + code, _, err = self.box.wf("run", "--mem", "1G", "--for", "1m", "--title", "x") + self.assertEqual((code, err), (1, "wf: no command after --\n")) + + def test_usage_error_exit_2(self): + with redirect_stderr(io.StringIO()): + self.assertEqual(self.box.wf("run", "--for", "1m")[0], 2) + + +class Status(IOBase): + def test_status_and_exit_pruned(self): + self.box.wf("run", "--mem", "4G", "--for", "40m", "--title", "build", "--", "make") + self.box.fake.units["wf-r-1.service"] = ("active", 2 * 1024 ** 3) + code, out, _ = self.box.wf("status") + self.assertEqual(out.splitlines()[0], 'r-1 proj "build" running 4.0 GB used 2.0 GB 1 cpu since 14:00 ETA ~14:40') + logs = self.box.env.state / "logs" + (logs / "r-1.rc").write_text("0\n") + (logs / "r-1.peak").write_text(str(3 * 1024 ** 3) + "\n") + self.box.fake.units["wf-r-1.service"] = ("inactive", None) + code, out, _ = self.box.wf("status") + self.assertEqual(out.splitlines()[0], 'r-1 proj "build" done (exited) rc=0 peak 3.0 GB at 14:00') + + def test_status_warns_throttled(self): + self.box.wf("run", "--mem", "4G", "--for", "40m", "--title", "build", "--", "make") + self.box.fake.units["wf-r-1.service"] = ("active", int(3.8 * 1024 ** 3)) + self.box.fake.cgroups["wf-r-1.service"] = "/x/wf-r-1.service" + self.box.psi["/x/wf-r-1.service"] = "some avg10=50.00 avg60=40.00 avg300=10.00 total=1\n" + out = self.box.wf("status")[1] + self.assertIn("r-1 throttled at its memory limit (3.8 of 4.0 GB, stalled 40% of the last minute): " + "likely too small; `wf res release r-1 --stop` and re-run with a bigger --mem\n", out) + + def test_status_warns_bare_units(self): + self.box.fake.bare["wh-t27.service"] = 2 * 1024 ** 3 + out = self.box.wf("status")[1] + self.assertIn("warning: 1 jobs outside wf res (bare systemd-run): wh-t27.service 2.0 GB; " + "start jobs with wf res run, also from project scripts\n", out) + self.assertIn(["systemctl", "--user", "list-units", "--type=service", "--state=running", "--no-legend", + "--plain"], self.box.fake.calls) + + def test_status_splits_slice_cache(self): + self.box.fake.units["agents.slice"] = ("active", 5 * 1024 ** 3) + self.box.fake.cgroups["agents.slice"] = "/a.slice" + self.box.stat["/a.slice"] = f"anon {3 * 1024 ** 3}\nfile {2 * 1024 ** 3}\nshmem {1024 ** 3 // 2}\n" + self.assertIn("unreserved agent memory 5.0 GB (1.5 GB of it file cache, reclaimable)\n", self.box.wf("status")[1]) + + def test_status_id_and_json(self): + self.box.wf("run", "--mem", "4G", "--for", "40m", "--title", "build", "--", "make", "a b") + out = self.box.wf("status", "r-1")[1] + self.assertIn("cmd: make 'a b'", out) + self.assertEqual(json.loads(self.box.wf("status", "--json")[1])["entries"][0]["id"], "r-1") + + def test_show_failure_keeps_ledger(self): + self.box.wf("run", "--mem", "4G", "--for", "40m", "--title", "build", "--", "make") + self.box.fake.fail["show"] = "Failed to connect to bus" + code, _, err = self.box.wf("status") + self.assertEqual((code, err), (1, "wf: systemctl show: Failed to connect to bus\n")) + self.assertEqual(self.box.ledger().get("r-1").state, "running") + + def test_corrupt_ledger_moved_aside(self): + self.box.env.state.mkdir(parents=True) + (self.box.env.state / "resources.json").write_text("{oops") + code, out, err = self.box.wf("status") + self.assertEqual(code, 0) + self.assertRegex(err, r"^wf: corrupt ledger: .*; moved to resources\.json\.bad-20261001-140000, starting empty\n$") + self.assertTrue((self.box.env.state / "resources.json.bad-20261001-140000").exists()) + self.assertEqual(out.splitlines()[0], "no reservations") + + +class LedgerCompat(IOBase): + """Readers of another code version (a long `wf res wait` started before a release) must keep entries.""" + + def _write(self, extra: dict, drop=()): + self.box.wf("run", "--mem", "1G", "--for", "5m", "--title", "t", "--", "x") + path = self.box.env.state / "resources.json" + d = json.loads(path.read_text()) + for e in d["entries"]: + for k in drop: + e.pop(k, None) + e.update(extra) + path.write_text(json.dumps(d)) + return path + + def _kept(self, path): + for argv in (("status",), ("tick",), ("wait", "r-1", "--timeout", "1m")): + self.box.wf(*argv) + self.assertEqual(self.box.ledger().get("r-1").state, "running", argv) + self.assertEqual(sorted(p.name for p in path.parent.glob("resources.json*")), ["resources.json"]) + self.assertIn("r-1", self.box.wf("status")[1]) + + def test_pre_owner_ledger_kept(self): # written before e84d7ca/9533211: no by field + self._kept(self._write({}, drop=("by", "env"))) + + def test_future_fields_kept(self): # written by newer code: unknown keys + self._kept(self._write({"by": {"name": "x"}, "added_later": 1})) + + def test_corrupt_ledger_keeps_id_counter(self): + self.box.env.state.mkdir(parents=True) + (self.box.env.state / "resources.json").write_text('{"next": 652, "entries": [{"id": "r-651"}]}') + self.box.wf("run", "--mem", "1G", "--for", "5m", "--title", "t", "--", "x") + self.assertEqual([e.id for e in self.box.ledger().entries], ["r-652"]) + + def test_start_clears_stale_results(self): + logs = self.box.env.state / "logs" + logs.mkdir(parents=True) + (logs / "r-1.rc").write_text("1\n") + (logs / "r-1.peak").write_text("5\n") + self.box.wf("run", "--mem", "1G", "--for", "5m", "--title", "t", "--", "x") + self.assertFalse((logs / "r-1.rc").exists() or (logs / "r-1.peak").exists()) + + +class HookIO(IOBase): + def hook(self, text): + self.box.env.stdin = lambda: text + return self.box.wf("hook") + + def test_game_on_rewrites_bash(self): + self.box.wf("game", "on", "--for", "1h") + code, out, err = self.hook('{"tool_name": "Bash", "tool_input": {"command": "./app"}}') + self.assertEqual((code, err), (0, "")) + self.assertEqual(json.loads(out)["hookSpecificOutput"]["updatedInput"], + {"command": "unset DISPLAY WAYLAND_DISPLAY; ./app"}) + calls = len(self.box.fake.calls) + self.hook('{"tool_name": "Bash", "tool_input": {"command": "./app"}}') + self.assertEqual(len(self.box.fake.calls), calls) # read-only: no systemctl, no prune + + def test_silent_otherwise(self): + self.assertEqual(self.hook('{"tool_name": "Bash", "tool_input": {"command": "./app"}}'), (0, "", "")) + self.box.wf("game", "on") + self.assertEqual(self.hook("not json"), (0, "", "")) + (self.box.env.state / "resources.json").write_text("{oops") + self.assertEqual(self.hook('{"tool_name": "Bash", "tool_input": {"command": "x"}}'), (0, "", "")) + self.assertTrue((self.box.env.state / "resources.json").exists()) # hook never moves a bad ledger + + +class Release(IOBase): + def test_release_running_needs_stop(self): + self.box.wf("run", "--mem", "4G", "--for", "40m", "--title", "build", "--", "make") + code, _, err = self.box.wf("release", "r-1") + self.assertEqual((code, err), (1, "wf: r-1 is running; --stop to kill it\n")) + code, out, _ = self.box.wf("release", "r-1", "--stop") + self.assertEqual((code, out), (0, "r-1 stopped\n")) + self.assertEqual(self.box.fake.ran("stop")[0], ["systemctl", "--user", "stop", "wf-r-1.service"]) + e = self.box.ledger().get("r-1") + self.assertEqual((e.state, e.why), ("done", "released")) + + def test_release_unknown(self): + self.assertEqual(self.box.wf("release", "r-7")[2], "wf: no entry 'r-7'\n") + + +class Dispatch(unittest.TestCase): + def test_wf_forwards_res(self): + r = subprocess.run([sys.executable, str(HERE / "wf.py"), "res", "-h"], capture_output=True, text=True) + self.assertEqual(r.returncode, 0) + self.assertIn("usage: wf res", r.stdout) + +class Wait(IOBase): + def test_wait_until_done(self): + self.box.wf("run", "--mem", "1G", "--for", "5m", "--title", "t", "--", "x") + logs = self.box.env.state / "logs" + polls = [] + + def sleep(seconds): + polls.append(seconds) + self.box.tick(seconds) + if len(polls) == 2: + (logs / "r-1.rc").write_text("0\n") + (logs / "r-1.peak").write_text(str(1024 ** 3 // 2) + "\n") + self.box.fake.units["wf-r-1.service"] = ("inactive", None) + self.box.env.sleep = sleep + code, out, _ = self.box.wf("wait", "r-1") + self.assertEqual((code, out, polls), (0, "r-1 done rc=0 peak 0.5 GB in 0 min\n", [15, 15])) + + def test_wait_warns_throttled_once(self): + self.box.wf("run", "--mem", "4G", "--for", "5m", "--title", "t", "--", "x") + self.box.fake.units["wf-r-1.service"] = ("active", int(3.8 * 1024 ** 3)) + self.box.fake.cgroups["wf-r-1.service"] = "/x/wf-r-1.service" + self.box.psi["/x/wf-r-1.service"] = "some avg10=50.00 avg60=40.00 avg300=10.00 total=1\n" + code, _, err = self.box.wf("wait", "r-1", "--timeout", "1m") + self.assertEqual(err, "wf: r-1 throttled at its memory limit (3.8 of 4.0 GB, stalled 40% of the last minute): " + "likely too small; `wf res release r-1 --stop` and re-run with a bigger --mem\n" + "wf: r-1 not done after 1m (running)\n") + + def test_wait_timeout(self): + self.box.wf("run", "--mem", "1G", "--for", "5m", "--title", "t", "--", "x") + code, _, err = self.box.wf("wait", "r-1", "--timeout", "1m") + self.assertEqual((code, err), (1, "wf: r-1 not done after 1m (running)\n")) + + def test_wait_killed_job(self): + self.box.wf("run", "--mem", "1G", "--for", "5m", "--title", "t", "--", "x") + self.box.fake.units["wf-r-1.service"] = ("failed", None) + self.assertEqual(self.box.wf("wait", "r-1")[1], + "r-1 done rc=? peak ? in 0 min (no exit code; see journalctl --user -u wf-r-1.service)\n") + + def test_wait_oom_killed_job(self): + self.box.wf("run", "--mem", "4G", "--for", "5m", "--title", "t", "--", "x") + self.box.fake.units["wf-r-1.service"] = ("failed", None) + self.box.fake.journal["wf-r-1.service"] = ( + "wf-r-1.service: systemd-oomd killed some process(es) in this unit.\n" + "wf-r-1.service: Failed with result 'oom-kill'.\n" + "wf-r-1.service: Consumed 11min 2.837s CPU time, 3.9G memory peak.\n") + self.assertEqual(self.box.wf("wait", "r-1")[1], "r-1 done rc=? peak 3.9 GB in 0 min " + "(killed: oom-kill by systemd-oomd, limit 4.0 GB; raise --mem)\n") + j = self.box.fake.ran("journalctl")[0] + self.assertEqual(j, ["journalctl", "--user", "-u", "wf-r-1.service", "-o", "cat", "--no-pager", + "--since", "@" + str(int(dt.datetime(2026, 10, 1, 14, 0, tzinfo=UTC).timestamp()))]) + out = self.box.wf("status")[1] + self.assertEqual(out.splitlines()[0], 'r-1 proj "t" done (killed: oom-kill by systemd-oomd, limit 4.0 GB; ' + 'raise --mem) rc=? peak 3.9 GB at 14:00') + + def test_rc_file_skips_journal(self): + self.box.wf("run", "--mem", "1G", "--for", "5m", "--title", "t", "--", "x") + (self.box.env.state / "logs" / "r-1.rc").write_text("2\n") + self.box.fake.units["wf-r-1.service"] = ("inactive", None) + self.assertEqual(self.box.wf("wait", "r-1")[1], "r-1 done rc=2 peak ? in 0 min\n") + self.assertEqual(self.box.fake.ran("journalctl"), []) + +class QueueIO(IOBase): + def test_queue_then_start_when_free(self): + self.box.wf("run", "--mem", "10G", "--for", "40m", "--title", "big", "--", "x") + code, out, _ = self.box.wf("run", "--mem", "8G", "--for", "10m", "--title", "two", "--queue", "--", "y") + self.assertEqual((code, out), (0, "r-2 queued, position 1; est. start ~14:40; cancel: wf res release r-2\n")) + self.box.fake.units["wf-r-1.service"] = ("inactive", None) + self.box.wf("status") + e = self.box.ledger().get("r-2") + self.assertEqual((e.state, e.unit), ("running", "wf-r-2.service")) + self.assertEqual(self.box.fake.ran("systemd-run")[-1][-2:], ["sh", "y"]) + + def test_direct_run_does_not_jump_queue(self): + self.box.wf("run", "--mem", "10G", "--for", "40m", "--title", "big", "--", "x") + self.box.wf("run", "--mem", "8G", "--for", "10m", "--title", "two", "--queue", "--", "y") + code, out, _ = self.box.wf("run", "--mem", "1G", "--for", "10m", "--title", "small", "--", "z") + self.assertEqual(code, 3) + + def test_queued_start_failure_moves_on(self): + self.box.wf("run", "--mem", "10G", "--for", "40m", "--title", "big", "--", "x") + self.box.wf("run", "--mem", "8G", "--for", "10m", "--title", "two", "--queue", "--", "y") + self.box.fake.units["wf-r-1.service"] = ("inactive", None) + self.box.fake.fail["systemd-run"] = "boom" + code, out, err = self.box.wf("status") + self.assertEqual(code, 0) + e = self.box.ledger().get("r-2") + self.assertEqual((e.state, e.why), ("done", "start failed: systemd-run: boom")) + self.box.fake.fail.clear() + self.assertEqual(self.box.wf("status")[0], 0) + +class Note(IOBase): + def test_note_and_owner_exit(self): + code, out, _ = self.box.wf("note", "--mem", "3G", "--for", "20m", "dotnet test") + self.assertEqual((code, out), (0, "r-1 noted 3.0 GB until ~14:20 (then freed; nothing is killed)\n")) + self.assertEqual(self.box.ledger().get("r-1").owner, 4242) + self.box.alive.clear() + self.box.wf("status") + self.assertEqual(self.box.ledger().get("r-1").why, "owner gone") + + def test_note_busy(self): + self.box.wf("note", "--mem", "10G", "--for", "20m", "a") + code, out, _ = self.box.wf("note", "--mem", "3G", "--for", "20m", "b") + self.assertEqual((code, out), (3, 'busy: 10.0 GB held by proj "a" (r-1) until ~14:20; 2.0 GB free for agents; ' + 'retry after ~14:20 or work on something else; 12.0 GB really free ' + 'beyond the reserve: --force starts it past the ledger (only if the ' + 'holders will not use what they reserved)\n')) + code, out, _ = self.box.wf("note", "--mem", "3G", "--for", "20m", "--force", "b") + self.assertEqual(code, 0) + self.assertTrue(out.startswith("r-2 noted 3.0 GB"), out) + + +def _race_child(tmp, q): + import time as _t + box = Box(Path(tmp), available_gb=14) # budget 14 − 6 − 2 = 6 GB: one 5G note fits, not two + slow = box.env.meminfo + box.env.meminfo = lambda: (_t.sleep(0.3), slow())[1] + with redirect_stdout(io.StringIO()), redirect_stderr(io.StringIO()): + q.put(wf_res.main(["note", "--mem", "5G", "--for", "10m", "x"], box.env)) + + +class LockRace(unittest.TestCase): + def test_two_processes_never_overbook(self): + import multiprocessing as mp + ctx = mp.get_context("fork") + with tempfile.TemporaryDirectory() as tmp: + q = ctx.Queue() + ps = [ctx.Process(target=_race_child, args=(tmp, q)) for _ in range(2)] + for p in ps: + p.start() + for p in ps: + p.join(10) + self.assertEqual([p.exitcode for p in ps], [0, 0]) # a crashed child must fail, not hang q.get() + self.assertEqual(sorted([q.get(timeout=5), q.get(timeout=5)]), [0, 3]) + +class GameIO(IOBase): + def props(self): + return [c[5:] for c in self.box.fake.ran("set-property")] + + def test_on_off(self): + code, out, _ = self.box.wf("game", "on") + self.assertEqual((code, out), (0, "game on until 18:00 (4h); CPU/IO now yours\n")) + self.assertEqual(self.props()[-1], ["CPUWeight=5", "IOWeight=5", f"MemoryHigh={18 * 1024 ** 3}"]) + self.assertEqual(self.box.fake.ran("set-property")[-1][:5], + ["systemctl", "--user", "set-property", "--runtime", "agents.slice"]) + self.assertEqual(self.box.wf("game", "off")[1], "game off\n") + self.assertEqual(self.props()[-1], ["CPUWeight=20", "IOWeight=20", f"MemoryHigh={24 * 1024 ** 3}"]) + + def test_expiry_restores(self): + self.box.wf("game", "on", "--for", "1h") + self.box.tick(3601) + self.box.wf("status") + self.assertIsNone(self.box.ledger().game_until) + self.assertEqual(self.props()[-1][0], "CPUWeight=20") + + def test_game_refuses_new_job(self): + self.box.wf("game", "on") + # 20 − 12 − 2 = 6 GB budget + self.assertEqual(self.box.wf("run", "--mem", "7G", "--for", "5m", "--title", "x", "--", "x")[0], 3) + + def test_on_again_extends(self): + self.box.wf("game", "on", "--for", "2h") + self.box.wf("game", "on", "--for", "1h") + self.assertEqual(self.box.ledger().game_until.hour, 16) + + +class CleanIO(IOBase): + def make(self, rel, age_h, size=10): + p = self.box.env.scratch / rel + p.parent.mkdir(parents=True, exist_ok=True) + p.write_bytes(b"x" * size) + t = self.box.clock[0].timestamp() - age_h * 3600 + os.utime(p, (t, t)) + for d in [p.parent, *p.parent.parents]: + if d == self.box.env.scratch.parent: + break + os.utime(d, (t, t)) + return p + + def test_auto_clean_sweeps_tmp_litter(self): + t = self.box.tmp / "tmpdir" + (t / "old-empty").mkdir(parents=True) + (t / "full").mkdir() + (t / "full" / "x").write_text("x") + (t / "clr-debug-pipe-999999-1-in").write_text("") + (t / "keep.txt").write_text("x") + old = self.box.clock[0].timestamp() - 3 * 3600 + for n in ("old-empty", "full", "keep.txt"): + os.utime(t / n, (old, old)) + self.box.env.tmp_dirs = [t] + self.box.env.pid_alive = lambda p: p != 999999 and p in self.box.alive + self.box.wf("status") + self.assertEqual(sorted(os.listdir(t)), ["full", "keep.txt"]) + + def test_auto_clean_keeps_live_session(self): + live = self.make("-projects-a/live-1/scratchpad/notes.txt", 9) + dead = self.make("-projects-a/dead-2/scratchpad/x.txt", 9) + self.box.env.live_sessions = lambda: {"live-1"} + self.box.wf("status") + self.assertEqual((live.exists(), dead.exists()), (True, False)) + + def test_auto_clean_on_status_and_throttle(self): + old = self.make("-projects-a/s1/scratchpad/big.bin", 3) + new = self.make("-projects-b/s2/scratchpad/x.txt", 0.5) + self.box.wf("status") + self.assertFalse(old.exists()) + self.assertTrue(new.exists()) + self.assertTrue(any(c[2] == "reset-failed" for c in self.box.fake.calls if c[0] == "systemctl")) + again = self.make("-projects-c/s3/a.txt", 3) + self.box.wf("status") + self.assertTrue(again.exists()) # < 10 min since last clean + self.box.tick(601) + self.box.wf("status") + self.assertFalse(again.exists()) + + def test_clean_prints_and_lists_project_patterns(self): + self.make("-projects-a/s1/f.bin", 3, size=2048) + proj = self.box.tmp / "proj" + (proj / "workflow.toml").write_text('cleanup = ["out/prof", "out/logs/*.log:30d"]\n') + (proj / "out" / "prof").mkdir(parents=True) + (proj / "out" / "prof" / "p.dat").write_bytes(b"x" * 1024) + logs = proj / "out" / "logs" + logs.mkdir() + (logs / "new.log").write_text("n") + old = logs / "old.log" + old.write_text("o") + t = self.box.clock[0].timestamp() - 31 * 86400 + os.utime(old, (t, t)) + (logs / "link.log").symlink_to("/etc/hostname") + code, out, _ = self.box.wf("clean") + self.assertEqual(out.splitlines(), [ + f"freed 0 MB: {self.box.env.scratch / '-projects-a'}", + "would delete 0 MB: out/prof (wf res clean --yes)", + "would delete 0 MB: out/logs/old.log (wf res clean --yes)"]) + self.assertTrue((proj / "out" / "prof").exists()) + self.box.wf("clean", "--yes") + self.assertFalse((proj / "out" / "prof").exists()) + self.assertFalse(old.exists()) + self.assertTrue((logs / "new.log").exists()) + self.assertTrue((logs / "link.log").is_symlink()) + + def test_status_warnings(self): + self.box.env.tmp_used_gb = lambda: 8.0 + self.box.env.procs = lambda: [(10, 1, "claude", "/u/app.slice/tab.scope"), + (20, 1, "claude", "/u/agents.slice/wf-claude-20.scope")] + out = self.box.wf("status")[1].splitlines() + self.assertEqual(out[-2:], ["warning: /tmp (RAM) holds 8.0 GB; wf res clean", + "warning: 1 claude sessions outside agents.slice (wf res adopt)"]) + +class TimerIO(IOBase): + def test_timer_on_off(self): + code, out, _ = self.box.wf("timer", "on") + self.assertEqual((code, out), (0, "timer on: wf-res.timer every 1 min\n")) + units = self.box.env.units + self.assertIn("res tick", (units / "wf-res.service").read_text()) + self.assertTrue((units / "wf-res.timer").exists()) + self.assertIn(["systemctl", "--user", "enable", "--now", "wf-res.timer"], self.box.fake.calls) + self.assertEqual(self.box.wf("timer", "off")[1], "timer off\n") + self.assertFalse((units / "wf-res.timer").exists()) + self.assertIn(["systemctl", "--user", "disable", "--now", "wf-res.timer"], self.box.fake.calls) + + def test_adopt(self): + self.box.env.procs = lambda: [(101, 1, "claude", "/u/app.slice/t.scope"), (102, 101, "bash", "/u/app.slice/t.scope"), + (201, 1, "claude", "/u/app.slice/u.scope")] + code, out, _ = self.box.wf("adopt") + self.assertEqual(out, "adopted claude 101 (2 processes)\nadopted claude 201 (1 processes)\n") + self.assertEqual(self.box.fake.ran("StartTransientUnit")[0][8], "wf-claude-101.scope") + + def test_adopt_failure_reported(self): + self.box.env.procs = lambda: [(101, 1, "claude", "/u/app.slice/t.scope")] + self.box.fake.fail["busctl"] = "Call failed: No such process" + self.assertEqual(self.box.wf("adopt")[1], "claude 101: Call failed: No such process\n") + + def test_adopt_none(self): + self.assertEqual(self.box.wf("adopt")[1], "all claude sessions already in agents.slice\n") + + def test_tick_adopts_silently(self): + self.box.env.procs = lambda: [(101, 1, "claude", "/u/app.slice/t.scope")] + self.assertEqual(self.box.wf("tick"), (0, "", "")) + self.assertEqual(len(self.box.fake.ran("StartTransientUnit")), 1) + + def test_tick_caps_sessions(self): + self.box.fake.scopes = {"wf-claude-5.scope": {"Slice": "agents.slice", "MemoryHigh": "infinity", + "MemoryCurrent": str(1024 ** 3), "ControlGroup": "/a/5"}, + "init.scope": {"Slice": "-.slice", "MemoryHigh": "infinity", "MemoryCurrent": "1"}} + self.assertEqual(self.box.wf("tick"), (0, "", "")) + self.assertEqual(self.box.fake.ran("set-property"), + [["systemctl", "--user", "set-property", "--runtime", "wf-claude-5.scope", + f"MemoryHigh={6 * 1024 ** 3}"]]) + self.box.wf("tick") + self.assertEqual(len(self.box.fake.ran("set-property")), 1) # already capped → no call + + def test_status_warns_session_at_cap(self): + self.box.fake.scopes = {"wf-claude-5.scope": {"Slice": "agents.slice", "MemoryHigh": str(6 * 1024 ** 3), + "MemoryCurrent": str(int(5.7 * 1024 ** 3)), "ControlGroup": "/a/5"}} + self.box.psi["/a/5"] = "some avg10=50.00 avg60=40.00 avg300=10.00 total=1\n" + self.assertIn("warning: claude session 5 at its memory cap (5.7 of 6.0 GB, stalled 40% of the last minute): " + "run big work with wf res run\n", self.box.wf("status")[1]) + + def test_tick_silent_and_starts_queue(self): + self.box.wf("run", "--mem", "10G", "--for", "40m", "--title", "big", "--", "x") + self.box.wf("run", "--mem", "8G", "--for", "10m", "--title", "two", "--queue", "--", "y") + self.box.fake.units["wf-r-1.service"] = ("inactive", None) + self.assertEqual(self.box.wf("tick"), (0, "", "")) + self.assertEqual(self.box.ledger().get("r-2").state, "running") + + def test_shell_init(self): + self.assertEqual(self.box.wf("shell-init")[1], + "alias claude='systemd-run --user --scope --quiet --slice=agents.slice claude'\n") + + +class Robust(IOBase): + def test_vanished_paths_are_skipped(self): + gone = self.box.tmp / "gone" + self.assertEqual((wf_res._newest(gone), wf_res._size(gone)), (0.0, 0)) + + def test_os_error_is_one_line(self): + def boom(): + raise PermissionError(13, "Permission denied", "/proc/meminfo") + self.box.env.meminfo = boom + code, _, err = self.box.wf("status") + self.assertEqual((code, err), (1, "wf: [Errno 13] Permission denied: '/proc/meminfo'\n")) + + +if __name__ == "__main__": + unittest.main() + + +class LockIO(IOBase): + """--lock / 'gate …' titles: one job per lock key and main tree at a time (shared gate checkout).""" + + def gate(self, title, *extra): + return self.box.wf("run", "--mem", "2G", "--for", "30m", "--title", title, *extra, "--", "x") + + def test_second_gate_refused_even_with_force(self): + self.assertEqual(self.gate("gate c61907f0")[0], 0) + for extra in ((), ("--force",)): + code, out, err = self.gate("gate 2c91cac7 (fix)", *extra) + self.assertEqual(code, 3, err) + self.assertIn("lock 'gate' held by r-1", out + err) + self.assertIn("--queue", out + err) + self.assertEqual(len(self.box.fake.ran("systemd-run")), 1) + + def test_second_gate_queues_and_starts_after_first(self): + self.gate("gate c61907f0") + code, out, _ = self.gate("gate 2c91cac7", "--queue") + self.assertEqual(code, 0) + self.assertTrue(out.startswith("r-2 queued, position 1 (lock held by r-1); est. start ~14:30"), out) + self.box.wf("status") # memory is free, lock is not + self.assertEqual(self.box.ledger().get("r-2").state, "queued") + other = self.box.wf("run", "--mem", "1G", "--for", "5m", "--title", "build", "--", "y") + self.assertEqual(other[0], 0) # unlocked work is not held by the locked queue + self.box.fake.units["wf-r-1.service"] = ("inactive", None) + self.box.wf("status") + self.assertEqual(self.box.ledger().get("r-2").state, "running") + + def test_two_queued_gates_start_one_at_a_time(self): + self.gate("gate a") + self.gate("gate b", "--queue") + self.gate("gate c", "--queue") + self.box.fake.units["wf-r-1.service"] = ("inactive", None) + self.box.wf("status") + led = self.box.ledger() + self.assertEqual([led.get(i).state for i in ("r-2", "r-3")], ["running", "queued"]) + + def test_explicit_lock_and_unlocked_titles(self): + self.box.wf("run", "--mem", "2G", "--for", "30m", "--title", "e2e", "--lock", "e2e", "--", "x") + self.assertEqual(self.box.wf("run", "--mem", "2G", "--for", "30m", "--title", "e2e 2", "--lock", "e2e", + "--", "x")[0], 3) + self.assertEqual(self.gate("gate a")[0], 0) # other key + self.assertEqual(self.box.wf("run", "--mem", "2G", "--for", "30m", "--title", "gatekeeper", "--", "x")[0], 0) + self.assertEqual(self.box.ledger().get("r-1").lock, f"e2e@{self.box.tmp / 'proj'}") + + def test_lock_scoped_to_main_tree(self): + self.gate("gate a") + self.box.fake.git["rev-parse"] = str(self.box.tmp / "other" / ".git") # another project's tree + self.assertEqual(self.gate("gate b")[0], 0) + self.box.fake.git["rev-parse"] = str(self.box.tmp / "proj" / ".git") # lane worktree of proj + self.assertEqual(self.gate("gate c")[0], 3) + + def test_project_is_main_tree_name_from_lane_worktree(self): + self.box.fake.git["rev-parse"] = str(self.box.tmp / "home" / ".git") # cwd = lane worktree 'proj' of home + self.box.wf("run", "--mem", "1G", "--for", "1m", "--title", "build 1", "--", "x") + self.box.fake.git["rev-parse"] = "" # no git → root name + self.box.wf("run", "--mem", "1G", "--for", "1m", "--title", "build 2", "--", "x") + self.box.fake.git["rev-parse"] = str(self.box.tmp / "home" / ".git" / "modules" / "m") # submodule → root + self.box.wf("run", "--mem", "1G", "--for", "1m", "--title", "build 3", "--", "x") + self.assertEqual([self.box.ledger().get(f"r-{i}").project for i in (1, 2, 3)], ["home", "proj", "proj"]) + + def test_percent_args_reach_systemd_verbatim(self): + # r-671 'fatal: ambiguous argument %s"': caller quoting, not wf; argv passes through untouched + self.box.wf("run", "--mem", "1G", "--for", "5m", "--title", "log", "--", "git", "log", "--format=%h %s", "HEAD") + self.assertEqual(self.box.fake.ran("systemd-run")[0][-4:], ["git", "log", "--format=%h %s", "HEAD"]) + + +def hist_rec(id, title="build 1234567", peak=1.0, minutes=10.0, project="proj", mem=4.0, est=40): + return {"id": id, "project": project, "title": title, "mem_gb": mem, "est_min": est, "rc": 0, + "peak_gb": peak, "min": minutes} + + +def seed_history(box, recs): + box.env.state.mkdir(parents=True, exist_ok=True) + (box.env.state / "resources-history.jsonl").write_text("".join(json.dumps(r) + "\n" for r in recs)) + + +class HistoryIO(IOBase): + def finish(self, id, peak_gb): + logs = self.box.env.state / "logs" + (logs / f"{id}.rc").write_text("0\n") + (logs / f"{id}.peak").write_text(str(int(peak_gb * 1024 ** 3)) + "\n") + self.box.fake.units[f"wf-{id}.service"] = ("inactive", None) + + def test_pruned_run_kept_in_history(self): + self.box.wf("run", "--mem", "4G", "--for", "40m", "--title", "build abc1234", "--", "make") + self.box.tick(12 * 60) + self.finish("r-1", 3.0) + self.box.wf("status") + hist = self.box.env.state / "resources-history.jsonl" + self.assertFalse(hist.exists()) # still in the ledger: not copied yet + self.box.tick(25 * 3600) + self.box.wf("status") + self.assertEqual(self.box.ledger().entries, []) + self.assertEqual([json.loads(x) for x in hist.read_text().splitlines()], [ + {"id": "r-1", "project": "proj", "title": "build abc1234", "mem_gb": 4.0, "est_min": 40, "rc": 0, + "peak_gb": 3.0, "min": 12.0}]) + + def test_history_trimmed(self): + seed_history(self.box, [hist_rec(f"r-{i}") for i in range(100, 100 + R.HIST_KEEP)]) + self.box.wf("run", "--mem", "4G", "--for", "40m", "--title", "build", "--", "make") + self.finish("r-1", 1.0) + self.box.wf("status") + self.box.tick(25 * 3600) + self.box.wf("status") + lines = (self.box.env.state / "resources-history.jsonl").read_text().splitlines() + self.assertEqual(len(lines), R.HIST_KEEP) + self.assertEqual((json.loads(lines[0])["id"], json.loads(lines[-1])["id"]), ("r-101", "r-1")) + + def test_run_hint_over_history(self): + # peaks 1,2,3 → p95 3 ×1.15 = 3.45 → 3.5 GB; durations 10,20,30 → p90 30 ×1.5 = 45 min + seed_history(self.box, [hist_rec("r-90", peak=1.0, minutes=10.0), hist_rec("r-91", peak=2.0, minutes=20.0), + hist_rec("r-92", title="build 89abcde (retry)", peak=3.0, minutes=30.0)]) + code, out, err = self.box.wf("run", "--mem", "8G", "--for", "40m", "--title", "build 7654321", "--", "make") + self.assertEqual((code, err), (0, "hint: history says ~3.5 GB / 45 min (3 runs)\n")) + code, out, err = self.box.wf("run", "--mem", "7G", "--for", "90m", "--title", "build", "--", "make") + self.assertEqual(err, "") + code, out, err = self.box.wf("run", "--mem", "1G", "--for", "91m", "--title", "build", "--", "make") + self.assertEqual(err, "hint: history says ~3.5 GB / 45 min (3 runs)\n") + code, out, err = self.box.wf("note", "--mem", "8G", "--for", "10m", "build") + self.assertEqual(err, ("hint: history says ~3.5 GB / 45 min (3 runs)\n")) + code, out, err = self.box.wf("run", "--mem", "8G", "--for", "40m", "--title", "other", "--", "make") + self.assertEqual(err, "") + + def test_hist_command(self): + seed_history(self.box, [hist_rec("r-90", peak=1.0, minutes=10.0), hist_rec("r-91", peak=2.0, minutes=20.0), + hist_rec("r-92", peak=3.0, minutes=30.0), hist_rec("r-93", project="zzz")]) + code, out, err = self.box.wf("hist") + self.assertEqual((code, err), (0, "")) + self.assertEqual(out, "proj · build · n 3 · req 4.0 GB · peak 2.0/3.0 GB · est 40m · dur 20m/30m · suggest 3.5 GB 45m\n" + "zzz · build · n 1 · req 4.0 GB · peak 1.0/1.0 GB · est 40m · dur 10m/10m · suggest - (< 3 runs)\n") + self.assertEqual(self.box.wf("hist", "--project", "zzz")[1].count("\n"), 1) diff --git a/tests/test_runner.py b/tests/test_runner.py new file mode 100644 index 0000000..a095f2b --- /dev/null +++ b/tests/test_runner.py @@ -0,0 +1,81 @@ +import sys +import unittest +from pathlib import Path + +HERE = Path(__file__).resolve().parent.parent +sys.path.insert(0, str(HERE)) +sys.path.insert(0, str(HERE / "tests")) + +import test_cli # noqa: E402 + +TASKS = """\ +# Tasks — demo + +## Awaiting your decision + +## Pending + +- **t-ok** [P1] (1h): Good. Headless. + Done: tests green. + +- **t-nodone** [P1] (1h): No done line. + +- **t-owner** [P1] (1h): Owner bound. + Done: tests green. + Sessions: owner + +- **t-confirm** [P1] (1h): Confirm. + Done: owner confirms it works. + +- **t-report** [P1] (1h): Report. + - Done: report to the owner. + +- **t-parent** [P1] (5h): Parent. + Done: all slices. + Slices: [[t-parent-1]] + +- **t-parent-1** [P1] (1h): Slice. + Done: ok. + Model: sonnet + +- **t-doneparent** [P1] (10h): Slices all archived. + Done: all slices. + - Slices: [[t-gone]] + +- **t-blk** [P1] (1h) (blocked: [[a-x]]): Blocked. + Done: ok. +""" + + +class Runner(test_cli.Cli): + tasks_text = TASKS + + def col(self, *args): + return [l.split()[0] for l in self.ok("list", *args).splitlines()[:-1]] + + def test_runner_filter(self): + self.assertEqual(self.col("--runner"), ["t-ok", "t-parent-1"]) + self.assertEqual(self.col("--runner", "--model", "sonnet"), ["t-parent-1"]) + + def test_add_hint_without_done(self): + code, out, err = self.wf("add", "-p", "2", "-e", "1h", "Plain thing", hints=True) + self.assertEqual((code, err), (0, "hint: no Done line; add one (--done) so runners can pick it\n")) + + def test_add_no_hint_with_done_or_awaiting(self): + code, out, err = self.wf("add", "-p", "2", "-e", "1h", "--body", "Body thing", stdin="Done: x\n", hints=True) + self.assertEqual((code, err), (0, "")) + code, out, err = self.wf("add", "-s", "awaiting", "Which one?", hints=True) + self.assertEqual((code, err), (0, "")) + + def test_add_done_flag_writes_done_line(self): + code, out, err = self.wf("add", "-p", "2", "-e", "1h", "--done", "gate ALL GREEN", "Flagged thing", hints=True) + self.assertEqual((code, err), (0, "")) + self.assertIn("\n Done: gate ALL GREEN\n", self.ok("show", "t-flagged-thing") + "\n") + + def test_add_p0_without_done_warns(self): + code, out, err = self.wf("add", "-p", "0", "-e", "1h", "Urgent thing", hints=True) + self.assertEqual((code, err), (0, 'wf: warning: P0 without Done line is not runner-pickable; use --done "<text>"\n')) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_search.py b/tests/test_search.py new file mode 100644 index 0000000..4c02520 --- /dev/null +++ b/tests/test_search.py @@ -0,0 +1,78 @@ +import sys +import unittest +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) +from wflib import search as S +from wflib import tasks as T + +TASKS = """\ +## Awaiting your decision + +## Pending + +- **t-guards** [P2] (1h): Guards notice lock picking. Port the notify function. + - Steps: reaction drop + Ref: docs/ai.md + +- **t-throw** [P2] (5h): Thrown weapons and grenades. NPCs throw items. + - Steps: guard against throwing into allies + +- **t-upgrade** [P3] (1h): Shop upgrenade typo. Fix spelling. + +## Needs human + +## Deferred +""" +ARCHIVE = "# Archive\n\n- 2026-09-01 **t-wear** Weapon wear — guards bust weapons\n- old entry about grenades\n" +DOCS = { + "docs/ai.md": "# AI notes\n\nIntro.\n\n## Guards and thieves\n\nText about catching.\n\n## Combat\n\nA guard attacks. Grenade use.\n", +} + + +def run(*words, **kw): + return S.search(list(words), T.parse(TASKS), ARCHIVE, DOCS, **kw) + + +class SearchTest(unittest.TestCase): + def test_title_outranks_body(self): + hits = run("guard", kinds={"task"}) + self.assertEqual([h.where for h in hits], ["t-guards", "t-throw"]) + self.assertEqual(hits[0].line, "Guards notice lock picking. Port the notify function.") + self.assertEqual(hits[1].line, "- Steps: guard against throwing into allies") + + def test_prefix_matches_word_start_only(self): + self.assertEqual([h.where for h in run("grenad", kinds={"task"})], ["t-throw"]) + + def test_all_words_outrank_one_strong_word(self): + hits = run("throwing", "allies", "guards", kinds={"task"}) + self.assertEqual([h.where for h in hits], ["t-throw", "t-guards"]) + + def test_kinds_and_tie_order(self): + hits = run("guard") + self.assertEqual([(h.kind, h.where) for h in hits], + [("task", "t-guards"), ("doc", "docs/ai.md:5"), ("archive", "archive:3"), + ("task", "t-throw"), ("doc", "docs/ai.md:9")]) + self.assertEqual(hits[1].label, "Guards and thieves") + self.assertEqual(hits[4].line, "A guard attacks. Grenade use.") + + def test_archive_filter(self): + hits = run("grenades", kinds={"archive"}) + self.assertEqual([(h.where, h.line) for h in hits], [("archive:4", "old entry about grenades")]) + + def test_limit(self): + self.assertEqual(len(run("guard", limit=2)), 2) + + def test_case_and_regex_chars(self): + self.assertEqual([h.where for h in run("GUARD", kinds={"task"})], ["t-guards", "t-throw"]) + self.assertEqual(run("a.i", "(x"), []) + + def test_id_matches(self): + self.assertEqual([h.where for h in run("t-throw", kinds={"task"})], ["t-throw"]) + + def test_no_words(self): + self.assertEqual(run(), []) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_sessions_field.py b/tests/test_sessions_field.py new file mode 100644 index 0000000..af1cb62 --- /dev/null +++ b/tests/test_sessions_field.py @@ -0,0 +1,197 @@ +import json +import os +import sys +import unittest +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) +from wflib import tasks as T +from wflib import lanes as L +from test_cli import Cli +from test_check import Base, CLEAN + +SESS = """\ +# Tasks — demo + +## Awaiting your decision + +## Pending + +- **t-solo** [P1] (1h): Solo. + - Sessions: solo — lib bump + Model: sonnet + +- **t-own** [P1] (1h, interactive): Own. + +- **t-owner** [P2] (1h): Owner. + Sessions: owner + +- **t-par** [P2] (<1h): Par. + Sessions: parallel + +- **t-plain** [P3] (<1h): Plain. + +## Needs human + +## Deferred +""" + + +class SessionsLineTest(unittest.TestCase): + def setUp(self): + self.doc = T.parse(SESS) + + def test_values(self): + self.assertEqual([(i.id, i.sessions) for i in self.doc.section("pending").items], + [("t-solo", "solo"), ("t-own", "owner"), ("t-owner", "owner"), + ("t-par", "parallel"), ("t-plain", "parallel")]) + + def test_set_replaces_in_place_and_keeps_tail_order(self): + T.set_fields(self.doc, "t-solo", sessions="owner") + self.assertEqual(self.doc.item("t-solo").body, [" Sessions: owner", " Model: sonnet"]) + + def test_set_adds_before_model(self): + T.set_fields(self.doc, "t-plain", sessions="solo") + self.assertEqual(self.doc.item("t-plain").body, [" Sessions: solo"]) + + def test_set_empty_removes(self): + T.set_fields(self.doc, "t-owner", sessions="") + self.assertEqual(self.doc.item("t-owner").body, []) + + def test_set_drops_interactive_flag(self): + T.set_fields(self.doc, "t-own", sessions="owner") + item = self.doc.item("t-own") + self.assertEqual(item.lines(), ["- **t-own** [P1] (1h): Own.", " Sessions: owner"]) + + def test_set_bad_value(self): + with self.assertRaises(T.TaskError): + T.set_fields(self.doc, "t-plain", sessions="many") + + def test_note_goes_before_sessions_line(self): + T.add_note(self.doc, "t-owner", "x") + self.assertEqual(self.doc.item("t-owner").body, [" - x", " Sessions: owner"]) + + +class SessionsPickTest(unittest.TestCase): + def setUp(self): + self.doc = T.parse(SESS) + + def pick(self, lane, model, **kw): + return L.pick(self.doc, set(), L.DEFAULT_LANES, "1h", lane, model, **kw) + + def test_alone_solo_picked_owner_skipped(self): + item, skipped = self.pick(None, "sonnet") + self.assertEqual(item.id, "t-solo") + item, skipped = self.pick("slow", "opus", others=1) + self.assertEqual((item.id, [(i.id, why) for i, why in skipped]), + ("t-par", [("t-solo", "solo: 1 other live session"), + ("t-own", "owner: needs the owner (wf next --owner)"), + ("t-owner", "owner: needs the owner (wf next --owner)")])) + + def test_owner_present(self): + item, _ = self.pick("slow", "opus", others=1, owner=True) + self.assertEqual(item.id, "t-own") + + def test_solo_skipped_with_other_live_sessions(self): + item, skipped = self.pick(None, "sonnet", others=1) + self.assertEqual((item, [(i.id, why) for i, why in skipped]), + (None, [("t-solo", "solo: 1 other live session")])) + _, skipped = self.pick(None, "sonnet", others=2) + self.assertEqual(skipped[0][1], "solo: 2 other live sessions") + + def test_solo_running(self): + self.assertEqual(T.solo_running(self.doc, {"t-solo": "sonnet session uds:/a"}), + ("t-solo", "sonnet session uds:/a")) + self.assertIsNone(T.solo_running(self.doc, {"t-par": "x"})) + self.assertIsNone(T.solo_running(self.doc, {})) + + def test_solo_done_block(self): + sessions = {"opus": {"socket": "/o", "alive": True, "pid": 1}, + "haiku": {"socket": "/h", "alive": False, "pid": 2}, + "sonnet": {"socket": "/s", "alive": True, "pid": 3}} + self.assertEqual(L.solo_done_block(["t-solo"], sessions, "3"), + ["notify opus uds:/o: solo t-solo done, run wf next"]) + self.assertEqual(L.solo_done_block([], sessions, "3"), []) + + +class SessionsCheckTest(Base): + def test_bad_word_error_interactive_warning(self): + text = CLEAN.replace("- **t-one** [P1] (1h)", "- **t-one** [P1] (1h, interactive)").replace( + "## Needs human", "- **t-x** [P3] (1h): X.\n Sessions: lots\n\n## Needs human") + self.tasks(text) + errors, warnings = self.run_check() + self.assertTrue(any(e.endswith("t-x: Sessions 'lots' (want parallel, solo, owner)") for e in errors), errors) + self.assertTrue(any(w.endswith("t-one: 'interactive' flag: write 'Sessions: owner' " + "(wf set t-one --sessions owner)") for w in warnings), warnings) + + +class SessionsCliTest(Cli): + tasks_text = SESS + + def env(self, name, pid): + sock = self.root / f"{name}.sock" + sock.write_text("") + return {"CLAUDE_CODE_MESSAGING_SOCKET": str(sock), "CLAUDE_PID": str(pid), "CLAUDE_CODE_SESSION_ID": name} + + def register(self, lane, model, sock, pid): + d = self.root / ".wf" / "sessions" + d.mkdir(parents=True, exist_ok=True) + (d / f"{lane}.json").write_text(json.dumps({"lane": lane, "model": model, "socket": str(sock), "pid": pid, + "session": "s", "at": "2026-10-04T10:00"})) + + def test_list_marks(self): + self.assertEqual(self.ok("list"), + "t-solo P1 1h - sonnet slow [solo] Solo\n" + "t-own P1 1h - opus slow [owner] Own\n" + "t-owner P2 1h - opus slow [owner] Owner\n" + "t-par P2 <1h - opus fast Par\n" + "t-plain P3 <1h - opus fast Plain\n" + "pending 5 · human 0 · awaiting 0 · deferred 0\n") + + def test_add_and_set(self): + self.ok("add", "Four.", "-p", "3", "-e", "1h", "--sessions", "solo", "--model", "haiku") + self.assertEqual(self.item("t-four"), "- **t-four** [P3] (1h): Four.\n Sessions: solo\n Model: haiku\n") + self.ok("set", "t-four", "--sessions", "") + self.assertEqual(self.item("t-four"), "- **t-four** [P3] (1h): Four.\n Model: haiku\n") + self.fails("set", "t-four", "--sessions", "many", code=2) + self.fails("add", "Five.", "-p", "3", "-e", "1h", "--sessions", "many", code=2) + + def test_interactive_option_is_old_spelling(self): + code, out, err = self.wf("add", "Four.", "-p", "3", "-e", "1h", "--interactive") + self.assertEqual((code, err), (0, "wf: --interactive is now --sessions owner\n")) + self.assertEqual(self.item("t-four"), "- **t-four** [P3] (1h): Four.\n Sessions: owner\n") + code, out, err = self.wf("set", "t-plain", "--interactive", "yes") + self.assertEqual((code, err), (0, "wf: --interactive is now --sessions owner\n")) + self.assertEqual(self.item("t-plain"), "- **t-plain** [P3] (<1h): Plain.\n Sessions: owner\n") + + def test_next_owner(self): + self.ok("done", "t-solo", "-m", "ok") + out = self.ok("next", "--lane", "slow", "--as", "opus", env={"CLAUDE_PID": "", "CLAUDE_CODE_MESSAGING_SOCKET": ""}) + self.assertIn("- t-own: owner: needs the owner (wf next --owner)\n", out) + self.assertIn("===== Next task =====\n- **t-par**", out) # slow empty → fallback fast + self.assertTrue(self.ok("next", "--lane", "slow", "--as", "opus", "--owner", "--brief").startswith("- **t-own**")) + + def test_next_solo_with_other_live_session(self): + sock = self.root / "o.sock" + sock.write_text("") + self.register("fast", "opus", sock, os.getppid()) + err = self.fails("next", "--as", "sonnet", env=self.env("s", os.getpid())) + self.assertEqual(err, "wf: nothing pickable for all lanes (sonnet) in Pending\n") + out = self.wf("next", "--as", "sonnet", env=self.env("s", os.getpid()))[1] + self.assertIn("- t-solo: solo: 1 other live session\n", out) + + def test_solo_in_progress_blocks_others_and_done_notifies(self): + s, o = self.env("s", os.getpid()), self.env("o", os.getppid()) + self.ok("next", "--lane", "slow", "--as", "sonnet", "--brief", env=s) + self.ok("status", "t-solo", "progress", "x", env=s) + code, out, err = self.wf("next", "--lane", "fast", "--as", "opus", env=o) + self.assertEqual((code, err), (1, f"wf: solo t-solo in progress by sonnet session uds:{self.root / 's.sock'}: " + "wait (its wf done notifies you)\n")) + self.assertNotIn("Next task", out) + out = self.ok("done", "t-solo", "-m", "ok", env=s) + self.assertIn(f"notify fast uds:{self.root / 'o.sock'}: solo t-solo done, run wf next\n", out) + self.assertTrue(self.ok("next", "--lane", "fast", "--as", "opus", "--brief", env=o).startswith("- **t-par**")) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_setup.py b/tests/test_setup.py new file mode 100644 index 0000000..ea44c33 --- /dev/null +++ b/tests/test_setup.py @@ -0,0 +1,164 @@ +import os +import sys +import tempfile +import unittest +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) +from wflib import config as C +from test_claims import FOUR, git +from test_cli import TOML, Cli + +SETUP = TOML + 'worktree_setup = ["mkdir -p out && ln -sfn \\"$WF_MAIN/out/data\\" out/data", "echo ran >> log.txt"]\n' + + +class SetupConfigTest(unittest.TestCase): + def setUp(self): + self.tmp = tempfile.TemporaryDirectory() + self.root = Path(self.tmp.name).resolve() + + def tearDown(self): + self.tmp.cleanup() + + def load(self, extra): + (self.root / "workflow.toml").write_text('format = 1\ntasks = "T.md"\narchive = "a.md"\n' + extra) + return C.load(self.root) + + def test_default_empty(self): + self.assertEqual(self.load("").worktree_setup, []) + + def test_list_of_commands(self): + self.assertEqual(self.load('worktree_setup = ["a", "b c"]\n').worktree_setup, ["a", "b c"]) + + def test_wrong_type(self): + with self.assertRaisesRegex(C.ConfigError, "'worktree_setup' must be a list of strings"): + self.load('worktree_setup = "a"\n') + + def test_unknown_key_still_errors(self): + with self.assertRaisesRegex(C.ConfigError, "unknown key 'worktree_setp'"): + self.load('worktree_setp = []\n') + + +class GitCli(Cli): + tasks_text = FOUR + toml = SETUP + + def setUp(self): + super().setUp() + git(self.root, "init", "-q", "-b", "master") + (self.root / ".gitignore").write_text(".worktrees/\n.wf/\nout/\n") + git(self.root, "add", "-A") + git(self.root, "commit", "-qm", "init") + + +class WorktreeCli(GitCli): + def setUp(self): + super().setUp() + self.wt = self.root / ".worktrees" / "opus" + git(self.root, "worktree", "add", "-q", str(self.wt), "-b", "opus/t-three") + (self.root / "out" / "data").mkdir(parents=True) + (self.root / "out" / "data" / "x.bin").write_text("main data\n") + + def setup_(self, *args, cwd=None): + return self.wf("setup", *args, project=False, cwd=cwd or self.wt) + + +class SetupCliTest(WorktreeCli): + def test_runs_commands_in_worktree_with_wf_main(self): + code, out, err = self.setup_() + self.assertEqual((code, err), (0, ""), out) + self.assertEqual((self.wt / "out" / "data" / "x.bin").read_text(), "main data\n") + self.assertEqual((self.wt / "log.txt").read_text(), "ran\n") + self.assertFalse((self.root / "log.txt").exists()) + self.assertIn("$ echo ran >> log.txt\n", out) + self.assertTrue(out.endswith("worktree_setup: 2 commands ok in .worktrees/opus\n"), out) + + def test_idempotent_rerun(self): + self.assertEqual(self.setup_()[0], 0) + self.assertEqual(self.setup_()[0], 0) + self.assertEqual((self.wt / "out" / "data" / "x.bin").read_text(), "main data\n") + + def test_from_subfolder_runs_at_worktree_project_root(self): + code, out, err = self.setup_(cwd=self.wt / "docs") + self.assertEqual((code, err), (0, ""), out) + self.assertTrue((self.wt / "log.txt").is_file()) + + def test_nonzero_exit_reported_and_rest_skipped(self): + (self.wt / "workflow.toml").write_text(TOML + 'worktree_setup = ["echo a > a.txt", "exit 3", "echo c > c.txt"]\n') + code, out, err = self.setup_() + self.assertEqual(code, 1, out + err) + self.assertEqual(err, "wf: worktree_setup 'exit 3' failed (exit 3): later commands skipped\n") + self.assertTrue((self.wt / "a.txt").is_file()) + self.assertFalse((self.wt / "c.txt").exists()) + + def test_outside_worktree_fails(self): + code, out, err = self.wf("setup", project=False, cwd=self.root) + self.assertEqual((code, err), (1, "wf: setup runs inside a linked git worktree (lane worktree)\n")) + + def test_none_configured(self): + (self.wt / "workflow.toml").write_text(TOML) + code, out, err = self.setup_() + self.assertEqual((code, out, err), (0, "no worktree_setup in workflow.toml: nothing to do\n", "")) + + +class WorktreeConfigTest(WorktreeCli): + """A lane worktree's own workflow.toml / area file: setup, gate, areas read it; --mark writes it.""" + + def edit_wt_toml(self, extra): + (self.wt / "workflow.toml").write_text(TOML + extra) + + def test_setup_and_gate_read_worktree_toml(self): + self.edit_wt_toml('worktree_setup = ["echo branch > b.txt"]\nquick_gate = ["echo g > g.txt"]\n') + code, out, err = self.setup_() + self.assertEqual((code, err), (0, ""), out) + self.assertEqual((self.wt / "b.txt").read_text(), "branch\n") + self.assertFalse((self.wt / "log.txt").exists()) + code, out, err = self.wf("gate", project=False, cwd=self.wt) + self.assertEqual((code, err), (0, ""), out) + self.assertTrue((self.wt / "g.txt").is_file()) + + def test_books_stay_main_tree(self): + self.edit_wt_toml("") + (self.wt / "TASKS.md").write_text("# other\n") + out = self.ok("show", "t-three", project=False, cwd=self.wt) + self.assertIn("t-three", out) + cfg = C.load_at(self.wt) + self.assertEqual((cfg.root, cfg.local, cfg.tasks), (self.root, self.wt, self.root / "TASKS.md")) + self.assertEqual(cfg.areas_file, self.wt / "CLAUDE.md") + + def test_areas_read_and_mark_worktree_copy(self): + notes = "## Areas\n### Core\n- Code map: `nothing_here`\n" + (self.root / "CLAUDE.md").write_text(notes.replace("Core", "Main")) + (self.wt / "CLAUDE.md").write_text(notes) + self.assertIn("Core: ", self.ok("areas", project=False, cwd=self.wt)) + self.ok("areas", "--mark", "Core", project=False, cwd=self.wt) + self.assertIn("- Checked: ", (self.wt / "CLAUDE.md").read_text()) + self.assertEqual((self.root / "CLAUDE.md").read_text(), notes.replace("Core", "Main")) + self.assertIn("Main: ", self.ok("areas", project=False, cwd=self.root)) + + def test_main_tree_ignores_worktree(self): + self.edit_wt_toml('quick_gate = ["exit 9"]\n') + self.assertIsNone(C.find_local(self.root)) + code, out, err = self.wf("gate", project=False, cwd=self.root) + self.assertEqual((code, out), (0, "no quick_gate in workflow.toml: nothing to do\n")) + + +class SetupHintTest(GitCli): + def env(self, name, pid): + sock = self.root / f"{name}.sock" + sock.write_text("") + return {"CLAUDE_CODE_MESSAGING_SOCKET": str(sock), "CLAUDE_PID": str(pid), "CLAUDE_CODE_SESSION_ID": name} + + def test_multi_session_hint_adds_setup(self): + son, me = self.env("s", os.getppid()), self.env("o", os.getpid()) + self.ok("next", "--lane", "slow", "--as", "sonnet", "--brief", env=son) + out = self.ok("next", "--lane", "fast", "--as", "opus", env=me) + self.assertIn(" git worktree add .worktrees/fast -b fast/<task> master && cd .worktrees/fast && wf setup" + " (wf there writes this TASKS.md)\n", out) + (self.root / ".worktrees" / "fast").mkdir(parents=True) + out = self.ok("next", "--lane", "fast", "--as", "opus", env=me) + self.assertIn(" cd .worktrees/fast && git switch -c fast/<task> master && wf setup" + " (wf there writes this TASKS.md)\n", out) + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_start.py b/tests/test_start.py new file mode 100644 index 0000000..e1a2d6e --- /dev/null +++ b/tests/test_start.py @@ -0,0 +1,109 @@ +import sys +import unittest +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) +from test_claims import git +from test_cli import WF +from test_setup import GitCli + +NONE = {"CLAUDE_CODE_MESSAGING_SOCKET": ""} + + +class StartTest(GitCli): + def setUp(self): + super().setUp() + self.wt = self.root / ".worktrees" / "slow" + + def start(self, *extra, code=0): + got, out, err = self.wf("start", "t-three", "--worktree", ".worktrees/slow", "--branch", "slow/t-three", + *extra, project=False, env=NONE) + self.assertEqual(got, code, out + err) + self.assertNotIn("Traceback", err) + return out, err + + def branch(self): + return (self.wt / ".git").is_file() and \ + subprocess_out(self.wt, "branch", "--show-current") + + def test_new_worktree_setup_progress_ctx_finish_line(self): + out, _ = self.start() + self.assertEqual(self.branch(), "slow/t-three") + self.assertEqual((self.wt / "log.txt").read_text(), "ran\n") # worktree_setup ran there + self.assertIn("**t-three** [P2] (<1h) (in progress: slow/t-three): Three.", self.tasks()) + self.assertIn("worktree: .worktrees/slow (new, branch slow/t-three from master)\n", out) + self.assertEqual(out.count("- **t-three**"), 1, out) + self.assertIn("worktree_setup: 2 commands ok in .worktrees/slow\n", out) + self.assertIn("- **t-three** [P2] (<1h) (in progress: slow/t-three): Three.\n\nSection: Pending\n", out) + self.assertTrue(out.endswith( + f"Finish (after the work; fill in the quoted parts and the paths):\n" + f" cd {self.wt} && (make test) && python3 {WF} finish t-three -m \"<entry>\" --commit \"<msg + footer>\" <paths>\n"), + out) + + def test_existing_branch_new_worktree_reports_wip(self): + git(self.root, "branch", "slow/t-three") + out, _ = self.start() + self.assertEqual(self.branch(), "slow/t-three") + self.assertIn("worktree: .worktrees/slow (new, existing branch slow/t-three: earlier WIP, read the notes)\n", + out) + + def test_existing_clean_worktree_switches(self): + git(self.root, "worktree", "add", "-q", "--detach", str(self.wt), "master") + out, _ = self.start() + self.assertEqual(self.branch(), "slow/t-three") + self.assertIn("worktree: .worktrees/slow (switched to new branch slow/t-three from master)\n", out) + (self.wt / "log.txt").unlink() # setup output, not ignored here + out, _ = self.start() # rerun: already there + self.assertIn("worktree: .worktrees/slow (on slow/t-three)\n", out) + + def test_dirty_worktree_refused_without_recovery(self): + git(self.root, "worktree", "add", "-q", "--detach", str(self.wt), "master") + (self.wt / "DESIGN.md").write_text("changed\n") + _, err = self.start(code=1) + self.assertEqual(err, "wf: worktree .worktrees/slow has uncommitted changes: M DESIGN.md " + "(hand back, or --recovery when a dead worker left them)\n") + self.assertNotIn("in progress", self.tasks()) + self.assertFalse((self.wt / "log.txt").exists()) + + def test_recovery_keeps_dirty_wip_and_shows_it(self): + git(self.root, "worktree", "add", "-q", str(self.wt), "-b", "slow/t-three") + (self.wt / "DESIGN.md").write_text("changed\n") + out, _ = self.start("--recovery") + self.assertEqual((self.wt / "DESIGN.md").read_text(), "changed\n") + self.assertIn("recovery: uncommitted:\n M DESIGN.md\n", out) + self.assertIn("recovery: commits master..HEAD: none\n", out) + self.assertIn("(in progress: slow/t-three)", self.tasks()) + + def test_recovery_dirty_on_other_branch_refused(self): + git(self.root, "worktree", "add", "-q", "--detach", str(self.wt), "master") + (self.wt / "DESIGN.md").write_text("changed\n") + _, err = self.start("--recovery", code=1) + self.assertIn("uncommitted changes on another branch", err) + + def test_setup_failure_stops_before_progress(self): + (self.root / "workflow.toml").write_text(self.toml.replace('"echo ran >> log.txt"', '"false"')) + git(self.root, "commit", "-qam", "red setup") + _, err = self.start(code=1) + self.assertIn("worktree_setup 'false' failed", err) + self.assertNotIn("in progress", self.tasks()) + + def test_unknown_id_creates_nothing(self): + got, out, err = self.wf("start", "t-nope", "--worktree", ".worktrees/slow", "--branch", "b", + project=False, env=NONE) + self.assertEqual(got, 1, out + err) + self.assertFalse(self.wt.exists()) + + def test_inside_worktree_refused(self): + git(self.root, "worktree", "add", "-q", "--detach", str(self.wt), "master") + got, out, err = self.wf("start", "t-three", "--worktree", str(self.wt), "--branch", "b", + project=False, cwd=self.wt, env=NONE) + self.assertEqual((got, err), (1, "wf: start runs in the main tree (it creates the worktree)\n")) + + +def subprocess_out(cwd, *args): + import subprocess + return subprocess.run(["git", "-C", str(cwd), *args], capture_output=True, text=True).stdout.strip() + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_tasks.py b/tests/test_tasks.py new file mode 100644 index 0000000..64f219c --- /dev/null +++ b/tests/test_tasks.py @@ -0,0 +1,523 @@ +import sys +import unittest +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) +from wflib import tasks as T +from wflib import lanes as L + +SAMPLE = """\ +# Tasks — demo + +Commands: `make test`. + +## Awaiting your decision + +- **a-smoke**: Smoke screenshots. Need you present. + +## Pending + +Intro prose. + +- **t-first** [P1] (<1h): First thing. Do it well. + - Steps: one + - After: [[t-zero]], [[t-other]] + Ref: DESIGN.md#terrain, docs/x.md (skill rows) + +- **t-second** [P2] (5h, interactive) (in progress: master): Second: with colon. Goal here. + +- **t-third** [P3] (1h) (blocked: [[a-smoke]]): Third. + +Trailing prose after items. + +## Needs human + +- **t-play** [P1] (<1h): Play-test feel. + - [ ] keyboard works + - [ ] motion smooth + +## Deferred + +## Known caveats + +- plain bullet, mentions [[t-first]] +""" + + +class HeaderTest(unittest.TestCase): + def setUp(self): + self.doc = T.parse(SAMPLE) + + def item(self, id): + section, i = self.doc.find(id) + return section.items[i] + + def test_task_with_priority_and_effort(self): + it = self.item("t-first") + self.assertEqual((it.id, it.prio, it.effort, it.interactive, it.status, it.text), + ("t-first", 1, "<1h", False, None, "First thing. Do it well.")) + + def test_interactive_and_in_progress(self): + it = self.item("t-second") + self.assertEqual((it.prio, it.effort, it.interactive, it.status), + (2, "5h", True, "in progress: master")) + + def test_blocked(self): + it = self.item("t-third") + self.assertEqual(it.status, "blocked: [[a-smoke]]") + self.assertEqual(it.blocked_on, "a-smoke") + self.assertIsNone(self.item("t-first").blocked_on) + + def test_awaiting_item_has_no_priority(self): + it = self.item("a-smoke") + self.assertEqual((it.prio, it.effort, it.text), (None, None, "Smoke screenshots. Need you present.")) + + def test_header_rebuilds_the_line(self): + self.assertEqual(self.item("t-first").header(), "- **t-first** [P1] (<1h): First thing. Do it well.") + self.assertEqual(self.item("t-second").header(), + "- **t-second** [P2] (5h, interactive) (in progress: master): Second: with colon. Goal here.") + self.assertEqual(self.item("t-third").header(), "- **t-third** [P3] (1h) (blocked: [[a-smoke]]): Third.") + self.assertEqual(self.item("a-smoke").header(), "- **a-smoke**: Smoke screenshots. Need you present.") + + def test_title_and_goal(self): + it = self.item("t-second") + self.assertEqual(it.title, "Second: with colon") + self.assertEqual(it.goal, "Goal here.") + self.assertEqual(self.item("t-third").title, "Third") + self.assertEqual(self.item("t-third").goal, "") + + def test_after(self): + self.assertEqual(self.item("t-first").after, ["t-zero", "t-other"]) + self.assertEqual(self.item("t-second").after, []) + + def test_refs(self): + self.assertEqual(self.item("t-first").refs, [("DESIGN.md", "terrain"), ("docs/x.md", None)]) + self.assertEqual(self.item("t-second").refs, []) + + def test_body_lines(self): + self.assertEqual(self.item("t-play").body, [" - [ ] keyboard works", " - [ ] motion smooth"]) + + +class StructureTest(unittest.TestCase): + def test_round_trip(self): + self.assertEqual(T.render(T.parse(SAMPLE)), SAMPLE) + + def test_sections_and_keys(self): + doc = T.parse(SAMPLE) + self.assertEqual([(s.heading, s.key) for s in doc.sections], + [("Awaiting your decision", "awaiting"), ("Pending", "pending"), + ("Needs human", "human"), ("Deferred", "deferred"), ("Known caveats", None)]) + self.assertEqual([i.id for i in doc.section("pending").items], ["t-first", "t-second", "t-third"]) + self.assertEqual(doc.section("pending").prefix, ["", "Intro prose.", ""]) + self.assertEqual(doc.section("pending").suffix, ["Trailing prose after items.", ""]) + self.assertEqual(doc.ids(), {"a-smoke", "t-first", "t-second", "t-third", "t-play"}) + + def test_prose_section_bullets_are_not_items(self): + doc = T.parse(SAMPLE) + self.assertEqual(doc.sections[-1].items, []) + self.assertEqual(doc.sections[-1].prefix, ["", "- plain bullet, mentions [[t-first]]"]) + + def test_missing_section_raises(self): + doc = T.parse("# x\n\n## Pending\n") + with self.assertRaisesRegex(T.TaskError, "no '## Deferred' section"): + doc.section("deferred") + + def test_find_unknown_id_names_nearest(self): + with self.assertRaisesRegex(T.TaskError, r"unknown id 't-frist' \(nearest: t-first"): + T.parse(SAMPLE).find("t-frist") + + def test_items_are_separated_by_one_blank_line(self): + text = "## Pending\n- **t-a** [P1] (1h): A.\n\n\n\n- **t-b** [P1] (1h): B.\n body\n- **t-c** [P1] (1h): C.\n" + self.assertEqual(T.render(T.parse(text)), + "## Pending\n- **t-a** [P1] (1h): A.\n\n- **t-b** [P1] (1h): B.\n body\n\n- **t-c** [P1] (1h): C.\n") + + def test_malformed_header_is_kept_and_flagged(self): + text = "## Pending\n\n- **t-x** [P9] (1h) no colon\n body\n\n- **t-ok** [P1] (1h): Fine.\n" + doc = T.parse(text) + bad = doc.section("pending").items[0] + self.assertEqual(bad.raw, "- **t-x** [P9] (1h) no colon") + self.assertIn("header", bad.error) + self.assertEqual(bad.id, "t-x") + self.assertEqual(T.render(doc), text) + + def test_flush_left_line_ends_the_items_and_survives(self): + text = ("## Pending\n\n- **t-a** [P1] (1h): A.\n body\nstray continuation\n" + "- **t-b** [P1] (1h): B.\n\n## Deferred\n") + doc = T.parse(text) + pending = doc.section("pending") + self.assertEqual([i.id for i in pending.items], ["t-a"]) + self.assertEqual(pending.suffix, ["stray continuation", "- **t-b** [P1] (1h): B.", ""]) + self.assertEqual(T.render(doc), + "## Pending\n\n- **t-a** [P1] (1h): A.\n body\n\nstray continuation\n" + "- **t-b** [P1] (1h): B.\n\n## Deferred\n") + + def test_crlf_is_kept(self): + text = "## Pending\r\n\r\n- **t-a** [P1] (1h): A.\r\n body\r\n" + doc = T.parse(text) + self.assertEqual(doc.newline, "\r\n") + self.assertEqual(doc.section("pending").items[0].body, [" body"]) + self.assertEqual(T.render(doc), text) + + def test_missing_final_newline_is_added(self): + self.assertEqual(T.render(T.parse("## Pending\n\n- **t-a** [P1] (1h): A.")), + "## Pending\n\n- **t-a** [P1] (1h): A.\n") + + +if __name__ == "__main__": + unittest.main() + + +EDIT = """\ +## Awaiting your decision + +- **a-key**: Key needed. Which one? + +## Pending + +- **t-p0** [P0] (1h): Zero. + +- **t-p1** [P1] (1h): One. + - Steps: a + - After: [[t-p0]] + Ref: docs/a.md + +- **t-p2** [P2] (1h) (blocked: [[a-key]]): Two. + +- **t-p3** [P3] (1h): Three. + - After: [[t-gone]] + +## Needs human + +- **t-play** [P1] (<1h): Play. + - [ ] keyboard works + - [x] motion smooth + - [ ] gait feels right + +## Deferred +""" + + +def ids(doc, key): + return [i.id for i in doc.section(key).items] + + +class MakeIdTest(unittest.TestCase): + def test_slug_of_title(self): + self.assertEqual(T.make_id("Guards notice lock picking and busting", set()), + "t-guards-notice-lock-picking-and-busting") + + def test_long_title_is_cut_at_a_word(self): + self.assertEqual(T.make_id("Exe function map plus port re-verification of everything", set()), + "t-exe-function-map-plus-port-re") + self.assertEqual(T.make_id("Thrown weapons and grenades for everyone here", set()), + "t-thrown-weapons-and-grenades-for-everyone") + + def test_taken_id_gets_a_number(self): + self.assertEqual(T.make_id("Cave seams", {"t-cave-seams"}), "t-cave-seams-2") + self.assertEqual(T.make_id("Cave seams", {"t-cave-seams", "t-cave-seams-2"}), "t-cave-seams-3") + + def test_empty_slug_is_refused(self): + with self.assertRaisesRegex(T.TaskError, "--id"): + T.make_id("???", set()) + + def test_non_ascii_is_dropped(self): + self.assertEqual(T.make_id("Åäö test", set()), "t-test") + + def test_awaiting_prefix(self): + self.assertEqual(T.make_id("Smoke screenshots", set(), "a-"), "a-smoke-screenshots") + + def test_slice_id(self): + self.assertEqual(T.slice_id("t-map", {"t-map"}), "t-map-1") + self.assertEqual(T.slice_id("t-map", {"t-map", "t-map-1", "t-map-2"}), "t-map-3") + + +class EditTest(unittest.TestCase): + def setUp(self): + self.doc = T.parse(EDIT) + + def new(self, id, prio, body=()): + return T.Item(id=id, prio=prio, effort="1h", text="New.", body=list(body)) + + def test_insert_by_priority(self): + T.insert(self.doc, self.new("t-n", 1), "pending") + self.assertEqual(ids(self.doc, "pending"), ["t-p0", "t-p1", "t-n", "t-p2", "t-p3"]) + + def test_insert_into_empty_section(self): + T.insert(self.doc, self.new("t-n", 0), "deferred") + self.assertEqual(ids(self.doc, "deferred"), ["t-n"]) + + def test_insert_goes_after_its_dependency(self): + T.insert(self.doc, self.new("t-n", 1, [" - After: [[t-p3]]"]), "pending") + self.assertEqual(ids(self.doc, "pending"), ["t-p0", "t-p1", "t-p2", "t-p3", "t-n"]) + + def test_insert_refuses_taken_id(self): + with self.assertRaisesRegex(T.TaskError, "'t-p1' already exists"): + T.insert(self.doc, self.new("t-p1", 1), "pending") + + def test_insert_refuses_wrong_kind_for_section(self): + with self.assertRaisesRegex(T.TaskError, "a- items belong in Awaiting"): + T.insert(self.doc, T.Item(id="a-x", text="X."), "pending") + with self.assertRaisesRegex(T.TaskError, "needs a priority"): + T.insert(self.doc, T.Item(id="t-x", effort="1h", text="X."), "pending") + + def test_remove(self): + item = T.remove(self.doc, "t-p1") + self.assertEqual(item.body, [" - Steps: a", " - After: [[t-p0]]", " Ref: docs/a.md"]) + self.assertEqual(ids(self.doc, "pending"), ["t-p0", "t-p2", "t-p3"]) + + def test_remove_unknown_names_nearest(self): + with self.assertRaisesRegex(T.TaskError, r"unknown id 't-p9' \(nearest: t-p"): + T.remove(self.doc, "t-p9") + + def test_set_prio_moves_the_item_and_keeps_its_body(self): + T.set_prio(self.doc, "t-p1", 3) + self.assertEqual(ids(self.doc, "pending"), ["t-p0", "t-p2", "t-p3", "t-p1"]) + item = self.doc.item("t-p1") + self.assertEqual((item.prio, item.body[0]), (3, " - Steps: a")) + + def test_set_prio_range(self): + with self.assertRaisesRegex(T.TaskError, "priority 0-3"): + T.set_prio(self.doc, "t-p1", 4) + + def test_move_to_section(self): + T.move_to(self.doc, "t-p3", "deferred") + self.assertEqual(ids(self.doc, "pending"), ["t-p0", "t-p1", "t-p2"]) + self.assertEqual(ids(self.doc, "deferred"), ["t-p3"]) + + def test_move_to_refuses_kind_change(self): + with self.assertRaisesRegex(T.TaskError, "a- items belong in Awaiting"): + T.move_to(self.doc, "a-key", "pending") + + def test_move_rel_refuses_priority_break(self): + with self.assertRaisesRegex(T.TaskError, "breaks priority order.*--force"): + T.move_rel(self.doc, "t-p3", "t-p0", before=True) + self.assertEqual(ids(self.doc, "pending"), ["t-p0", "t-p1", "t-p2", "t-p3"]) + + def test_move_rel_with_force(self): + T.move_rel(self.doc, "t-p3", "t-p0", before=True, force=True) + self.assertEqual(ids(self.doc, "pending"), ["t-p3", "t-p0", "t-p1", "t-p2"]) + + def test_move_rel_never_puts_an_item_before_its_dependency(self): + with self.assertRaisesRegex(T.TaskError, "'t-p1' is After: \\[\\[t-p0\\]\\]"): + T.move_rel(self.doc, "t-p1", "t-p0", before=True, force=True) + + def test_move_rel_after_into_other_section(self): + T.move_rel(self.doc, "t-p1", "t-play", before=False) + self.assertEqual(ids(self.doc, "human"), ["t-play", "t-p1"]) + + def test_status(self): + T.set_status(self.doc, "t-p1", "in progress: feature/x") + self.assertEqual(self.doc.item("t-p1").header(), "- **t-p1** [P1] (1h) (in progress: feature/x): One.") + T.set_status(self.doc, "t-p1", "blocked: [[a-key]]") + self.assertEqual(self.doc.item("t-p1").blocked_on, "a-key") + T.set_status(self.doc, "t-p1", None) + self.assertEqual(self.doc.item("t-p1").header(), "- **t-p1** [P1] (1h): One.") + + def test_status_blocked_on_unknown_awaiting_item(self): + with self.assertRaisesRegex(T.TaskError, "'a-none' is not an open Awaiting item"): + T.set_status(self.doc, "t-p1", "blocked: [[a-none]]") + + def test_set_fields_header(self): + T.set_fields(self.doc, "t-p1", title="Uno", effort="5h", interactive=True) + self.assertEqual(self.doc.item("t-p1").header(), "- **t-p1** [P1] (5h, interactive): Uno.") + T.set_fields(self.doc, "t-p0", title="Nil") + self.assertEqual(self.doc.item("t-p0").text, "Nil.") + + def test_set_fields_title_keeps_goal(self): + doc = T.parse("## Pending\n\n- **t-a** [P1] (1h): Old name. The goal. More.\n") + T.set_fields(doc, "t-a", title="New name") + self.assertEqual(doc.item("t-a").text, "New name. The goal. More.") + + def test_set_fields_full_text_replaces_goal(self): + doc = T.parse("## Awaiting your decision\n\n- **a-q** : Colour. OK?\n\n## Pending\n\n" + "- **t-a** [P1] (1h): Old name. Old goal.\n".replace(" :", ":")) + T.set_fields(doc, "a-q", title="Colour red. OK?") + self.assertEqual(doc.item("a-q").text, "Colour red. OK?") + T.set_fields(doc, "t-a", title="New name. New goal") + self.assertEqual(doc.item("t-a").text, "New name. New goal.") + T.set_fields(doc, "t-a", title="Why not?") + self.assertEqual(doc.item("t-a").text, "Why not?") + + def test_set_fields_bad_effort(self): + with self.assertRaisesRegex(T.TaskError, "effort '2h'"): + T.set_fields(self.doc, "t-p1", effort="2h") + + def test_set_after(self): + T.set_fields(self.doc, "t-p1", after=[]) + self.assertEqual(self.doc.item("t-p1").body, [" - Steps: a", " Ref: docs/a.md"]) + T.set_fields(self.doc, "t-p1", after=["t-p0", "t-p2"]) + self.assertEqual(self.doc.item("t-p1").body, + [" - Steps: a", " - After: [[t-p0]], [[t-p2]]", " Ref: docs/a.md"]) + + def test_set_done(self): + doc = T.parse("## Pending\n\n- **t-a** [P1] (1h): A.\n - Steps: a\n - note\n Model: sonnet\n Ref: x.md\n") + T.set_fields(doc, "t-a", done="a works") + self.assertEqual(doc.item("t-a").body, + [" - Steps: a", " - note", " Done: a works", " Model: sonnet", " Ref: x.md"]) + T.set_fields(doc, "t-a", done=" b works ") + self.assertEqual(doc.item("t-a").body[2], " Done: b works") + T.set_fields(doc, "t-a", done="") + self.assertEqual(doc.item("t-a").body, [" - Steps: a", " - note", " Model: sonnet", " Ref: x.md"]) + + def test_set_refs(self): + T.set_fields(self.doc, "t-p1", refs=["docs/b.md#x (rows)", "DESIGN.md"]) + self.assertEqual(self.doc.item("t-p1").body[-1], " Ref: docs/b.md#x (rows), DESIGN.md") + T.set_fields(self.doc, "t-p0", refs=["docs/a.md"]) + self.assertEqual(self.doc.item("t-p0").body, [" Ref: docs/a.md"]) + T.set_fields(self.doc, "t-p0", refs=[]) + self.assertEqual(self.doc.item("t-p0").body, []) + + def test_add_note_goes_before_after_and_ref(self): + T.add_note(self.doc, "t-p1", "found the cause") + self.assertEqual(self.doc.item("t-p1").body, + [" - Steps: a", " - found the cause", " - After: [[t-p0]]", " Ref: docs/a.md"]) + + def test_set_body_keeps_after_and_ref(self): + T.set_body(self.doc, "t-p1", ["- Steps: b", " - sub", "", "- Done: c"]) + self.assertEqual(self.doc.item("t-p1").body, + [" - Steps: b", " - sub", "", " - Done: c", " - After: [[t-p0]]", " Ref: docs/a.md"]) + + def test_set_body_bare_after_ids_become_links(self): + T.set_body(self.doc, "t-p1", ["- Done: c", "- After: t-a, [[t-b]]"]) + self.assertEqual(self.doc.item("t-p1").body[:2], [" - Done: c", " - After: [[t-a]], [[t-b]]"]) + + def test_tick_by_number_and_text(self): + self.assertEqual(T.tick(self.doc, "t-play", "1"), "keyboard works") + self.assertEqual(T.tick(self.doc, "t-play", "gait"), "gait feels right") + self.assertEqual(self.doc.item("t-play").body, + [" - [x] keyboard works", " - [x] motion smooth", " - [x] gait feels right"]) + + def test_tick_errors(self): + with self.assertRaisesRegex(T.TaskError, "already ticked"): + T.tick(self.doc, "t-play", "2") + with self.assertRaisesRegex(T.TaskError, "no box matches 'jump'"): + T.tick(self.doc, "t-play", "jump") + with self.assertRaisesRegex(T.TaskError, "no box 7"): + T.tick(self.doc, "t-play", "7") + + def test_unblock(self): + self.assertEqual(T.unblock(self.doc, "a-key"), ["t-p2"]) + self.assertIsNone(self.doc.item("t-p2").status) + + +class PickTest(unittest.TestCase): + def test_skips_blocked_and_open_dependencies(self): + doc = T.parse(EDIT) + T.remove(doc, "t-p0") + T.remove(doc, "t-p1") + item, skipped = L.pick(doc, set(), L.DEFAULT_LANES, "1h") + self.assertIsNone(item) + self.assertEqual([(i.id, why) for i, why in skipped], + [("t-p2", "blocked: a-key"), ("t-p3", "after: t-gone")]) + + def test_archived_dependency_counts_as_done(self): + doc = T.parse(EDIT) + T.remove(doc, "t-p0") + T.remove(doc, "t-p1") + item, skipped = L.pick(doc, {"t-gone"}, L.DEFAULT_LANES, "1h") + self.assertEqual(item.id, "t-p3") + self.assertEqual([i.id for i, _ in skipped], ["t-p2"]) + + def test_first_item_when_free(self): + item, skipped = L.pick(T.parse(EDIT), set(), L.DEFAULT_LANES, "1h") + self.assertEqual((item.id, skipped), ("t-p0", [])) + + def test_open_dependency_in_file(self): + doc = T.parse(EDIT) + T.move_to(doc, "t-p0", "deferred") + item, skipped = L.pick(doc, set(), L.DEFAULT_LANES, "1h") + self.assertIsNone(item) + self.assertEqual(skipped[0][1], "after: t-p0") + + +class SliceTest(unittest.TestCase): + def test_open_slices(self): + doc = T.parse("## Pending\n\n- **t-p** [P1] (10h): Parent.\n\n- **t-p-2** [P1] (1h): Two.\n\n" + "- **t-px-1** [P1] (1h): Other.\n") + self.assertEqual(T.open_slices(doc, "t-p"), ["t-p-2"]) + + def test_open_slices_by_name(self): + doc = T.parse(SLICED) + self.assertEqual(T.open_slices(doc, "t-p"), ["t-map", "t-p-3"]) + self.assertEqual(T.parent_of(doc, "t-map"), "t-p") + self.assertEqual(T.parent_of(doc, "t-p-3"), "t-p") + self.assertIsNone(T.parent_of(doc, "t-q")) + + def test_pick_skips_parent_with_open_slices(self): + item, skipped = L.pick(T.parse(SLICED), {"t-done"}, L.DEFAULT_LANES, "1h") + self.assertEqual(item.id, "t-q") + self.assertEqual([(i.id, why) for i, why in skipped], + [("t-map", "after: t-x"), ("t-p-3", "after: t-map"), ("t-p", "open slices: t-map, t-p-3")]) + + +class RenameTest(unittest.TestCase): + def test_rename_changes_id_and_links(self): + doc = T.parse(SLICED) + n = T.rename(doc, "t-map", "t-map-scaffold", taken={"t-old"}) + self.assertEqual(n, 2) + self.assertEqual(T.render(doc), SLICED.replace("t-map", "t-map-scaffold")) + + def test_rename_awaiting_updates_blocked_status(self): + doc = T.parse("## Awaiting your decision\n\n- **a-a-foo**: Foo?\n\n## Pending\n\n" + "- **t-a** [P1] (1h) (blocked: [[a-a-foo]]): A.\n") + self.assertEqual(T.rename(doc, "a-a-foo", "a-foo", taken=set()), 1) + self.assertEqual(doc.item("t-a").blocked_on, "a-foo") + + def test_rename_refusals(self): + doc = T.parse(SLICED) + for new, err in [("t-q", "id 't-q' already exists"), ("t-old", "id 't-old' already used in the archive"), + ("a-map", "a rename keeps the kind"), ("t-Bad", "bad id 't-Bad'")]: + with self.assertRaisesRegex(T.TaskError, err): + T.rename(doc, "t-map", new, taken={"t-old"}) + with self.assertRaisesRegex(T.TaskError, "unknown id 't-none'"): + T.rename(doc, "t-none", "t-x", taken=set()) + self.assertEqual(T.render(doc), SLICED) + + +SLICED = """\ +## Pending + +- **t-p** [P1] (10h): Parent. + - Slices: [[t-done]], [[t-map]], [[t-p-3]] + +- **t-map** [P1] (1h): Map. + - After: [[t-x]] + +- **t-p-3** [P1] (1h): Three. + - After: [[t-map]] + +- **t-q** [P2] (1h): Free. +""" + + +class ArchiveTest(unittest.TestCase): + def test_line(self): + item = T.Item(id="t-x", prio=1, effort="1h", text="Cave seams. Close the slit.") + self.assertEqual(T.archive_line("2026-09-29", item, "closed with side walls"), + "- 2026-09-29 **t-x** Cave seams — closed with side walls") + + def test_prepend(self): + self.assertEqual(T.archive_prepend("# Archive (newest first)\n\n- old one\n- older\n", "- new"), + "# Archive (newest first)\n\n- new\n- old one\n- older\n") + + def test_prepend_to_empty_list(self): + self.assertEqual(T.archive_prepend("# Archive (newest first)\n", "- new"), + "# Archive (newest first)\n\n- new\n") + + def test_ids(self): + text = "# A\n\n- 2026-09-29 **t-x** X — done\n- old line without id\n- 2026-09-01 **t-y-2** Y — z\n" + self.assertEqual(T.archive_ids(text), {"t-x", "t-y-2"}) + + +class ParseBlockTest(unittest.TestCase): + def test_block_is_reindented(self): + item = T.parse_block("- **t-n** [P2] (1h): New. Goal.\n - Steps: a\n - sub\n Ref: docs/a.md\n") + self.assertEqual(item.lines(), ["- **t-n** [P2] (1h): New. Goal.", " - Steps: a", " - sub", " Ref: docs/a.md"]) + + def test_old_numbered_format_is_refused(self): + with self.assertRaisesRegex(T.TaskError, "old numbered format"): + T.parse_block("3. **[P2] Old** (Effort: 1h) — goal.") + + def test_bad_header_is_refused(self): + with self.assertRaisesRegex(T.TaskError, "bad header"): + T.parse_block("- **t-n** no colon") diff --git a/tests/test_usage.py b/tests/test_usage.py new file mode 100644 index 0000000..f2d55e3 --- /dev/null +++ b/tests/test_usage.py @@ -0,0 +1,337 @@ +import json +import os +import subprocess +import sys +import tempfile +import unittest +from pathlib import Path + +HERE = Path(__file__).resolve().parent.parent +WF = HERE / "wf.py" +sys.path.insert(0, str(HERE)) + +from wflib import usage # noqa: E402 + + +def entry(mid, req, model, ts, inp=0, cw5=0, cw1h=0, cr=0, out=0, type="assistant", stop=None, content=None, + uuid=None): + u = {"input_tokens": inp, "cache_creation_input_tokens": cw5 + cw1h, "cache_read_input_tokens": cr, + "output_tokens": out} + if cw1h: + u["cache_creation"] = {"ephemeral_5m_input_tokens": cw5, "ephemeral_1h_input_tokens": cw1h} + d = {"type": type, "requestId": req, "timestamp": ts, + "message": {"id": mid, "model": model, "role": "assistant", "usage": u}} + if stop: + d["message"]["stop_reason"] = stop + if content is not None: + d["message"]["content"] = content + if uuid: + d["uuid"] = uuid + return json.dumps(d) + + +# Subagent transcripts: stop_reason always null, output_tokens = message_start placeholder. +# e1: thinking (signature 1800 chars) + text 1000 chars + tool_use (input json 100 chars), placeholder 8. +# by hand: 0.3*(1800-800) + 0.3*1000 + (30 + 0.44*100) = 300 + 300 + 74 = 674 +# e2: same tool_use again but stop_reason set (final): reported 50 is trusted. +# e3: no content, placeholder 9 kept (nothing to estimate). +TOOL = {"type": "tool_use", "id": "t1", "name": "Bash", "input": {"a": "y" * 91}} # json.dumps → 100 chars +SUB = "\n".join([ + entry("e1", "q1", "claude-sonnet-5-5", "2026-10-04T10:00:00Z", out=8, uuid="u1", + content=[{"type": "thinking", "thinking": "", "signature": "z" * 1800}]), + entry("e1", "q1", "claude-sonnet-5-5", "2026-10-04T10:00:01Z", out=8, uuid="u2", + content=[{"type": "text", "text": "x" * 1000}]), + entry("e1", "q1", "claude-sonnet-5-5", "2026-10-04T10:00:01Z", out=8, uuid="u2", + content=[{"type": "text", "text": "x" * 1000}]), # duplicated line: counted once + entry("e1", "q1", "claude-sonnet-5-5", "2026-10-04T10:00:02Z", out=8, uuid="u3", content=[TOOL]), + entry("e2", "q2", "claude-sonnet-5-5", "2026-10-04T10:01:00Z", out=50, stop="tool_use", content=[TOOL]), + entry("e3", "q3", "claude-sonnet-5-5", "2026-10-04T10:02:00Z", out=9), +]) + "\n" + + +# main session, opus 5.5: m1 streamed as 2 entries (same usage), m2 as 2 entries (out 4000 then 10000). +# by hand: m1 = 1000*4 + 100000*5 + 2000*20 = 544000 µ$; m2 = 50000*8 + 1e6*0.2 + 10000*20 = 800000 µ$ +MAIN = "\n".join([ + json.dumps({"type": "user", "timestamp": "2026-10-04T09:00:00Z", "message": {"role": "user", "content": "hi"}}), + entry("m1", "r1", "claude-opus-5-5", "2026-10-04T09:00:01Z", inp=1000, cw5=100000, out=2000), + entry("m1", "r1", "claude-opus-5-5", "2026-10-04T09:00:02Z", inp=1000, cw5=100000, out=2000), + "not json", + json.dumps({"type": "attachment", "timestamp": "2026-10-04T09:00:03Z"}), + entry("m2", "r2", "claude-opus-5-5", "2026-10-04T09:01:00Z", cw1h=50000, cr=1000000, out=4000), + entry("m2", "r2", "claude-opus-5-5", "2026-10-04T09:01:05Z", cw1h=50000, cr=1000000, out=10000), + entry("m9", "r9", "<synthetic>", "2026-10-04T09:02:00Z", out=5), +]) + "\n" + +# subagent, sonnet 5.5: s1 = 10*2 + 20000*2.5 + 30000*0.2 + 500*10 = 61020 µ$; s2 = 50000*0.2 + 1500*10 = 25000 µ$ +AGENT = "\n".join([ + entry("s1", "q1", "claude-sonnet-5-5", "2026-10-04T10:00:00Z", inp=10, cw5=20000, cr=30000, out=500), + entry("s2", "q2", "claude-sonnet-5-5", "2026-10-04T12:00:00Z", cr=50000, out=1500), + entry("s3", "q3", "claude-mystery-1", "2026-10-04T12:30:00Z", inp=7, out=3), +]) + "\n" + +SID = "11111111-2222-3333-4444-555555555555" + + +class Parse(unittest.TestCase): + def test_dedupes_streamed_entries_and_takes_last_output(self): + got = usage.parse(MAIN) + self.assertEqual(list(got), ["claude-opus-5-5"]) + u = got["claude-opus-5-5"] + self.assertEqual((u.turns, u.inp, u.cw5, u.cw1h, u.cr, u.out), (2, 1000, 100000, 50000, 1000000, 12000)) + self.assertEqual(u.cw, 150000) + self.assertAlmostEqual(usage.cost("claude-opus-5-5", u), 1.344) + + def test_models_kept_apart_unknown_price_none(self): + got = usage.parse(AGENT) + s = got["claude-sonnet-5-5"] + self.assertEqual((s.turns, s.cr, s.out), (2, 80000, 2000)) + self.assertAlmostEqual(usage.cost("claude-sonnet-5-5", s), 0.08602) + self.assertIsNone(usage.cost("claude-mystery-1", got["claude-mystery-1"])) + + def test_since_until_filter_on_utc_timestamp(self): + s = usage.parse(AGENT, since="2026-10-04T11:00", until="2026-10-04T12:10")["claude-sonnet-5-5"] + self.assertEqual((s.turns, s.cr), (1, 50000)) + self.assertAlmostEqual(usage.cost("claude-sonnet-5-5", s), 0.025) + + def test_dated_model_id_priced_by_prefix(self): + u = usage.Usage(turns=1, inp=1000000, out=1000000) + self.assertAlmostEqual(usage.cost("claude-haiku-4-5-20251001", u), 6.0) + self.assertAlmostEqual(usage.cost("claude-opus-5", u), 30.0) + + def test_subagent_output_estimated_from_content(self): + u = usage.parse(SUB)["claude-sonnet-5-5"] + self.assertEqual((u.turns, u.out, u.est), (3, 674 + 50 + 9, 1)) + + def test_log_line_marks_estimate(self): + line = usage.log_line("T", "p", "t-x", "1h", "done", "a1", usage.parse(SUB)) + self.assertIn(" out=733 est=1 usd=", line) + self.assertNotIn(" est=", usage.log_line("T", "p", "t-x", "1h", "done", "a1", usage.parse(AGENT))) + + def test_log_line_lane_and_model(self): + by = usage.parse(SUB) + self.assertIn(" lane=fast model=sonnet effort=", usage.log_line("T", "p", "t-x", "1h", "done", "a1", by, lane="fast")) + self.assertIn(" lane=sonnet model=sonnet effort=", usage.log_line("T", "p", "t-x", "1h", "done", "a1", by)) + + def test_report_lane_model_key(self): + es = [{"lane": "fast", "model": "sonnet", "outcome": "done", "usd": 1.0, "turns": 10, "effort": "1h"}, + {"lane": "fast", "model": "opus", "outcome": "done", "usd": 2.0, "turns": 20, "effort": "1h"}, + {"lane": "opus", "outcome": "done", "usd": 3.0, "turns": 30, "effort": "1h"}] + self.assertEqual(usage.report(es, ("1h",)), [ + ("fast/opus", "all", 1, 1, 0, 2.0, 2.0, 2.0, 20, None), ("fast/opus", "1h", 1, 1, 0, 2.0, 2.0, 2.0, 20, None), + ("fast/sonnet", "all", 1, 1, 0, 1.0, 1.0, 1.0, 10, None), ("fast/sonnet", "1h", 1, 1, 0, 1.0, 1.0, 1.0, 10, None), + ("opus", "all", 1, 1, 0, 3.0, 3.0, 3.0, 30, None), ("opus", "1h", 1, 1, 0, 3.0, 3.0, 3.0, 30, None)]) + + def test_fmt(self): + self.assertEqual([usage.fmt(n) for n in (999, 1500, 8948413)], ["999", "1.5k", "8.95M"]) + + +def tcall(mid, ctx, name, inp, tid): + return json.dumps({"type": "assistant", "timestamp": "2026-10-04T10:00:00Z", "message": { + "id": mid, "model": "claude-sonnet-5-5", "usage": {"input_tokens": 10, "cache_read_input_tokens": ctx - 10}, + "content": [{"type": "tool_use", "id": tid, "name": name, "input": inp}]}}) + + +def tres(tid, text): + return json.dumps({"type": "user", "message": {"role": "user", "content": [ + {"type": "tool_result", "tool_use_id": tid, "content": text}]}}) + + +# agent A: ctx 1000 Read(400 chars=100 tok), 2000 grep(800=200), 3000 Edit, 4000 git commit. +# exp 3000 / mut 3000 / book 4000 of 10000; first edit after 2 calls, pre 3000 = 30%, context +2000 +EXA = "\n".join([ + tcall("a1", 1000, "Read", {"file_path": "/w/proj/.worktrees/x/src/a.py"}, "t1"), tres("t1", "x" * 400), + tcall("a2", 2000, "Bash", {"command": "grep -n foo src/a.py 2>/dev/null"}, "t2"), tres("t2", "y" * 800), + tcall("a3", 3000, "Edit", {"file_path": "/w/proj/src/a.py"}, "t3"), tres("t3", "ok"), + tcall("a4", 4000, "Bash", {"command": "git commit -m x"}, "t4"), tres("t4", "done"), +]) + "\n" +# agent B: 4 exploring calls of 1000 each, reads src/a.py (100 tok) + sed -n of b.py (400 chars) +EXB = "\n".join([ + tcall("b1", 1000, "Read", {"file_path": "/w/proj/src/a.py"}, "u1"), tres("u1", "x" * 400), + tcall("b2", 1000, "Bash", {"command": "sed -n 1,9p src/b.py"}, "u2"), tres("u2", "z" * 400), + tcall("b3", 1000, "Bash", {"command": "ls"}, "u3"), tres("u3", "q"), + tcall("b4", 1000, "Glob", {}, "u4"), tres("u4", "q"), +]) + "\n" + + +class Explore(unittest.TestCase): + def test_one_agent(self): + e = usage.explore(EXA, root="/w/proj") + self.assertEqual((e.calls, e.first_edit, e.pre, e.ctx_growth), (4, 2, 3000, 2000)) + self.assertEqual(e.cost, {"exp": 3000, "mut": 3000, "book": 4000}) + self.assertEqual(e.results, {"Read": 100, "grep": 200, "Edit": 0, "other": 1}) + self.assertEqual(e.files, {"src/a.py": 300}) + + def test_call_kind_book_is_writes_only(self): + kind = lambda c: usage.call_kind({"name": "Bash", "input": {"command": c}}) + for c in ("git worktree list", "git branch --list 'fast/*'", "git branch --show-current", "git status --short"): + self.assertEqual(kind(c), "exp", c) + for c in ("git worktree add .w/x -b x master", "git worktree remove .w/x", "git branch -d x", + "python3 /projects/public/workflow/wf.py finish t-x -m ok", "git switch -c x"): + self.assertEqual(kind(c), "book", c) + + def test_since_drops_early_calls(self): + e = usage.explore(EXA.replace("10:00:00Z", "09:00:00Z", 2), since="2026-10-04T10") + self.assertEqual(e.calls, 2) + + def test_report(self): + a, b = usage.explore(EXA, root="/w/proj"), usage.explore(EXB, root="/w/proj") + r = usage.explore_report([("A", a), ("B", b)], min_pre=4) + self.assertEqual(r["rows"], [("A", 4, 2, 0.3, 0.3), ("B", 4, 4, 1.0, 0.0)]) + self.assertEqual((r["total"], r["cost"]["exp"]), (14000, 7000)) + self.assertEqual((r["pre_n"], r["pre_share"], r["pre_calls"], r["pre_ctx"]), (1, 0.3, 2, 2000)) + self.assertEqual(r["results"]["sed-cat"], 100) + self.assertEqual(r["files"], [("src/a.py", 2, 400)]) + self.assertEqual(usage.explore_report([("A", a)], min_calls=5)["rows"], []) + + +class Cli(unittest.TestCase): + def setUp(self): + self.tmp = tempfile.TemporaryDirectory() + self.home = Path(self.tmp.name) + proj = self.home / "projects" / "-work-demo" + sub = proj / SID / "subagents" + sub.mkdir(parents=True) + (proj / f"{SID}.jsonl").write_text(MAIN) + (sub / "agent-abc123.jsonl").write_text(AGENT) + (sub / "agent-abc123.meta.json").write_text(json.dumps({"description": "wf-worker sonnet t-x"})) + + def tearDown(self): + self.tmp.cleanup() + + def wf(self, *args, sid="", cwd=None): + env = {**os.environ, "CLAUDE_CONFIG_DIR": str(self.home), "CLAUDE_CODE_SESSION_ID": sid} + r = subprocess.run([sys.executable, str(WF), "usage", *args], capture_output=True, text=True, + cwd=cwd or self.tmp.name, env=env, timeout=30) + return r.returncode, r.stdout, r.stderr + + def test_explore_skips_main_and_short_agents(self): + code, out, err = self.wf("--session", SID[:8], "--explore") + self.assertEqual((code, err), (0, "")) + self.assertEqual(out.strip(), "no subagent with >= 4 calls") + + def test_session_rows_and_total(self): + code, out, err = self.wf("--session", SID[:8]) + self.assertEqual((code, err), (0, "")) + rows = [l.split() for l in out.strip().split("\n")[1:]] + self.assertEqual(rows[0], ["main", "opus-5-5", "2", "1.0k", "150.0k", "1.00M", "12.0k", "1.34"]) + self.assertEqual(rows[1], ["abc123", "wf-worker", "sonnet", "t-x", + "sonnet-5-5", "2", "10", "20.0k", "80.0k", "2.0k", "0.09"]) + self.assertEqual(rows[2][-7:], ["mystery-1", "1", "7", "0", "0", "3", "?"]) + self.assertEqual(rows[3], ["total", "5", "1.0k", "170.0k", "1.08M", "14.0k", "1.43+"]) + + def test_session_from_env(self): + code, out, _ = self.wf(sid=SID) + self.assertEqual(code, 0) + self.assertIn("abc123", out) + + def test_agent_only(self): + code, out, err = self.wf("--agent", "agent-abc123", "--since", "2026-10-04T11:00") + self.assertEqual((code, err), (0, "")) + lines = out.strip().split("\n") + self.assertNotIn("main", out) + self.assertEqual(lines[1].split()[-7:], ["sonnet-5-5", "1", "0", "0", "50.0k", "1.5k", "0.03"]) + + def test_log_appends_one_key_value_line(self): + root = self.home / "demo" + root.mkdir() + (root / "workflow.toml").write_text("format = 1\n") + code, out, err = self.wf("--agent", "abc123", "--log", "t-x", "done", "--effort", "1h", cwd=root) + self.assertEqual((code, err), (0, "")) + code, out, err = self.wf("--agent", "abc123", "--log", "t-x", "handback", cwd=root) + log = (root / "out" / "wf-cost.log").read_text().split("\n") + self.assertEqual(len(log), 3) # two lines + trailing newline + # lane = model with most turns (sonnet 2 vs mystery 1); usd = known models only + self.assertRegex(log[0], r"^\d{4}-\d\d-\d\dT\d\d:\d\d:\d\dZ project=demo task=t-x lane=sonnet model=sonnet effort=1h " + r"outcome=done turns=3 in=17 cw=20000 cr=80000 out=2003 usd=0\.0860 agent=abc123$") + self.assertIn(" effort=- outcome=handback ", log[1]) + self.assertEqual(out, log[1] + "\n") + + def test_estimated_out_marked_with_tilde(self): + other = self.home / "projects" / "-work-demo" / "other-session" / "subagents" + other.mkdir(parents=True) + (other / "agent-est1.jsonl").write_text(SUB) + code, out, err = self.wf("--agent", "est1") + self.assertEqual((code, err), (0, "")) + self.assertEqual(out.strip().split("\n")[1].split()[-3:], ["0", "~733", "0.01"]) + + def test_log_needs_agent(self): + code, out, err = self.wf("--log", "t-x", "done") + self.assertEqual((code, err), (2, "wf: --log needs --agent\n")) + + def test_log_effort_must_be_an_estimate(self): + code, out, err = self.wf("--agent", "abc123", "--log", "t-x", "done", "--effort", "medium") + self.assertEqual(code, 2) + self.assertIn("invalid choice: 'medium'", err) + + def test_missing_agent_one_line_error(self): + code, out, err = self.wf("--agent", "nope") + self.assertEqual((code, out), (1, "")) + self.assertEqual(err, "wf: no transcript for agent nope\n") + + +LOG_A = """\ +2026-10-01T10:00:00Z project=a task=t-1 lane=sonnet effort=1h outcome=done turns=10 usd=0.50 +2026-10-01T11:00:00Z project=a task=t-2 lane=sonnet effort=1h outcome=handback turns=30 usd=1.50 +garbage line +""" +LOG_B = """\ +2026-10-02T10:00:00Z project=b task=t-3 lane=sonnet effort=<1h outcome=done turns=20 usd=0.40 +2026-10-02T11:00:00Z project=b task=t-4 lane=opus effort=5h outcome=done turns=80 usd=3.00 +2026-10-03T11:00:00Z project=b task=t-5 lane=opus effort=1h outcome=done turns=40 usd=2.00 +""" + + +class Report(unittest.TestCase): + def test_rows_by_lane_then_effort(self): + entries = usage.parse_log(LOG_A + LOG_B) + self.assertEqual(len(entries), 5) + rows = usage.report(entries, ("<1h", "1h", "5h")) + # lane, effort, n, done, other, median $, total $, $/done, median turns — all by hand + self.assertEqual(rows, [ + ("opus", "all", 2, 2, 0, 2.5, 5.0, 2.5, 60, None), + ("opus", "1h", 1, 1, 0, 2.0, 2.0, 2.0, 40, None), + ("opus", "5h", 1, 1, 0, 3.0, 3.0, 3.0, 80, None), + ("sonnet", "all", 3, 2, 1, 0.5, 2.4, 1.2, 20, None), + ("sonnet", "<1h", 1, 1, 0, 0.4, 0.4, 0.4, 20, None), + ("sonnet", "1h", 2, 1, 1, 1.0, 2.0, 2.0, 20, None), + ]) + + def test_duration(self): + line = usage.log_line("T", "p", "t-x", "1h", "done", "a1", usage.parse(SUB), dur=125) + self.assertIn(" dur=125 agent=a1", line) + self.assertNotIn("dur=", usage.log_line("T", "p", "t-x", "1h", "done", "a1", usage.parse(SUB))) + log = ("2026-10-04T10:00:00Z project=a task=t-6 lane=fast effort=<1h outcome=done turns=1 usd=1 dur=100\n" + "2026-10-04T10:00:00Z project=a task=t-7 lane=fast effort=<1h outcome=done turns=1 usd=1 dur=300\n" + "2026-10-04T10:00:00Z project=a task=t-8 lane=fast effort=<1h outcome=done turns=1 usd=1\n") + self.assertEqual(usage.report(usage.parse_log(log), ("<1h",))[0][9], 200) + + def test_done_gate_red_counts_as_done(self): + log = ("2026-10-04T10:00:00Z project=a task=t-6 lane=fast effort=<1h outcome=done+gate-red turns=12 usd=1.00\n" + "2026-10-04T11:00:00Z project=a task=t-7 lane=fast effort=<1h outcome=donex turns=8 usd=0.50\n") + rows = usage.report(usage.parse_log(log), ("<1h",)) + self.assertEqual(rows[0], ("fast", "all", 2, 1, 1, 0.75, 1.5, 1.5, 10, None)) + + def test_since_and_no_done(self): + entries = usage.parse_log(LOG_A + LOG_B, since="2026-10-01T10:30") + rows = usage.report(entries, ("1h",)) + self.assertIn(("sonnet", "all", 2, 1, 1, 0.95, 1.9, 1.9, 25, None), rows) + rows = usage.report(usage.parse_log(LOG_A, since="2026-10-01T10:30"), ("1h",)) + self.assertEqual(rows[0], ("sonnet", "all", 1, 0, 1, 1.5, 1.5, None, 30, None)) + + def test_cli_reads_every_project(self): + with tempfile.TemporaryDirectory() as tmp: + for name, text in (("a", LOG_A), ("b", LOG_B)): + (Path(tmp) / name / "out").mkdir(parents=True) + (Path(tmp) / name / "workflow.toml").write_text("format = 1\n") + (Path(tmp) / name / "out" / "wf-cost.log").write_text(text) + r = subprocess.run([sys.executable, str(WF), "usage", "--report"], capture_output=True, text=True, + cwd=tmp, env={**os.environ, "WF_ROOT": tmp}, timeout=30) + self.assertEqual((r.returncode, r.stderr), (0, "")) + lines = [l.split() for l in r.stdout.strip().split("\n")] + self.assertEqual(lines[0], ["lane", "effort", "n", "done", "other", "med$", "total$", "$/done", "med_turns", "med_dur"]) + self.assertEqual(lines[1], ["opus", "all", "2", "2", "0", "2.50", "5.00", "2.50", "60", "-"]) + self.assertEqual(lines[6], ["sonnet", "1h", "2", "1", "1", "1.00", "2.00", "2.00", "20", "-"]) + + +if __name__ == "__main__": + unittest.main() |
