aboutsummaryrefslogtreecommitdiffziptar.gz
path: root/wflib/areas.py
diff options
context:
space:
mode:
Diffstat (limited to 'wflib/areas.py')
-rw-r--r--wflib/areas.py199
1 files changed, 199 insertions, 0 deletions
diff --git a/wflib/areas.py b/wflib/areas.py
new file mode 100644
index 0000000..ce23d12
--- /dev/null
+++ b/wflib/areas.py
@@ -0,0 +1,199 @@
+"""Area notes (`## Areas` in the project's CLAUDE.md): code-map anchors, staleness, Checked marker."""
+from __future__ import annotations
+
+import fnmatch
+import re
+import subprocess
+from dataclasses import dataclass, replace
+from pathlib import Path
+
+
+@dataclass(frozen=True)
+class Area:
+ name: str
+ slug: str
+ anchors: list[str]
+ paths: list[str]
+ checked: str | None
+ line: int
+
+
+def slug(name: str) -> str:
+ return re.sub(r"[^a-z0-9]+", "-", name.lower()).strip("-")
+
+
+def parse(text: str) -> list[Area]:
+ out: list[Area] = []
+ in_areas = False
+ cur: dict | None = None
+
+ def flush():
+ if cur:
+ out.append(Area(cur["name"], slug(cur["name"]), cur["anchors"], cur["paths"], cur["checked"], cur["line"]))
+
+ for n, line in enumerate(text.replace("\r\n", "\n").split("\n"), 1):
+ if line.startswith("## "):
+ flush()
+ cur = None
+ in_areas = line[3:].strip().lower() == "areas"
+ elif in_areas and line.startswith("### "):
+ flush()
+ cur = {"name": line[4:].strip(), "anchors": [], "paths": [], "checked": None, "line": n}
+ elif cur is not None:
+ if line.startswith("- Code map:"):
+ cur["anchors"] = re.findall(r"`([^`]+)`", line)
+ elif line.startswith("- Paths:"):
+ cur["paths"] = line[len("- Paths:"):].split()
+ elif line.startswith("- Checked:"):
+ words = line[len("- Checked:"):].split()
+ cur["checked"] = words[0] if words and re.fullmatch(r"[0-9a-fA-F]{4,40}", words[0]) else None
+ flush()
+ return out
+
+
+def matches(area: Area, text: str, skip: set[str] = frozenset()) -> bool:
+ """text (a task) names the area, one of its anchors or Paths (whole word / path, case-insensitive name);
+ Paths in `skip` (shared by several areas) never match."""
+ def word(tok: str, flags=0) -> bool:
+ return re.search(r"(?<![\w.])" + re.escape(tok) + r"(?![\w])", text, flags) is not None
+ paths = [p.removeprefix("./").rstrip("/") for p in area.paths if not any(c in p for c in "*?[")]
+ paths = [p for p in paths if p not in skip]
+ return (word(area.name, re.I) or any(word(t) for t in area.anchors)
+ or any(word(p) for p in paths if p))
+
+
+def shared_paths(found: list[Area]) -> set[str]:
+ """Paths (normalised) listed by more than one area."""
+ seen: dict[str, int] = {}
+ for a in found:
+ for p in {p.removeprefix("./").rstrip("/") for p in a.paths}:
+ seen[p] = seen.get(p, 0) + 1
+ return {p for p, n in seen.items() if n > 1}
+
+
+def block(text: str, name: str) -> str:
+ """The area's notes: its ### heading up to the next heading, trailing blank lines dropped; unknown → ""."""
+ lines = text.replace("\r\n", "\n").split("\n")
+ in_areas, out = False, None
+ for line in lines:
+ if line.startswith("#"):
+ if out is not None:
+ break
+ if line.startswith("## "):
+ in_areas = line[3:].strip().lower() == "areas"
+ elif in_areas and line.startswith("### ") and line[4:].strip() == name:
+ out = [line]
+ elif out is not None:
+ out.append(line)
+ while out and not out[-1].strip():
+ out.pop()
+ return "\n".join(out or [])
+
+
+def covers(paths: list[str], file: str) -> bool:
+ """file (project-relative) matches one of an area's Paths: dir prefix, exact file or glob."""
+ for p in paths:
+ p = p.removeprefix("./").rstrip("/")
+ if file == p or file.startswith(p + "/") or fnmatch.fnmatchcase(file, p):
+ return True
+ return False
+
+
+def uncovered(files: list[str], found: list[Area], ignore: list[str]) -> list[tuple[str, int]]:
+ """Folders (≤ 2 levels) of files no area's Paths covers → [(folder, file count)], most files first.
+ Root files, dot folders and `ignore` prefixes skipped; no area with Paths → []."""
+ if not any(a.paths for a in found):
+ return []
+ count: dict[str, int] = {}
+ for f in files:
+ parts = f.split("/")[:-1]
+ if not parts or any(x.startswith(".") for x in parts) or covers(ignore, f):
+ continue
+ if any(covers(a.paths, f) for a in found):
+ continue
+ d = "/".join(parts[:2])
+ count[d] = count.get(d, 0) + 1
+ return sorted(count.items(), key=lambda kv: (-kv[1], kv[0]))
+
+
+def changed_files(here: Path, base: str | None) -> list[str]:
+ """Files (relative to here) changed since merge-base(HEAD, base) incl. uncommitted + untracked;
+ base None → HEAD (uncommitted only)."""
+ ref = "HEAD"
+ if base:
+ r = _git(here, "merge-base", "HEAD", base)
+ if r.returncode == 0 and r.stdout.strip():
+ ref = r.stdout.strip()
+ out = []
+ for args in (("diff", "--name-only", "--relative", ref), ("ls-files", "--others", "--exclude-standard")):
+ r = _git(here, *args)
+ if r.returncode == 0:
+ out += [l for l in r.stdout.splitlines() if l]
+ return sorted(set(out))
+
+
+def with_anchor_paths(root: Path, area: Area) -> Area:
+ """No Paths → the folders of the files holding its anchors (git grep) stand in for them."""
+ if area.paths or not area.anchors:
+ return area
+ args = [x for tok in area.anchors for x in ("-e", tok)]
+ r = _git(root, "grep", "-l", "-F", *args)
+ files = r.stdout.splitlines() if r.returncode == 0 else []
+ return replace(area, paths=sorted({f.rsplit("/", 1)[0] if "/" in f else f for f in files}))
+
+
+def _git(root: Path, *args: str) -> subprocess.CompletedProcess:
+ return subprocess.run(["git", "-C", str(root), *args], capture_output=True, text=True)
+
+
+def missing(root: Path, area: Area) -> list[str]:
+ out = []
+ for tok in area.anchors:
+ r = _git(root, "grep", "-F", "-q", "-e", tok, *(["--", *area.paths] if area.paths else []))
+ if r.returncode != 0:
+ out.append(tok)
+ return out
+
+
+def commits_since(root: Path, area: Area) -> int | None:
+ if not area.checked or _git(root, "cat-file", "-e", f"{area.checked}^{{commit}}").returncode != 0:
+ return None
+ r = _git(root, "rev-list", "--count", f"{area.checked}..HEAD", *(["--", *area.paths] if area.paths else []))
+ return int(r.stdout.strip()) if r.returncode == 0 and r.stdout.strip().isdigit() else None
+
+
+def status(root: Path, area: Area, limit: int) -> tuple[bool, list[str], int | None]:
+ miss = missing(root, area)
+ commits = commits_since(root, area)
+ return (bool(miss) or (commits is not None and commits >= limit)), miss, commits
+
+
+def mark(text: str, name: str, sha: str) -> str:
+ lines = text.split("\n")
+ in_areas = False
+ start = None
+ for i, line in enumerate(lines):
+ if line.startswith("## "):
+ if start is not None:
+ break
+ in_areas = line[3:].strip().lower() == "areas"
+ elif in_areas and line.startswith("### "):
+ if start is not None:
+ break
+ if line[4:].strip() == name:
+ start = i
+ if start is None:
+ return text
+ end = start + 1
+ while end < len(lines) and not lines[end].startswith("#"):
+ end += 1
+ block_end = end
+ while block_end > start + 1 and not lines[block_end - 1].strip():
+ block_end -= 1
+ new = f"- Checked: {sha}"
+ for i in range(start + 1, block_end):
+ if lines[i].startswith("- Checked:"):
+ lines[i] = new
+ return "\n".join(lines)
+ lines.insert(block_end, new)
+ return "\n".join(lines)