workflow

git clone https://git.godosa.eu/workflow

master

raw · 7566 bytes

"""Area notes (`## Areas` in the project's CLAUDE.md): code-map anchors, staleness, Checked marker."""
from __future__ import annotations

import fnmatch
import re
import subprocess
from dataclasses import dataclass, replace
from pathlib import Path


@dataclass(frozen=True)
class Area:
    name: str
    slug: str
    anchors: list[str]
    paths: list[str]
    checked: str | None
    line: int


def slug(name: str) -> str:
    return re.sub(r"[^a-z0-9]+", "-", name.lower()).strip("-")


def parse(text: str) -> list[Area]:
    out: list[Area] = []
    in_areas = False
    cur: dict | None = None

    def flush():
        if cur:
            out.append(Area(cur["name"], slug(cur["name"]), cur["anchors"], cur["paths"], cur["checked"], cur["line"]))

    for n, line in enumerate(text.replace("\r\n", "\n").split("\n"), 1):
        if line.startswith("## "):
            flush()
            cur = None
            in_areas = line[3:].strip().lower() == "areas"
        elif in_areas and line.startswith("### "):
            flush()
            cur = {"name": line[4:].strip(), "anchors": [], "paths": [], "checked": None, "line": n}
        elif cur is not None:
            if line.startswith("- Code map:"):
                cur["anchors"] = re.findall(r"`([^`]+)`", line)
            elif line.startswith("- Paths:"):
                cur["paths"] = line[len("- Paths:"):].split()
            elif line.startswith("- Checked:"):
                words = line[len("- Checked:"):].split()
                cur["checked"] = words[0] if words and re.fullmatch(r"[0-9a-fA-F]{4,40}", words[0]) else None
    flush()
    return out


def matches(area: Area, text: str, skip: set[str] = frozenset()) -> bool:
    """text (a task) names the area, one of its anchors or Paths (whole word / path, case-insensitive name);
    Paths in `skip` (shared by several areas) never match."""
    def word(tok: str, flags=0) -> bool:
        return re.search(r"(?<![\w.])" + re.escape(tok) + r"(?![\w])", text, flags) is not None
    paths = [p.removeprefix("./").rstrip("/") for p in area.paths if not any(c in p for c in "*?[")]
    paths = [p for p in paths if p not in skip]
    return (word(area.name, re.I) or any(word(t) for t in area.anchors)
            or any(word(p) for p in paths if p))


def shared_paths(found: list[Area]) -> set[str]:
    """Paths (normalised) listed by more than one area."""
    seen: dict[str, int] = {}
    for a in found:
        for p in {p.removeprefix("./").rstrip("/") for p in a.paths}:
            seen[p] = seen.get(p, 0) + 1
    return {p for p, n in seen.items() if n > 1}


def block(text: str, name: str) -> str:
    """The area's notes: its ### heading up to the next heading, trailing blank lines dropped; unknown → ""."""
    lines = text.replace("\r\n", "\n").split("\n")
    in_areas, out = False, None
    for line in lines:
        if line.startswith("#"):
            if out is not None:
                break
            if line.startswith("## "):
                in_areas = line[3:].strip().lower() == "areas"
            elif in_areas and line.startswith("### ") and line[4:].strip() == name:
                out = [line]
        elif out is not None:
            out.append(line)
    while out and not out[-1].strip():
        out.pop()
    return "\n".join(out or [])


def covers(paths: list[str], file: str) -> bool:
    """file (project-relative) matches one of an area's Paths: dir prefix, exact file or glob."""
    for p in paths:
        p = p.removeprefix("./").rstrip("/")
        if file == p or file.startswith(p + "/") or fnmatch.fnmatchcase(file, p):
            return True
    return False


def uncovered(files: list[str], found: list[Area], ignore: list[str]) -> list[tuple[str, int]]:
    """Folders (≤ 2 levels) of files no area's Paths covers → [(folder, file count)], most files first.
    Root files, dot folders and `ignore` prefixes skipped; no area with Paths → []."""
    if not any(a.paths for a in found):
        return []
    count: dict[str, int] = {}
    for f in files:
        parts = f.split("/")[:-1]
        if not parts or any(x.startswith(".") for x in parts) or covers(ignore, f):
            continue
        if any(covers(a.paths, f) for a in found):
            continue
        d = "/".join(parts[:2])
        count[d] = count.get(d, 0) + 1
    return sorted(count.items(), key=lambda kv: (-kv[1], kv[0]))


def changed_files(here: Path, base: str | None) -> list[str]:
    """Files (relative to here) changed since merge-base(HEAD, base) incl. uncommitted + untracked;
    base None → HEAD (uncommitted only)."""
    ref = "HEAD"
    if base:
        r = _git(here, "merge-base", "HEAD", base)
        if r.returncode == 0 and r.stdout.strip():
            ref = r.stdout.strip()
    out = []
    for args in (("diff", "--name-only", "--relative", ref), ("ls-files", "--others", "--exclude-standard")):
        r = _git(here, *args)
        if r.returncode == 0:
            out += [l for l in r.stdout.splitlines() if l]
    return sorted(set(out))


def with_anchor_paths(root: Path, area: Area) -> Area:
    """No Paths → the folders of the files holding its anchors (git grep) stand in for them."""
    if area.paths or not area.anchors:
        return area
    args = [x for tok in area.anchors for x in ("-e", tok)]
    r = _git(root, "grep", "-l", "-F", *args)
    files = r.stdout.splitlines() if r.returncode == 0 else []
    return replace(area, paths=sorted({f.rsplit("/", 1)[0] if "/" in f else f for f in files}))


def _git(root: Path, *args: str) -> subprocess.CompletedProcess:
    return subprocess.run(["git", "-C", str(root), *args], capture_output=True, text=True)


def missing(root: Path, area: Area) -> list[str]:
    out = []
    for tok in area.anchors:
        r = _git(root, "grep", "-F", "-q", "-e", tok, *(["--", *area.paths] if area.paths else []))
        if r.returncode != 0:
            out.append(tok)
    return out


def commits_since(root: Path, area: Area) -> int | None:
    if not area.checked or _git(root, "cat-file", "-e", f"{area.checked}^{{commit}}").returncode != 0:
        return None
    r = _git(root, "rev-list", "--count", f"{area.checked}..HEAD", *(["--", *area.paths] if area.paths else []))
    return int(r.stdout.strip()) if r.returncode == 0 and r.stdout.strip().isdigit() else None


def status(root: Path, area: Area, limit: int) -> tuple[bool, list[str], int | None]:
    miss = missing(root, area)
    commits = commits_since(root, area)
    return (bool(miss) or (commits is not None and commits >= limit)), miss, commits


def mark(text: str, name: str, sha: str) -> str:
    lines = text.split("\n")
    in_areas = False
    start = None
    for i, line in enumerate(lines):
        if line.startswith("## "):
            if start is not None:
                break
            in_areas = line[3:].strip().lower() == "areas"
        elif in_areas and line.startswith("### "):
            if start is not None:
                break
            if line[4:].strip() == name:
                start = i
    if start is None:
        return text
    end = start + 1
    while end < len(lines) and not lines[end].startswith("#"):
        end += 1
    block_end = end
    while block_end > start + 1 and not lines[block_end - 1].strip():
        block_end -= 1
    new = f"- Checked: {sha}"
    for i in range(start + 1, block_end):
        if lines[i].startswith("- Checked:"):
            lines[i] = new
            return "\n".join(lines)
    lines.insert(block_end, new)
    return "\n".join(lines)