workflow

git clone https://git.godosa.eu/workflow

master

raw · 19483 bytes

"""Validation: everything `wf check` reports."""
from __future__ import annotations

import re
import subprocess
import time
from dataclasses import dataclass
from pathlib import Path

from . import areas, refs, tasks
from .config import NAME, WF_HOME, Config, git_top

NUMBERED_RE = re.compile(r"^\d+\. ")
NUMBER_REF_RE = re.compile(r"(?<![\w/&#])#\d+\b")
MD_LINK_RE = re.compile(r"\[[^\]]*\]\(([^)\s]+)\)")
TASK_LINK_RE = re.compile(r"^[ta]-")
EFFORTS = ", ".join(tasks.EFFORTS)
KINDS = "a- items belong in Awaiting, t- elsewhere"
STALE_DAYS = 30


@dataclass(frozen=True)
class Problem:
    path: str
    line: int | None
    id: str
    message: str

    @property
    def key(self) -> tuple[str, str, str]:
        return (self.path, self.id, self.message)

    def __str__(self) -> str:
        where = self.path + (f":{self.line}" if self.line else "")
        return f"{where}: " + (f"{self.id}: " if self.id else "") + self.message


def _read(path: Path) -> str:
    return path.read_text(encoding="utf-8", errors="replace").replace("\r\n", "\n")


def _prose_lines(text: str):
    """(1-based line number, line) outside code fences."""
    fenced = False
    for n, line in enumerate(text.split("\n"), 1):
        if refs.FENCE_RE.match(line):
            fenced = not fenced
        elif not fenced:
            yield n, line


def _link_problems(path: str, text: str, known: set[str]) -> list[Problem]:
    return [Problem(path, n, f"[[{id}]]", "no such id in TASKS or archive")
            for n, line in _prose_lines(text)
            for id in dict.fromkeys(refs.links(line))
            if TASK_LINK_RE.match(id) and id not in known]


def _cycle(start: str, edges: dict[str, list[str]]) -> list[str] | None:
    def walk(node: str, path: list[str]) -> list[str] | None:
        for nxt in edges.get(node, []):
            if nxt == start:
                return path + [nxt]
            if nxt not in path and (found := walk(nxt, path + [nxt])):
                return found
        return None
    return walk(start, [start])


def _index_sections(cfg: Config) -> list[refs.DocSection] | None:
    """Headings under the anchors index section; None when no [anchors]."""
    if not cfg.anchors_index or not cfg.anchors_index.is_file():
        return None
    found = refs.sections(_read(cfg.anchors_index))
    if not cfg.anchors_section:
        return [s for s in found if s.level]
    top = next((s for s in found if s.level and s.heading == cfg.anchors_section), None)
    if top is None:
        return []
    return [s for s in found if s.level > top.level and top.start < s.start < top.end]


def check_tasks(text: str, archive_text: str, cfg: Config) -> tuple[list[Problem], list[Problem]]:
    path = cfg.rel(cfg.tasks)
    errors: list[Problem] = []
    warnings: list[Problem] = []
    doc = tasks.parse(text)
    archived = tasks.archive_ids(archive_text)

    def err(line, id, message):
        errors.append(Problem(path, line, id, message))

    for key, heading in tasks.SECTIONS.items():
        if not any(s.key == key for s in doc.sections):
            err(None, "", f"no '## {heading}' section")

    errors += _link_problems(path, text, doc.ids() | archived)

    index = _index_sections(cfg)
    index_anchors = {a for s in index for a in s.anchors} if index is not None else None
    open_awaiting = {i.id for s in doc.sections if s.key == "awaiting" for i in s.items}
    deferred = {i.id for s in doc.sections if s.key == "deferred" for i in s.items}
    edges = {i.id: [a for a in i.after if a in doc.ids()] for i in doc.all_items()}
    seen: dict[str, int] = {}
    in_cycle: set[str] = set()

    for section in doc.sections:
        if section.key is None:
            continue
        for n, line in enumerate(section.prefix, section.line + 1):
            if NUMBERED_RE.match(line):
                err(n, "", "old numbered item (wf migrate)")
        for n, line in enumerate(section.suffix, section.suffix_line or 0):
            if NUMBERED_RE.match(line):
                err(n, "", "old numbered item (wf migrate)")
            elif line.startswith(tasks.ITEM_START):
                err(n, tasks.parse_header(line).id,
                    f"item after a flush-left prose line (line {section.suffix_line}): indent or move the prose")
        for at, item in enumerate(section.items):
            if item.error:
                err(item.line, item.id, item.error)
                continue
            if not tasks.ID_RE.match(item.id):
                err(item.line, item.id, "bad id (want t-… or a-…, lowercase a-z 0-9 -)")
                continue
            if item.id in seen:
                err(item.line, item.id, f"duplicate id (also line {seen[item.id]})")
            seen.setdefault(item.id, item.line)
            if item.id in archived:
                err(item.line, item.id, "id already used in the archive (ids are never reused)")
            if item.id.startswith("a-") != (section.key == "awaiting"):
                err(item.line, item.id, f"{item.id[:2]} item in {section.heading} ({KINDS})")
                continue
            if section.key == "awaiting":
                continue
            if item.prio is None:
                err(item.line, item.id, "no priority [P0]-[P3]")
            elif item.prio > 3:
                err(item.line, item.id, f"priority P{item.prio} (want P0-P3)")
            if item.effort is None:
                err(item.line, item.id, f"no effort (want {EFFORTS})")
            elif item.effort not in tasks.EFFORTS:
                err(item.line, item.id, f"effort '{item.effort}' (want {EFFORTS})")
            if item.status and item.status.startswith("blocked:") and item.blocked_on not in open_awaiting:
                err(item.line, item.id, f"blocked on '{item.blocked_on}', which is not an open Awaiting item")
            if item.status and re.fullmatch(r"in progress:\s*", item.status):
                warnings.append(Problem(path, item.line, item.id, "in progress without a branch or note"))
            if item.id not in in_cycle and (cycle := _cycle(item.id, edges)):
                in_cycle.update(cycle)
                err(item.line, item.id, "After: cycle " + " → ".join(cycle))
            later = {o.id for o in section.items[at + 1:]}
            for dep in item.after:
                if dep in later:
                    err(item.line, item.id, f"placed before '{dep}', which it is After:")
            for label, pattern in (("After", tasks.AFTER_RE), ("Slices", tasks.SLICES_RE)):
                if (i := item._line(pattern)) is None:
                    continue
                rest = tasks.LINK_RE.sub(" ", pattern.match(item.body[i]).group(1))
                for word in re.split(r"[\s,;]+", rest):
                    if tasks.ID_RE.match(word):
                        err(item.line + 1 + i, item.id, f"{label}: '{word}' is not a link (want [[{word}]])")
            if (i := item._line(tasks.AFTER_RE)) is not None:
                rest = tasks.LINK_RE.sub(" ", tasks.AFTER_RE.match(item.body[i]).group(1))
                if re.search(r"[A-Za-z0-9]", rest):
                    warnings.append(Problem(path, item.line + 1 + i, item.id,
                                            "After: line has prose; every [[id]] in it is a dependency"))
                for dep in item.after:
                    if dep in deferred and section.key != "deferred":
                        warnings.append(Problem(path, item.line + 1 + i, item.id,
                                                f"After: '{dep}' is deferred (never runs; blocks this task)"))
            models = [(n, m.group(1).split()) for n, l in enumerate(item.body) if (m := tasks.MODEL_RE.match(l))]
            for n, words in models[:1]:
                if not words or words[0] not in tasks.MODELS:
                    err(item.line + 1 + n, item.id,
                        f"Model '{' '.join(words)}' (want {', '.join(tasks.MODELS)})")
                elif len(words) > 1:
                    warnings.append(Problem(path, item.line + 1 + n, item.id, f"Model line '{' '.join(words)}': "
                                            f"write 'Model: {words[0]}' (wf set --model)"))
            if len(models) > 1:
                err(item.line + 1 + models[1][0], item.id, "two Model lines")
            sess = [(n, m.group(1).split()) for n, l in enumerate(item.body) if (m := tasks.SESSIONS_RE.match(l))]
            for n, words in sess[:1]:
                if not words or words[0] not in tasks.SESSIONS:
                    err(item.line + 1 + n, item.id,
                        f"Sessions '{' '.join(words)}' (want {', '.join(tasks.SESSIONS)})")
            if len(sess) > 1:
                err(item.line + 1 + sess[1][0], item.id, "two Sessions lines")
            clouds = [(n, m.group(1).split()) for n, l in enumerate(item.body) if (m := tasks.CLOUD_RE.match(l))]
            for n, words in clouds[:1]:
                if not words or words[0] not in tasks.CLOUDS:
                    err(item.line + 1 + n, item.id, f"Cloud '{' '.join(words)}' (want {', '.join(tasks.CLOUDS)})")
            if len(clouds) > 1:
                err(item.line + 1 + clouds[1][0], item.id, "two Cloud lines")
            if item.interactive:
                warnings.append(Problem(path, item.line, item.id, "'interactive' flag: write 'Sessions: owner' "
                                        f"(wf set {item.id} --sessions owner)"))
            if section.key == "pending" and item.human_done_match and item.sessions != "owner":
                warnings.append(Problem(path, item.line, item.id, f"not runner-ready: Done reads as human action "
                                        f"('{item.human_done_match}'); reword or set Sessions: owner"))
            ref_at = item._line(tasks.REF_RE)
            ref_line = item.line + 1 + ref_at if ref_at is not None else item.line
            for target, anchor in item.refs:
                shown = target + (f"#{anchor}" if anchor else "")
                file = cfg.root / target
                if not file.exists():
                    err(ref_line, item.id, f"Ref '{shown}' does not exist")
                elif anchor and file.is_file():
                    if anchor not in refs.anchors(_read(file)):
                        err(ref_line, item.id, f"Ref '{shown}': no such anchor")
                    elif index_anchors is not None and file.resolve() == cfg.anchors_index.resolve() \
                            and anchor not in index_anchors:
                        err(ref_line, item.id, f"Ref '{shown}' is not a heading under '{cfg.anchors_section}'")

    for n, line in _prose_lines(text):
        for m in NUMBER_REF_RE.finditer(refs.strip_code(line)):
            warnings.append(Problem(path, n, "", f"'{m.group(0)}': number ref (tasks have ids: [[t-…]])"))
    return errors, warnings


def _anchor_problems(cfg: Config) -> list[Problem]:
    index = _index_sections(cfg)
    if index is None:
        return []
    out: list[Problem] = []
    path = cfg.rel(cfg.anchors_index)
    lines = _read(cfg.anchors_index).split("\n")
    where = f"'{cfg.anchors_section}'" if cfg.anchors_section else "the index"
    seen: dict[str, int] = {}
    for s in index:
        slug = refs.slugify(s.heading)
        if slug in seen:
            out.append(Problem(path, s.start + 1, "", f"two {where} headings give anchor '{slug}' (also line {seen[slug]})"))
        seen.setdefault(slug, s.start + 1)
        for n in range(s.start, s.end):
            for target in MD_LINK_RE.findall(refs.strip_code(lines[n])):
                if re.match(r"^[a-z][a-z0-9+.-]*:", target):
                    continue
                file_part, _, frag = target.partition("#")
                file = (cfg.anchors_index.parent / file_part) if file_part else cfg.anchors_index
                if not file.exists():
                    out.append(Problem(path, n + 1, "", f"link '{target}': file does not exist"))
                elif frag and file.suffix == ".md" and frag not in refs.anchors(_read(file)):
                    out.append(Problem(path, n + 1, "", f"link '{target}': no such anchor"))
    if cfg.anchors_specs and cfg.anchors_specs.is_dir():
        known = {a for s in index for a in s.anchors}
        for spec in sorted(cfg.anchors_specs.rglob("*.md")):
            for n, line in _prose_lines(_read(spec)):
                for anchor in refs.ANCHOR_RE.findall(line):
                    if anchor not in known:
                        out.append(Problem(cfg.rel(spec), n, "",
                                           f"explicit anchor '{anchor}' has no heading under {where} in {path}"))
    return out


def doc_files(cfg: Config) -> list[Path]:
    out: list[Path] = []
    for d in cfg.docs:
        if d.is_file():
            out.append(d)
        elif d.is_dir():
            out += sorted(p for p in d.rglob("*.md") if p.is_file())
    skip = {cfg.tasks.resolve(), cfg.archive.resolve()}
    return [p for p in dict.fromkeys(out) if p.resolve() not in skip]


def _age_days(root: Path, path: Path, line: int) -> float | None:
    try:
        out = subprocess.run(["git", "-C", str(root), "blame", "-L", f"{line},{line}", "--porcelain", "--", str(path)],
                             capture_output=True, text=True, timeout=10)
    except (OSError, subprocess.SubprocessError):
        return None
    m = re.search(r"^author-time (\d+)$", out.stdout, re.M) if out.returncode == 0 else None
    if not m or re.match(r"^0{40}", out.stdout):
        return None
    return (time.time() - int(m.group(1))) / 86400


def _stale_awaiting(cfg: Config, text: str) -> list[Problem]:
    doc = tasks.parse(text)
    waited = {i.blocked_on for i in doc.all_items()} | {l for i in doc.all_items() if i.id.startswith("t-")
                                                        for l in refs.links("\n".join(i.lines()))}
    out = []
    for s in doc.sections:
        if s.key != "awaiting":
            continue
        for item in s.items:
            if item.id in waited or item.error:
                continue
            age = _age_days(cfg.root, cfg.tasks, item.line)
            if age is not None and age > STALE_DAYS:
                out.append(Problem(cfg.rel(cfg.tasks), item.line, item.id,
                                   f"waiting {int(age)} days, no task references it"))
    return out


def _git(root: Path, *args: str) -> str | None:
    try:
        out = subprocess.run(["git", "-C", str(root), *args], capture_output=True, text=True, timeout=10)
    except (OSError, subprocess.SubprocessError):
        return None
    return out.stdout.strip() if out.returncode == 0 else None


def merged_tool_worktrees(tool_root: Path, label: str = "tool") -> list[Problem]:
    """Worktrees of the wf tool repo whose branch is behind master (merged, leftover after release).
    A branch equal to master, or a worktree created < 2h ago, is not flagged."""
    listing = _git(tool_root, "worktree", "list", "--porcelain")
    master = _git(tool_root, "rev-parse", "master")
    if not listing or not master:
        return []
    out = []
    for block in listing.split("\n\n"):
        path = branch = head = None
        for l in block.splitlines():
            if l.startswith("worktree "):
                path = l[9:]
            elif l.startswith("HEAD "):
                head = l[5:]
            elif l.startswith("branch refs/heads/"):
                branch = l[18:]
        if not path or not branch or branch == "master" or head == master:
            continue
        try:  # created < 2h ago: a running worker's fresh worktree, not a leftover
            if time.time() - (Path(path) / ".git").stat().st_mtime < 7200:
                continue
        except OSError:
            pass
        if _git(tool_root, "merge-base", "--is-ancestor", branch, "master") is not None:
            out.append(Problem("wf-tool", None, "", f"{label} worktree '{path}' (branch {branch}) is merged into "
                               f"master: git worktree remove it, git branch -d {branch}"))
    return out


def split_problems(cfg: Config) -> tuple[list[Problem], list[Problem]]:
    """Split project rules (code_root = another repo): cloud refused, push expected, .wf-home ignored, merged code worktrees."""
    if not cfg.split:
        return [], []
    code = git_top(cfg.code_root)
    errors, warnings = [], []
    if cfg.cloud:
        errors.append(Problem("workflow.toml", None, "", "cloud lane needs a single repo: split project (code_root is another repo)"))
    if not cfg.push:
        warnings.append(Problem("workflow.toml", None, "", "split project without push: public repo pushed unscanned or not at all"))
    if _git(code, "check-ignore", "-q", WF_HOME) is None:
        warnings.append(Problem(f"{code}/.gitignore", None, "", f"{WF_HOME} not git-ignored (wf start writes it into code worktrees)"))
    warnings += merged_tool_worktrees(code, label="code")
    return errors, warnings


def _area_problems(cfg: Config) -> list[Problem]:
    file = cfg.areas_file
    if not file.is_file():
        return []
    out = []
    for a in areas.parse(_read(file)):
        for tok in areas.missing(cfg.code, a):
            out.append(Problem(cfg.rel(file), a.line, "", f"area {a.name}: anchor {tok} not found"))
        if a.checked and areas.commits_since(cfg.code, a) is None:
            out.append(Problem(cfg.rel(file), a.line, "", f"area {a.name}: Checked {a.checked} unknown"))
    return out


def check(cfg: Config, tasks_text: str | None = None, archive_text: str | None = None,
          slow: bool = True) -> tuple[list[Problem], list[Problem]]:
    """All errors and warnings of the project. `tasks_text` / `archive_text`
    replace what is on disk (to judge a change before it is written);
    `slow=False` leaves out the git-based warnings."""
    errors: list[Problem] = []
    for name, path in (("tasks", cfg.tasks), ("archive", cfg.archive)):
        given = tasks_text if name == "tasks" else archive_text
        if given is None and not path.is_file():
            errors.append(Problem(NAME, None, "", f"{name} '{cfg.rel(path)}' does not exist"))
    for name, path in [("docs", d) for d in cfg.docs] + [("anchors.index", cfg.anchors_index),
                                                          ("anchors.specs", cfg.anchors_specs)]:
        if path is not None and not path.exists():
            errors.append(Problem(NAME, None, "", f"{name} '{cfg.rel(path)}' does not exist"))
    if tasks_text is None and not cfg.tasks.is_file():
        return errors, []
    text = (_read(cfg.tasks) if tasks_text is None else tasks_text).replace("\r\n", "\n")
    archive = archive_text if archive_text is not None else (_read(cfg.archive) if cfg.archive.is_file() else "")
    found, warnings = check_tasks(text, archive, cfg)
    errors += found
    known = tasks.parse(text).ids() | tasks.archive_ids(archive)
    for file in doc_files(cfg):
        errors += _link_problems(cfg.rel(file), _read(file), known)
    errors += _anchor_problems(cfg)
    if slow:
        warnings += _stale_awaiting(cfg, text)
        warnings += _area_problems(cfg)
        e, w = split_problems(cfg)
        errors += e
        warnings += w
    order: dict[str, int] = {}
    for p in errors + warnings:
        order.setdefault(p.path, len(order))
    by_place = lambda p: (order[p.path], p.line or 0)
    return sorted(errors, key=by_place), sorted(warnings, key=by_place)