From 81d4e80fd5aabe4e80f58e960affa795cf7d34ec Mon Sep 17 00:00:00 2001 From: godosa Date: Wed, 7 Oct 2026 07:27:17 +0200 Subject: workflow: initial public history --- wflib/refs.py | 107 ++++++++++++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 107 insertions(+) create mode 100644 wflib/refs.py (limited to 'wflib/refs.py') diff --git a/wflib/refs.py b/wflib/refs.py new file mode 100644 index 0000000..d1e853d --- /dev/null +++ b/wflib/refs.py @@ -0,0 +1,107 @@ +"""Links, anchors and doc sections: what `[[id]]` and `Ref: path#anchor` point at.""" +from __future__ import annotations + +import re +from dataclasses import dataclass +from pathlib import Path + +LINK_RE = re.compile(r"\[\[([^\]|#]+)(?:#[^\]|]+)?(?:\|[^\]]+)?\]\]") +HEADING_RE = re.compile(r"^(#{1,6})\s+(.*?)\s*$") +ANCHOR_RE = re.compile(r'\s*', re.I) +FENCE_RE = re.compile(r"^\s*(```|~~~)") + + +def strip_code(text: str) -> str: + text = re.sub(r"```.*?```", "", text, flags=re.S) + return re.sub(r"`[^`\n]*`", "", text) + + +def links(text: str) -> list[str]: + return [m.group(1).strip() for m in LINK_RE.finditer(strip_code(text))] + + +def slugify(heading: str) -> str: + text = ANCHOR_RE.sub("", heading).strip().lower() + text = re.sub(r"[^\w\s-]", "", text) + return re.sub(r"[\s_]+", "-", text).strip("-").replace("--", "-") + + +@dataclass +class DocSection: + level: int # 0 = an anchor standing alone, not a heading + heading: str + anchors: set[str] + start: int # 0-based line of the heading (or of the lone anchor) + end: int # exclusive + + +def _lines(text: str) -> list[str]: + lines = text.replace("\r\n", "\n").split("\n") + while lines and not lines[-1].strip(): + lines.pop() + return lines + + +def _is_lone_anchor(line: str) -> bool: + return bool(ANCHOR_RE.search(line)) and not ANCHOR_RE.sub("", line).strip() + + +def sections(text: str) -> list[DocSection]: + lines = _lines(text) + out: list[DocSection] = [] + fenced = False + for n, line in enumerate(lines): + if FENCE_RE.match(line): + fenced = not fenced + if fenced or FENCE_RE.match(line): + continue + m = HEADING_RE.match(line) + tags = set(ANCHOR_RE.findall(line)) + if m: + heading = ANCHOR_RE.sub("", m.group(2)).strip() + above = set(ANCHOR_RE.findall(lines[n - 1])) if n and _is_lone_anchor(lines[n - 1]) else set() + out.append(DocSection(len(m.group(1)), heading, {slugify(heading)} | tags | above, n, len(lines))) + elif tags and not (_is_lone_anchor(line) and n + 1 < len(lines) and HEADING_RE.match(lines[n + 1])): + out.append(DocSection(0, "", tags, n, len(lines))) + headings = [s for s in out if s.level] + for s in out: + nxt = next((h for h in headings if h.start > s.start and (s.level == 0 or h.level <= s.level)), None) + if nxt: + above = nxt.start - 1 + s.end = above if above > s.start and _is_lone_anchor(lines[above]) else nxt.start + return out + + +def anchors(text: str) -> set[str]: + return {a for s in sections(text) for a in s.anchors} + + +def find(text: str, anchor: str) -> DocSection | None: + found = sections(text) + return next((s for s in found if anchor in s.anchors and s.level), None) or \ + next((s for s in found if anchor in s.anchors), None) + + +def resolve(root: Path, path: str, anchor: str | None, limit: int = 80) -> str: + target = root / path + shown = path.rstrip("/") + if target.is_dir(): + return f"{shown}/ (directory)" + if not target.is_file(): + return f"(missing: {shown})" + text = target.read_text(encoding="utf-8", errors="replace") + lines = _lines(text) + if anchor is None: + heading = next((m.group(2) for l in lines if (m := HEADING_RE.match(l))), "") + goal = next((l[len("**Goal:**"):].strip() for l in lines if l.startswith("**Goal:**")), "") + return shown + (f": {heading}" if heading else "") + (f" — {goal}" if goal else "") + s = find(text, anchor) + if s is None: + return f"(no anchor '{anchor}' in {shown})" + body = lines[s.start:s.end] + while body and not body[-1].strip(): + body.pop() + out = [f"===== {shown}#{anchor} (line {s.start + 1}) =====", *body[:limit]] + if len(body) > limit: + out.append(f"… ({len(body) - limit} more lines: {shown}:{s.start + limit + 1})") + return "\n".join(out) -- cgit