aboutsummaryrefslogtreecommitdiffziptar.gz
path: root/wflib/refs.py
diff options
context:
space:
mode:
Diffstat (limited to 'wflib/refs.py')
-rw-r--r--wflib/refs.py107
1 files changed, 107 insertions, 0 deletions
diff --git a/wflib/refs.py b/wflib/refs.py
new file mode 100644
index 0000000..d1e853d
--- /dev/null
+++ b/wflib/refs.py
@@ -0,0 +1,107 @@
+"""Links, anchors and doc sections: what `[[id]]` and `Ref: path#anchor` point at."""
+from __future__ import annotations
+
+import re
+from dataclasses import dataclass
+from pathlib import Path
+
+LINK_RE = re.compile(r"\[\[([^\]|#]+)(?:#[^\]|]+)?(?:\|[^\]]+)?\]\]")
+HEADING_RE = re.compile(r"^(#{1,6})\s+(.*?)\s*$")
+ANCHOR_RE = re.compile(r'<a\s+id="([^"]+)"\s*>\s*</a>', re.I)
+FENCE_RE = re.compile(r"^\s*(```|~~~)")
+
+
+def strip_code(text: str) -> str:
+ text = re.sub(r"```.*?```", "", text, flags=re.S)
+ return re.sub(r"`[^`\n]*`", "", text)
+
+
+def links(text: str) -> list[str]:
+ return [m.group(1).strip() for m in LINK_RE.finditer(strip_code(text))]
+
+
+def slugify(heading: str) -> str:
+ text = ANCHOR_RE.sub("", heading).strip().lower()
+ text = re.sub(r"[^\w\s-]", "", text)
+ return re.sub(r"[\s_]+", "-", text).strip("-").replace("--", "-")
+
+
+@dataclass
+class DocSection:
+ level: int # 0 = an anchor standing alone, not a heading
+ heading: str
+ anchors: set[str]
+ start: int # 0-based line of the heading (or of the lone anchor)
+ end: int # exclusive
+
+
+def _lines(text: str) -> list[str]:
+ lines = text.replace("\r\n", "\n").split("\n")
+ while lines and not lines[-1].strip():
+ lines.pop()
+ return lines
+
+
+def _is_lone_anchor(line: str) -> bool:
+ return bool(ANCHOR_RE.search(line)) and not ANCHOR_RE.sub("", line).strip()
+
+
+def sections(text: str) -> list[DocSection]:
+ lines = _lines(text)
+ out: list[DocSection] = []
+ fenced = False
+ for n, line in enumerate(lines):
+ if FENCE_RE.match(line):
+ fenced = not fenced
+ if fenced or FENCE_RE.match(line):
+ continue
+ m = HEADING_RE.match(line)
+ tags = set(ANCHOR_RE.findall(line))
+ if m:
+ heading = ANCHOR_RE.sub("", m.group(2)).strip()
+ above = set(ANCHOR_RE.findall(lines[n - 1])) if n and _is_lone_anchor(lines[n - 1]) else set()
+ out.append(DocSection(len(m.group(1)), heading, {slugify(heading)} | tags | above, n, len(lines)))
+ elif tags and not (_is_lone_anchor(line) and n + 1 < len(lines) and HEADING_RE.match(lines[n + 1])):
+ out.append(DocSection(0, "", tags, n, len(lines)))
+ headings = [s for s in out if s.level]
+ for s in out:
+ nxt = next((h for h in headings if h.start > s.start and (s.level == 0 or h.level <= s.level)), None)
+ if nxt:
+ above = nxt.start - 1
+ s.end = above if above > s.start and _is_lone_anchor(lines[above]) else nxt.start
+ return out
+
+
+def anchors(text: str) -> set[str]:
+ return {a for s in sections(text) for a in s.anchors}
+
+
+def find(text: str, anchor: str) -> DocSection | None:
+ found = sections(text)
+ return next((s for s in found if anchor in s.anchors and s.level), None) or \
+ next((s for s in found if anchor in s.anchors), None)
+
+
+def resolve(root: Path, path: str, anchor: str | None, limit: int = 80) -> str:
+ target = root / path
+ shown = path.rstrip("/")
+ if target.is_dir():
+ return f"{shown}/ (directory)"
+ if not target.is_file():
+ return f"(missing: {shown})"
+ text = target.read_text(encoding="utf-8", errors="replace")
+ lines = _lines(text)
+ if anchor is None:
+ heading = next((m.group(2) for l in lines if (m := HEADING_RE.match(l))), "")
+ goal = next((l[len("**Goal:**"):].strip() for l in lines if l.startswith("**Goal:**")), "")
+ return shown + (f": {heading}" if heading else "") + (f" — {goal}" if goal else "")
+ s = find(text, anchor)
+ if s is None:
+ return f"(no anchor '{anchor}' in {shown})"
+ body = lines[s.start:s.end]
+ while body and not body[-1].strip():
+ body.pop()
+ out = [f"===== {shown}#{anchor} (line {s.start + 1}) =====", *body[:limit]]
+ if len(body) > limit:
+ out.append(f"… ({len(body) - limit} more lines: {shown}:{s.start + limit + 1})")
+ return "\n".join(out)