"""Links, anchors and doc sections: what `[[id]]` and `Ref: path#anchor` point at.""" from __future__ import annotations import re from dataclasses import dataclass from pathlib import Path LINK_RE = re.compile(r"\[\[([^\]|#]+)(?:#[^\]|]+)?(?:\|[^\]]+)?\]\]") HEADING_RE = re.compile(r"^(#{1,6})\s+(.*?)\s*$") ANCHOR_RE = re.compile(r'\s*', re.I) FENCE_RE = re.compile(r"^\s*(```|~~~)") def strip_code(text: str) -> str: text = re.sub(r"```.*?```", "", text, flags=re.S) return re.sub(r"`[^`\n]*`", "", text) def links(text: str) -> list[str]: return [m.group(1).strip() for m in LINK_RE.finditer(strip_code(text))] def slugify(heading: str) -> str: text = ANCHOR_RE.sub("", heading).strip().lower() text = re.sub(r"[^\w\s-]", "", text) return re.sub(r"[\s_]+", "-", text).strip("-").replace("--", "-") @dataclass class DocSection: level: int # 0 = an anchor standing alone, not a heading heading: str anchors: set[str] start: int # 0-based line of the heading (or of the lone anchor) end: int # exclusive def _lines(text: str) -> list[str]: lines = text.replace("\r\n", "\n").split("\n") while lines and not lines[-1].strip(): lines.pop() return lines def _is_lone_anchor(line: str) -> bool: return bool(ANCHOR_RE.search(line)) and not ANCHOR_RE.sub("", line).strip() def sections(text: str) -> list[DocSection]: lines = _lines(text) out: list[DocSection] = [] fenced = False for n, line in enumerate(lines): if FENCE_RE.match(line): fenced = not fenced if fenced or FENCE_RE.match(line): continue m = HEADING_RE.match(line) tags = set(ANCHOR_RE.findall(line)) if m: heading = ANCHOR_RE.sub("", m.group(2)).strip() above = set(ANCHOR_RE.findall(lines[n - 1])) if n and _is_lone_anchor(lines[n - 1]) else set() out.append(DocSection(len(m.group(1)), heading, {slugify(heading)} | tags | above, n, len(lines))) elif tags and not (_is_lone_anchor(line) and n + 1 < len(lines) and HEADING_RE.match(lines[n + 1])): out.append(DocSection(0, "", tags, n, len(lines))) headings = [s for s in out if s.level] for s in out: nxt = next((h for h in headings if h.start > s.start and (s.level == 0 or h.level <= s.level)), None) if nxt: above = nxt.start - 1 s.end = above if above > s.start and _is_lone_anchor(lines[above]) else nxt.start return out def anchors(text: str) -> set[str]: return {a for s in sections(text) for a in s.anchors} def find(text: str, anchor: str) -> DocSection | None: found = sections(text) return next((s for s in found if anchor in s.anchors and s.level), None) or \ next((s for s in found if anchor in s.anchors), None) def resolve(root: Path, path: str, anchor: str | None, limit: int = 80) -> str: target = root / path shown = path.rstrip("/") if target.is_dir(): return f"{shown}/ (directory)" if not target.is_file(): return f"(missing: {shown})" text = target.read_text(encoding="utf-8", errors="replace") lines = _lines(text) if anchor is None: heading = next((m.group(2) for l in lines if (m := HEADING_RE.match(l))), "") goal = next((l[len("**Goal:**"):].strip() for l in lines if l.startswith("**Goal:**")), "") return shown + (f": {heading}" if heading else "") + (f" — {goal}" if goal else "") s = find(text, anchor) if s is None: return f"(no anchor '{anchor}' in {shown})" body = lines[s.start:s.end] while body and not body[-1].strip(): body.pop() out = [f"===== {shown}#{anchor} (line {s.start + 1}) =====", *body[:limit]] if len(body) > limit: out.append(f"… ({len(body) - limit} more lines: {shown}:{s.start + limit + 1})") return "\n".join(out)