1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
|
"""Links, anchors and doc sections: what `[[id]]` and `Ref: path#anchor` point at."""
from __future__ import annotations
import re
from dataclasses import dataclass
from pathlib import Path
LINK_RE = re.compile(r"\[\[([^\]|#]+)(?:#[^\]|]+)?(?:\|[^\]]+)?\]\]")
HEADING_RE = re.compile(r"^(#{1,6})\s+(.*?)\s*$")
ANCHOR_RE = re.compile(r'<a\s+id="([^"]+)"\s*>\s*</a>', re.I)
FENCE_RE = re.compile(r"^\s*(```|~~~)")
def strip_code(text: str) -> str:
text = re.sub(r"```.*?```", "", text, flags=re.S)
return re.sub(r"`[^`\n]*`", "", text)
def links(text: str) -> list[str]:
return [m.group(1).strip() for m in LINK_RE.finditer(strip_code(text))]
def slugify(heading: str) -> str:
text = ANCHOR_RE.sub("", heading).strip().lower()
text = re.sub(r"[^\w\s-]", "", text)
return re.sub(r"[\s_]+", "-", text).strip("-").replace("--", "-")
@dataclass
class DocSection:
level: int # 0 = an anchor standing alone, not a heading
heading: str
anchors: set[str]
start: int # 0-based line of the heading (or of the lone anchor)
end: int # exclusive
def _lines(text: str) -> list[str]:
lines = text.replace("\r\n", "\n").split("\n")
while lines and not lines[-1].strip():
lines.pop()
return lines
def _is_lone_anchor(line: str) -> bool:
return bool(ANCHOR_RE.search(line)) and not ANCHOR_RE.sub("", line).strip()
def sections(text: str) -> list[DocSection]:
lines = _lines(text)
out: list[DocSection] = []
fenced = False
for n, line in enumerate(lines):
if FENCE_RE.match(line):
fenced = not fenced
if fenced or FENCE_RE.match(line):
continue
m = HEADING_RE.match(line)
tags = set(ANCHOR_RE.findall(line))
if m:
heading = ANCHOR_RE.sub("", m.group(2)).strip()
above = set(ANCHOR_RE.findall(lines[n - 1])) if n and _is_lone_anchor(lines[n - 1]) else set()
out.append(DocSection(len(m.group(1)), heading, {slugify(heading)} | tags | above, n, len(lines)))
elif tags and not (_is_lone_anchor(line) and n + 1 < len(lines) and HEADING_RE.match(lines[n + 1])):
out.append(DocSection(0, "", tags, n, len(lines)))
headings = [s for s in out if s.level]
for s in out:
nxt = next((h for h in headings if h.start > s.start and (s.level == 0 or h.level <= s.level)), None)
if nxt:
above = nxt.start - 1
s.end = above if above > s.start and _is_lone_anchor(lines[above]) else nxt.start
return out
def anchors(text: str) -> set[str]:
return {a for s in sections(text) for a in s.anchors}
def find(text: str, anchor: str) -> DocSection | None:
found = sections(text)
return next((s for s in found if anchor in s.anchors and s.level), None) or \
next((s for s in found if anchor in s.anchors), None)
def resolve(root: Path, path: str, anchor: str | None, limit: int = 80) -> str:
target = root / path
shown = path.rstrip("/")
if target.is_dir():
return f"{shown}/ (directory)"
if not target.is_file():
return f"(missing: {shown})"
text = target.read_text(encoding="utf-8", errors="replace")
lines = _lines(text)
if anchor is None:
heading = next((m.group(2) for l in lines if (m := HEADING_RE.match(l))), "")
goal = next((l[len("**Goal:**"):].strip() for l in lines if l.startswith("**Goal:**")), "")
return shown + (f": {heading}" if heading else "") + (f" — {goal}" if goal else "")
s = find(text, anchor)
if s is None:
return f"(no anchor '{anchor}' in {shown})"
body = lines[s.start:s.end]
while body and not body[-1].strip():
body.pop()
out = [f"===== {shown}#{anchor} (line {s.start + 1}) =====", *body[:limit]]
if len(body) > limit:
out.append(f"… ({len(body) - limit} more lines: {shown}:{s.start + limit + 1})")
return "\n".join(out)
|