workflow

git clone https://git.godosa.eu/workflow

master

raw · 3959 bytes

"""Links, anchors and doc sections: what `[[id]]` and `Ref: path#anchor` point at."""
from __future__ import annotations

import re
from dataclasses import dataclass
from pathlib import Path

LINK_RE = re.compile(r"\[\[([^\]|#]+)(?:#[^\]|]+)?(?:\|[^\]]+)?\]\]")
HEADING_RE = re.compile(r"^(#{1,6})\s+(.*?)\s*$")
ANCHOR_RE = re.compile(r'<a\s+id="([^"]+)"\s*>\s*</a>', re.I)
FENCE_RE = re.compile(r"^\s*(```|~~~)")


def strip_code(text: str) -> str:
    text = re.sub(r"```.*?```", "", text, flags=re.S)
    return re.sub(r"`[^`\n]*`", "", text)


def links(text: str) -> list[str]:
    return [m.group(1).strip() for m in LINK_RE.finditer(strip_code(text))]


def slugify(heading: str) -> str:
    text = ANCHOR_RE.sub("", heading).strip().lower()
    text = re.sub(r"[^\w\s-]", "", text)
    return re.sub(r"[\s_]+", "-", text).strip("-").replace("--", "-")


@dataclass
class DocSection:
    level: int          # 0 = an anchor standing alone, not a heading
    heading: str
    anchors: set[str]
    start: int          # 0-based line of the heading (or of the lone anchor)
    end: int            # exclusive


def _lines(text: str) -> list[str]:
    lines = text.replace("\r\n", "\n").split("\n")
    while lines and not lines[-1].strip():
        lines.pop()
    return lines


def _is_lone_anchor(line: str) -> bool:
    return bool(ANCHOR_RE.search(line)) and not ANCHOR_RE.sub("", line).strip()


def sections(text: str) -> list[DocSection]:
    lines = _lines(text)
    out: list[DocSection] = []
    fenced = False
    for n, line in enumerate(lines):
        if FENCE_RE.match(line):
            fenced = not fenced
        if fenced or FENCE_RE.match(line):
            continue
        m = HEADING_RE.match(line)
        tags = set(ANCHOR_RE.findall(line))
        if m:
            heading = ANCHOR_RE.sub("", m.group(2)).strip()
            above = set(ANCHOR_RE.findall(lines[n - 1])) if n and _is_lone_anchor(lines[n - 1]) else set()
            out.append(DocSection(len(m.group(1)), heading, {slugify(heading)} | tags | above, n, len(lines)))
        elif tags and not (_is_lone_anchor(line) and n + 1 < len(lines) and HEADING_RE.match(lines[n + 1])):
            out.append(DocSection(0, "", tags, n, len(lines)))
    headings = [s for s in out if s.level]
    for s in out:
        nxt = next((h for h in headings if h.start > s.start and (s.level == 0 or h.level <= s.level)), None)
        if nxt:
            above = nxt.start - 1
            s.end = above if above > s.start and _is_lone_anchor(lines[above]) else nxt.start
    return out


def anchors(text: str) -> set[str]:
    return {a for s in sections(text) for a in s.anchors}


def find(text: str, anchor: str) -> DocSection | None:
    found = sections(text)
    return next((s for s in found if anchor in s.anchors and s.level), None) or \
        next((s for s in found if anchor in s.anchors), None)


def resolve(root: Path, path: str, anchor: str | None, limit: int = 80) -> str:
    target = root / path
    shown = path.rstrip("/")
    if target.is_dir():
        return f"{shown}/ (directory)"
    if not target.is_file():
        return f"(missing: {shown})"
    text = target.read_text(encoding="utf-8", errors="replace")
    lines = _lines(text)
    if anchor is None:
        heading = next((m.group(2) for l in lines if (m := HEADING_RE.match(l))), "")
        goal = next((l[len("**Goal:**"):].strip() for l in lines if l.startswith("**Goal:**")), "")
        return shown + (f": {heading}" if heading else "") + (f" — {goal}" if goal else "")
    s = find(text, anchor)
    if s is None:
        return f"(no anchor '{anchor}' in {shown})"
    body = lines[s.start:s.end]
    while body and not body[-1].strip():
        body.pop()
    out = [f"===== {shown}#{anchor} (line {s.start + 1}) =====", *body[:limit]]
    if len(body) > limit:
        out.append(f"… ({len(body) - limit} more lines: {shown}:{s.start + limit + 1})")
    return "\n".join(out)