workflow

git clone https://git.godosa.eu/workflow

master

raw · 8904 bytes

"""One-time conversion of the numbered TASKS.md format to the id format.

Old:  `N. **[P1] Title** (Effort: 1h) (in progress: x) — goal.` + indented body,
      `After: <title>`, `Reference: …`, Awaiting bullets without ids.
New:  docs/design.md, section 3. Input already in the new format comes back unchanged.
"""
from __future__ import annotations

import re
from dataclasses import dataclass, field

from . import tasks

NUMBERED_RE = re.compile(r"^\d+\. ")
OLD_HEADER_RE = re.compile(r"^\d+\. (?:N\. )?\*\*\[P(\d)\] (.+?)\*\*(.*)$")
SLICE_RE = re.compile(r"^(.+?) (\d+)/\d+(?::|$)")
MD_LINK_RE = re.compile(r"\[[^\]]*\]\(([^)\s]+)\)")
AFTER_RE = re.compile(r"^(\s*)(?:-\s+)?After:\s*(.*)$")
REFERENCE_RE = re.compile(r"^\s*(?:-\s+)?(?:Reference|Ref):\s*(.*)$")
BULLET_RE = re.compile(r"^- (?:\*\*(.+?)\*\*:?\s*)?(.*)$")


@dataclass
class Report:
    ids: dict[str, str] = field(default_factory=dict)       # old title -> new id
    notes: list[str] = field(default_factory=list)


@dataclass
class _Old:
    title: str
    item: tasks.Item
    after: list[str]
    refs: str | None


def _groups(rest: str) -> tuple[list[str], str]:
    """Leading balanced '(…)' groups of `rest`, and what follows them."""
    out, i = [], 0
    while True:
        j = i
        while j < len(rest) and rest[j] == " ":
            j += 1
        if j >= len(rest) or rest[j] != "(":
            return out, rest[i:]
        depth, k = 0, j
        while k < len(rest):
            depth += rest[k] == "("
            depth -= rest[k] == ")"
            k += 1
            if depth == 0:
                break
        if depth:
            return out, rest[i:]
        out.append(rest[j + 1:k - 1])
        i = k


def _id_title(title: str) -> str:
    return re.sub(r"\s*\([^)]*\)", "", title).strip() or title


def _body(lines: list[str]) -> list[str]:
    """Re-indented body. A line one column deeper than the top level is a
    child when the line above it opens a list (ends with ':'), else a typo."""
    out: list[str] = []
    child = False
    for line in tasks.indent_body(lines):
        if re.match(r"^   \S", line):
            line = (2 * tasks.INDENT if child else tasks.INDENT) + line[3:]
        elif re.match(r"^  \S", line):
            child = line.rstrip().endswith(":")
        out.append(line)
    return out


def _is_path(part: str) -> bool:
    first = part.split(" ", 1)[0]
    return "/" in first or bool(re.search(r"\.\w+(#|$)", first))


def _refs(text: str) -> tuple[str | None, str | None]:
    """(Ref: value, leftover words that name no file)."""
    parts: list[str] = []
    for part in tasks.split_refs(MD_LINK_RE.sub(r"\1", text).replace(" · ", ", ").strip().rstrip(".")):
        if _is_path(part):
            parts.append(part)
        elif parts:
            last = parts[-1]
            parts[-1] = f"{last[:-1]}; {part})" if last.endswith(")") else f"{last} ({part})"
        else:
            return None, text.strip()
    return (", ".join(parts) or None), None


def _sentence(title: str, goal: str) -> str:
    title = title.strip().rstrip(".").replace(". ", ", ")
    return f"{title}. {goal.strip()}" if goal.strip() else f"{title}."


def _split_body(lines: list[str]) -> tuple[list[str], list[str], str | None]:
    """(body re-indented, After: titles, Ref: value)."""
    body, after, ref, notes = [], [], None, []
    for line in lines:
        if (m := AFTER_RE.match(line)) and not tasks.LINK_RE.search(line):
            after.append(m.group(2).strip())
        elif (m := REFERENCE_RE.match(line)):
            ref, words = _refs(m.group(1))
            if words:
                notes.append(f"{tasks.INDENT}- Reference: {words}")
        else:
            body.append(line)
    return _body(body) + notes, after, ref


def _task(lines: list[str], taken: set[str]) -> _Old | None:
    m = OLD_HEADER_RE.match(lines[0])
    if not m:
        return None
    title = m.group(2).strip()
    groups, rest = _groups(m.group(3))
    effort, interactive, status, extra = None, False, None, []
    for g in groups:
        if g.startswith("Effort:"):
            parts = [p.strip() for p in g[len("Effort:"):].split(",")]
            effort, interactive = parts[0], "interactive" in parts[1:]
        elif g.startswith(("in progress:", "blocked:")):
            status = g.replace("(", "[").replace(")", "]")
        else:
            extra.append(f"({g})")
    goal = re.sub(r"^\s*[—–-]\s*", "", rest).strip()
    goal = " ".join(extra + ([goal] if goal else []))
    slice_ = SLICE_RE.match(title)
    if slice_:
        base = tasks.make_id(_id_title(slice_.group(1)), set())
        id = f"{base}-{int(slice_.group(2))}"
        n = 2
        while id in taken:
            id, n = f"{base}-{int(slice_.group(2))}-{n}", n + 1
    else:
        id = tasks.make_id(_id_title(title), taken)
    body, after, ref = _split_body(lines[1:])
    item = tasks.Item(id=id, prio=int(m.group(1)), effort=effort, interactive=interactive,
                      status=status, text=_sentence(title, goal), body=body)
    return _Old(title, item, after, ref)


def _awaiting(lines: list[str], taken: set[str]) -> _Old:
    m = BULLET_RE.match(lines[0])
    bold, rest = m.group(1), m.group(2).strip()
    if bold:
        title, text = bold.strip(), _sentence(bold, rest)
    else:
        title = rest.split(". ", 1)[0].rstrip(".")
        text = rest
    id = tasks.make_id(_id_title(title), taken, "a-")
    body, _, _ = _split_body(lines[1:])
    return _Old(title, tasks.Item(id=id, text=text, body=body), [], None)


def _is_new_item(line: str) -> bool:
    m = re.match(r"^- \*\*([^*\s]+)\*\*", line)
    return bool(m and tasks.ID_RE.match(m.group(1)))


def migrate(text: str, taken: set[str] | None = None) -> tuple[str, Report]:
    taken = set(taken or ())
    report = Report()
    newline = "\r\n" if "\r\n" in text else "\n"
    lines = text.replace("\r\n", "\n").split("\n")
    while lines and not lines[-1].strip():
        lines.pop()
    taken |= {m.group(1) for l in lines if (m := re.match(r"^- \*\*([^*\s]+)\*\*", l)) and _is_new_item(l)}

    out: list[object] = []          # str lines and _Old items, in order
    key = None
    i = 0
    while i < len(lines):
        line = lines[i]
        if line.startswith("## "):
            heading = line[3:].strip()
            key = next((k for k, h in tasks.SECTIONS.items() if heading == h or heading.startswith(h + " (")), None)
            if key and heading != tasks.SECTIONS[key]:
                out += [f"## {tasks.SECTIONS[key]}", "", heading[len(tasks.SECTIONS[key]):].strip()]
            else:
                out.append(line)
            i += 1
            continue
        old_task = key in tasks.TASK_KEYS and NUMBERED_RE.match(line)
        old_bullet = key == "awaiting" and line.startswith("- ") and not _is_new_item(line)
        if not (old_task or old_bullet):
            out.append(line)
            i += 1
            continue
        j = i + 1
        while j < len(lines) and (not lines[j].strip() or lines[j][0].isspace()):
            j += 1
        block = lines[i:j]
        while block and not block[-1].strip():
            block.pop()
        old = _task(block, taken) if old_task else _awaiting(block, taken)
        if old is None:
            report.notes.append(f"line {i + 1}: numbered item not understood, left as it is: {line}")
            out += lines[i:j]
        else:
            taken.add(old.item.id)
            report.ids[old.title] = old.item.id
            out += [old, ""]
        i = j

    by_title = {t.lower().rstrip("."): id for t, id in report.ids.items()}
    result: list[str] = []
    for entry in out:
        if isinstance(entry, str):
            result.append(entry)
            continue
        item, ids = entry.item, []
        for wanted in entry.after:
            whole = by_title.get(wanted.lower().rstrip("."))
            parts = [whole] if whole else [by_title.get(p.strip().lower().rstrip(".")) for p in wanted.split("; ")]
            if not any(parts):
                report.notes.append(f"{item.id}: dropped 'After: {wanted}' (no open task with that title: done)")
            ids += [p for p in parts if p]
        if ids:
            item.body.append(f"{tasks.INDENT}- After: " + ", ".join(f"[[{a}]]" for a in ids))
        if entry.refs:
            item.body.append(f"{tasks.INDENT}Ref: {entry.refs}")
        result += item.lines()

    doc = tasks.parse("\n".join(result) + "\n")
    for key, heading in tasks.SECTIONS.items():
        if not any(s.key == key for s in doc.sections):
            if doc.sections and doc.sections[-1].lines()[-1].strip():
                doc.sections[-1].suffix.append("") if doc.sections[-1].key else doc.sections[-1].prefix.append("")
            doc.sections.append(tasks.Section(heading, key, prefix=[""]))
    doc.newline = newline
    return tasks.render(doc), report