diff options
Diffstat (limited to 'wflib/migrate.py')
| -rw-r--r-- | wflib/migrate.py | 237 |
1 files changed, 237 insertions, 0 deletions
diff --git a/wflib/migrate.py b/wflib/migrate.py new file mode 100644 index 0000000..04f606c --- /dev/null +++ b/wflib/migrate.py @@ -0,0 +1,237 @@ +"""One-time conversion of the numbered TASKS.md format to the id format. + +Old: `N. **[P1] Title** (Effort: 1h) (in progress: x) — goal.` + indented body, + `After: <title>`, `Reference: …`, Awaiting bullets without ids. +New: docs/design.md, section 3. Input already in the new format comes back unchanged. +""" +from __future__ import annotations + +import re +from dataclasses import dataclass, field + +from . import tasks + +NUMBERED_RE = re.compile(r"^\d+\. ") +OLD_HEADER_RE = re.compile(r"^\d+\. (?:N\. )?\*\*\[P(\d)\] (.+?)\*\*(.*)$") +SLICE_RE = re.compile(r"^(.+?) (\d+)/\d+(?::|$)") +MD_LINK_RE = re.compile(r"\[[^\]]*\]\(([^)\s]+)\)") +AFTER_RE = re.compile(r"^(\s*)(?:-\s+)?After:\s*(.*)$") +REFERENCE_RE = re.compile(r"^\s*(?:-\s+)?(?:Reference|Ref):\s*(.*)$") +BULLET_RE = re.compile(r"^- (?:\*\*(.+?)\*\*:?\s*)?(.*)$") + + +@dataclass +class Report: + ids: dict[str, str] = field(default_factory=dict) # old title -> new id + notes: list[str] = field(default_factory=list) + + +@dataclass +class _Old: + title: str + item: tasks.Item + after: list[str] + refs: str | None + + +def _groups(rest: str) -> tuple[list[str], str]: + """Leading balanced '(…)' groups of `rest`, and what follows them.""" + out, i = [], 0 + while True: + j = i + while j < len(rest) and rest[j] == " ": + j += 1 + if j >= len(rest) or rest[j] != "(": + return out, rest[i:] + depth, k = 0, j + while k < len(rest): + depth += rest[k] == "(" + depth -= rest[k] == ")" + k += 1 + if depth == 0: + break + if depth: + return out, rest[i:] + out.append(rest[j + 1:k - 1]) + i = k + + +def _id_title(title: str) -> str: + return re.sub(r"\s*\([^)]*\)", "", title).strip() or title + + +def _body(lines: list[str]) -> list[str]: + """Re-indented body. A line one column deeper than the top level is a + child when the line above it opens a list (ends with ':'), else a typo.""" + out: list[str] = [] + child = False + for line in tasks.indent_body(lines): + if re.match(r"^ \S", line): + line = (2 * tasks.INDENT if child else tasks.INDENT) + line[3:] + elif re.match(r"^ \S", line): + child = line.rstrip().endswith(":") + out.append(line) + return out + + +def _is_path(part: str) -> bool: + first = part.split(" ", 1)[0] + return "/" in first or bool(re.search(r"\.\w+(#|$)", first)) + + +def _refs(text: str) -> tuple[str | None, str | None]: + """(Ref: value, leftover words that name no file).""" + parts: list[str] = [] + for part in tasks.split_refs(MD_LINK_RE.sub(r"\1", text).replace(" · ", ", ").strip().rstrip(".")): + if _is_path(part): + parts.append(part) + elif parts: + last = parts[-1] + parts[-1] = f"{last[:-1]}; {part})" if last.endswith(")") else f"{last} ({part})" + else: + return None, text.strip() + return (", ".join(parts) or None), None + + +def _sentence(title: str, goal: str) -> str: + title = title.strip().rstrip(".").replace(". ", ", ") + return f"{title}. {goal.strip()}" if goal.strip() else f"{title}." + + +def _split_body(lines: list[str]) -> tuple[list[str], list[str], str | None]: + """(body re-indented, After: titles, Ref: value).""" + body, after, ref, notes = [], [], None, [] + for line in lines: + if (m := AFTER_RE.match(line)) and not tasks.LINK_RE.search(line): + after.append(m.group(2).strip()) + elif (m := REFERENCE_RE.match(line)): + ref, words = _refs(m.group(1)) + if words: + notes.append(f"{tasks.INDENT}- Reference: {words}") + else: + body.append(line) + return _body(body) + notes, after, ref + + +def _task(lines: list[str], taken: set[str]) -> _Old | None: + m = OLD_HEADER_RE.match(lines[0]) + if not m: + return None + title = m.group(2).strip() + groups, rest = _groups(m.group(3)) + effort, interactive, status, extra = None, False, None, [] + for g in groups: + if g.startswith("Effort:"): + parts = [p.strip() for p in g[len("Effort:"):].split(",")] + effort, interactive = parts[0], "interactive" in parts[1:] + elif g.startswith(("in progress:", "blocked:")): + status = g.replace("(", "[").replace(")", "]") + else: + extra.append(f"({g})") + goal = re.sub(r"^\s*[—–-]\s*", "", rest).strip() + goal = " ".join(extra + ([goal] if goal else [])) + slice_ = SLICE_RE.match(title) + if slice_: + base = tasks.make_id(_id_title(slice_.group(1)), set()) + id = f"{base}-{int(slice_.group(2))}" + n = 2 + while id in taken: + id, n = f"{base}-{int(slice_.group(2))}-{n}", n + 1 + else: + id = tasks.make_id(_id_title(title), taken) + body, after, ref = _split_body(lines[1:]) + item = tasks.Item(id=id, prio=int(m.group(1)), effort=effort, interactive=interactive, + status=status, text=_sentence(title, goal), body=body) + return _Old(title, item, after, ref) + + +def _awaiting(lines: list[str], taken: set[str]) -> _Old: + m = BULLET_RE.match(lines[0]) + bold, rest = m.group(1), m.group(2).strip() + if bold: + title, text = bold.strip(), _sentence(bold, rest) + else: + title = rest.split(". ", 1)[0].rstrip(".") + text = rest + id = tasks.make_id(_id_title(title), taken, "a-") + body, _, _ = _split_body(lines[1:]) + return _Old(title, tasks.Item(id=id, text=text, body=body), [], None) + + +def _is_new_item(line: str) -> bool: + m = re.match(r"^- \*\*([^*\s]+)\*\*", line) + return bool(m and tasks.ID_RE.match(m.group(1))) + + +def migrate(text: str, taken: set[str] | None = None) -> tuple[str, Report]: + taken = set(taken or ()) + report = Report() + newline = "\r\n" if "\r\n" in text else "\n" + lines = text.replace("\r\n", "\n").split("\n") + while lines and not lines[-1].strip(): + lines.pop() + taken |= {m.group(1) for l in lines if (m := re.match(r"^- \*\*([^*\s]+)\*\*", l)) and _is_new_item(l)} + + out: list[object] = [] # str lines and _Old items, in order + key = None + i = 0 + while i < len(lines): + line = lines[i] + if line.startswith("## "): + heading = line[3:].strip() + key = next((k for k, h in tasks.SECTIONS.items() if heading == h or heading.startswith(h + " (")), None) + if key and heading != tasks.SECTIONS[key]: + out += [f"## {tasks.SECTIONS[key]}", "", heading[len(tasks.SECTIONS[key]):].strip()] + else: + out.append(line) + i += 1 + continue + old_task = key in tasks.TASK_KEYS and NUMBERED_RE.match(line) + old_bullet = key == "awaiting" and line.startswith("- ") and not _is_new_item(line) + if not (old_task or old_bullet): + out.append(line) + i += 1 + continue + j = i + 1 + while j < len(lines) and (not lines[j].strip() or lines[j][0].isspace()): + j += 1 + block = lines[i:j] + while block and not block[-1].strip(): + block.pop() + old = _task(block, taken) if old_task else _awaiting(block, taken) + if old is None: + report.notes.append(f"line {i + 1}: numbered item not understood, left as it is: {line}") + out += lines[i:j] + else: + taken.add(old.item.id) + report.ids[old.title] = old.item.id + out += [old, ""] + i = j + + by_title = {t.lower().rstrip("."): id for t, id in report.ids.items()} + result: list[str] = [] + for entry in out: + if isinstance(entry, str): + result.append(entry) + continue + item, ids = entry.item, [] + for wanted in entry.after: + whole = by_title.get(wanted.lower().rstrip(".")) + parts = [whole] if whole else [by_title.get(p.strip().lower().rstrip(".")) for p in wanted.split("; ")] + if not any(parts): + report.notes.append(f"{item.id}: dropped 'After: {wanted}' (no open task with that title: done)") + ids += [p for p in parts if p] + if ids: + item.body.append(f"{tasks.INDENT}- After: " + ", ".join(f"[[{a}]]" for a in ids)) + if entry.refs: + item.body.append(f"{tasks.INDENT}Ref: {entry.refs}") + result += item.lines() + + doc = tasks.parse("\n".join(result) + "\n") + for key, heading in tasks.SECTIONS.items(): + if not any(s.key == key for s in doc.sections): + if doc.sections and doc.sections[-1].lines()[-1].strip(): + doc.sections[-1].suffix.append("") if doc.sections[-1].key else doc.sections[-1].prefix.append("") + doc.sections.append(tasks.Section(heading, key, prefix=[""])) + doc.newline = newline + return tasks.render(doc), report |
