"""One-time conversion of the numbered TASKS.md format to the id format. Old: `N. **[P1] Title** (Effort: 1h) (in progress: x) — goal.` + indented body, `After: `, `Reference: …`, Awaiting bullets without ids. New: docs/design.md, section 3. Input already in the new format comes back unchanged. """ from __future__ import annotations import re from dataclasses import dataclass, field from . import tasks NUMBERED_RE = re.compile(r"^\d+\. ") OLD_HEADER_RE = re.compile(r"^\d+\. (?:N\. )?\*\*\[P(\d)\] (.+?)\*\*(.*)$") SLICE_RE = re.compile(r"^(.+?) (\d+)/\d+(?::|$)") MD_LINK_RE = re.compile(r"\[[^\]]*\]\(([^)\s]+)\)") AFTER_RE = re.compile(r"^(\s*)(?:-\s+)?After:\s*(.*)$") REFERENCE_RE = re.compile(r"^\s*(?:-\s+)?(?:Reference|Ref):\s*(.*)$") BULLET_RE = re.compile(r"^- (?:\*\*(.+?)\*\*:?\s*)?(.*)$") @dataclass class Report: ids: dict[str, str] = field(default_factory=dict) # old title -> new id notes: list[str] = field(default_factory=list) @dataclass class _Old: title: str item: tasks.Item after: list[str] refs: str | None def _groups(rest: str) -> tuple[list[str], str]: """Leading balanced '(…)' groups of `rest`, and what follows them.""" out, i = [], 0 while True: j = i while j < len(rest) and rest[j] == " ": j += 1 if j >= len(rest) or rest[j] != "(": return out, rest[i:] depth, k = 0, j while k < len(rest): depth += rest[k] == "(" depth -= rest[k] == ")" k += 1 if depth == 0: break if depth: return out, rest[i:] out.append(rest[j + 1:k - 1]) i = k def _id_title(title: str) -> str: return re.sub(r"\s*\([^)]*\)", "", title).strip() or title def _body(lines: list[str]) -> list[str]: """Re-indented body. A line one column deeper than the top level is a child when the line above it opens a list (ends with ':'), else a typo.""" out: list[str] = [] child = False for line in tasks.indent_body(lines): if re.match(r"^ \S", line): line = (2 * tasks.INDENT if child else tasks.INDENT) + line[3:] elif re.match(r"^ \S", line): child = line.rstrip().endswith(":") out.append(line) return out def _is_path(part: str) -> bool: first = part.split(" ", 1)[0] return "/" in first or bool(re.search(r"\.\w+(#|$)", first)) def _refs(text: str) -> tuple[str | None, str | None]: """(Ref: value, leftover words that name no file).""" parts: list[str] = [] for part in tasks.split_refs(MD_LINK_RE.sub(r"\1", text).replace(" · ", ", ").strip().rstrip(".")): if _is_path(part): parts.append(part) elif parts: last = parts[-1] parts[-1] = f"{last[:-1]}; {part})" if last.endswith(")") else f"{last} ({part})" else: return None, text.strip() return (", ".join(parts) or None), None def _sentence(title: str, goal: str) -> str: title = title.strip().rstrip(".").replace(". ", ", ") return f"{title}. {goal.strip()}" if goal.strip() else f"{title}." def _split_body(lines: list[str]) -> tuple[list[str], list[str], str | None]: """(body re-indented, After: titles, Ref: value).""" body, after, ref, notes = [], [], None, [] for line in lines: if (m := AFTER_RE.match(line)) and not tasks.LINK_RE.search(line): after.append(m.group(2).strip()) elif (m := REFERENCE_RE.match(line)): ref, words = _refs(m.group(1)) if words: notes.append(f"{tasks.INDENT}- Reference: {words}") else: body.append(line) return _body(body) + notes, after, ref def _task(lines: list[str], taken: set[str]) -> _Old | None: m = OLD_HEADER_RE.match(lines[0]) if not m: return None title = m.group(2).strip() groups, rest = _groups(m.group(3)) effort, interactive, status, extra = None, False, None, [] for g in groups: if g.startswith("Effort:"): parts = [p.strip() for p in g[len("Effort:"):].split(",")] effort, interactive = parts[0], "interactive" in parts[1:] elif g.startswith(("in progress:", "blocked:")): status = g.replace("(", "[").replace(")", "]") else: extra.append(f"({g})") goal = re.sub(r"^\s*[—–-]\s*", "", rest).strip() goal = " ".join(extra + ([goal] if goal else [])) slice_ = SLICE_RE.match(title) if slice_: base = tasks.make_id(_id_title(slice_.group(1)), set()) id = f"{base}-{int(slice_.group(2))}" n = 2 while id in taken: id, n = f"{base}-{int(slice_.group(2))}-{n}", n + 1 else: id = tasks.make_id(_id_title(title), taken) body, after, ref = _split_body(lines[1:]) item = tasks.Item(id=id, prio=int(m.group(1)), effort=effort, interactive=interactive, status=status, text=_sentence(title, goal), body=body) return _Old(title, item, after, ref) def _awaiting(lines: list[str], taken: set[str]) -> _Old: m = BULLET_RE.match(lines[0]) bold, rest = m.group(1), m.group(2).strip() if bold: title, text = bold.strip(), _sentence(bold, rest) else: title = rest.split(". ", 1)[0].rstrip(".") text = rest id = tasks.make_id(_id_title(title), taken, "a-") body, _, _ = _split_body(lines[1:]) return _Old(title, tasks.Item(id=id, text=text, body=body), [], None) def _is_new_item(line: str) -> bool: m = re.match(r"^- \*\*([^*\s]+)\*\*", line) return bool(m and tasks.ID_RE.match(m.group(1))) def migrate(text: str, taken: set[str] | None = None) -> tuple[str, Report]: taken = set(taken or ()) report = Report() newline = "\r\n" if "\r\n" in text else "\n" lines = text.replace("\r\n", "\n").split("\n") while lines and not lines[-1].strip(): lines.pop() taken |= {m.group(1) for l in lines if (m := re.match(r"^- \*\*([^*\s]+)\*\*", l)) and _is_new_item(l)} out: list[object] = [] # str lines and _Old items, in order key = None i = 0 while i < len(lines): line = lines[i] if line.startswith("## "): heading = line[3:].strip() key = next((k for k, h in tasks.SECTIONS.items() if heading == h or heading.startswith(h + " (")), None) if key and heading != tasks.SECTIONS[key]: out += [f"## {tasks.SECTIONS[key]}", "", heading[len(tasks.SECTIONS[key]):].strip()] else: out.append(line) i += 1 continue old_task = key in tasks.TASK_KEYS and NUMBERED_RE.match(line) old_bullet = key == "awaiting" and line.startswith("- ") and not _is_new_item(line) if not (old_task or old_bullet): out.append(line) i += 1 continue j = i + 1 while j < len(lines) and (not lines[j].strip() or lines[j][0].isspace()): j += 1 block = lines[i:j] while block and not block[-1].strip(): block.pop() old = _task(block, taken) if old_task else _awaiting(block, taken) if old is None: report.notes.append(f"line {i + 1}: numbered item not understood, left as it is: {line}") out += lines[i:j] else: taken.add(old.item.id) report.ids[old.title] = old.item.id out += [old, ""] i = j by_title = {t.lower().rstrip("."): id for t, id in report.ids.items()} result: list[str] = [] for entry in out: if isinstance(entry, str): result.append(entry) continue item, ids = entry.item, [] for wanted in entry.after: whole = by_title.get(wanted.lower().rstrip(".")) parts = [whole] if whole else [by_title.get(p.strip().lower().rstrip(".")) for p in wanted.split("; ")] if not any(parts): report.notes.append(f"{item.id}: dropped 'After: {wanted}' (no open task with that title: done)") ids += [p for p in parts if p] if ids: item.body.append(f"{tasks.INDENT}- After: " + ", ".join(f"[[{a}]]" for a in ids)) if entry.refs: item.body.append(f"{tasks.INDENT}Ref: {entry.refs}") result += item.lines() doc = tasks.parse("\n".join(result) + "\n") for key, heading in tasks.SECTIONS.items(): if not any(s.key == key for s in doc.sections): if doc.sections and doc.sections[-1].lines()[-1].strip(): doc.sections[-1].suffix.append("") if doc.sections[-1].key else doc.sections[-1].prefix.append("") doc.sections.append(tasks.Section(heading, key, prefix=[""])) doc.newline = newline return tasks.render(doc), report