"""One-time conversion of the numbered TASKS.md format to the id format.
Old: `N. **[P1] Title** (Effort: 1h) (in progress: x) — goal.` + indented body,
`After:
`, `Reference: …`, Awaiting bullets without ids.
New: docs/design.md, section 3. Input already in the new format comes back unchanged.
"""
from __future__ import annotations
import re
from dataclasses import dataclass, field
from . import tasks
NUMBERED_RE = re.compile(r"^\d+\. ")
OLD_HEADER_RE = re.compile(r"^\d+\. (?:N\. )?\*\*\[P(\d)\] (.+?)\*\*(.*)$")
SLICE_RE = re.compile(r"^(.+?) (\d+)/\d+(?::|$)")
MD_LINK_RE = re.compile(r"\[[^\]]*\]\(([^)\s]+)\)")
AFTER_RE = re.compile(r"^(\s*)(?:-\s+)?After:\s*(.*)$")
REFERENCE_RE = re.compile(r"^\s*(?:-\s+)?(?:Reference|Ref):\s*(.*)$")
BULLET_RE = re.compile(r"^- (?:\*\*(.+?)\*\*:?\s*)?(.*)$")
@dataclass
class Report:
ids: dict[str, str] = field(default_factory=dict) # old title -> new id
notes: list[str] = field(default_factory=list)
@dataclass
class _Old:
title: str
item: tasks.Item
after: list[str]
refs: str | None
def _groups(rest: str) -> tuple[list[str], str]:
"""Leading balanced '(…)' groups of `rest`, and what follows them."""
out, i = [], 0
while True:
j = i
while j < len(rest) and rest[j] == " ":
j += 1
if j >= len(rest) or rest[j] != "(":
return out, rest[i:]
depth, k = 0, j
while k < len(rest):
depth += rest[k] == "("
depth -= rest[k] == ")"
k += 1
if depth == 0:
break
if depth:
return out, rest[i:]
out.append(rest[j + 1:k - 1])
i = k
def _id_title(title: str) -> str:
return re.sub(r"\s*\([^)]*\)", "", title).strip() or title
def _body(lines: list[str]) -> list[str]:
"""Re-indented body. A line one column deeper than the top level is a
child when the line above it opens a list (ends with ':'), else a typo."""
out: list[str] = []
child = False
for line in tasks.indent_body(lines):
if re.match(r"^ \S", line):
line = (2 * tasks.INDENT if child else tasks.INDENT) + line[3:]
elif re.match(r"^ \S", line):
child = line.rstrip().endswith(":")
out.append(line)
return out
def _is_path(part: str) -> bool:
first = part.split(" ", 1)[0]
return "/" in first or bool(re.search(r"\.\w+(#|$)", first))
def _refs(text: str) -> tuple[str | None, str | None]:
"""(Ref: value, leftover words that name no file)."""
parts: list[str] = []
for part in tasks.split_refs(MD_LINK_RE.sub(r"\1", text).replace(" · ", ", ").strip().rstrip(".")):
if _is_path(part):
parts.append(part)
elif parts:
last = parts[-1]
parts[-1] = f"{last[:-1]}; {part})" if last.endswith(")") else f"{last} ({part})"
else:
return None, text.strip()
return (", ".join(parts) or None), None
def _sentence(title: str, goal: str) -> str:
title = title.strip().rstrip(".").replace(". ", ", ")
return f"{title}. {goal.strip()}" if goal.strip() else f"{title}."
def _split_body(lines: list[str]) -> tuple[list[str], list[str], str | None]:
"""(body re-indented, After: titles, Ref: value)."""
body, after, ref, notes = [], [], None, []
for line in lines:
if (m := AFTER_RE.match(line)) and not tasks.LINK_RE.search(line):
after.append(m.group(2).strip())
elif (m := REFERENCE_RE.match(line)):
ref, words = _refs(m.group(1))
if words:
notes.append(f"{tasks.INDENT}- Reference: {words}")
else:
body.append(line)
return _body(body) + notes, after, ref
def _task(lines: list[str], taken: set[str]) -> _Old | None:
m = OLD_HEADER_RE.match(lines[0])
if not m:
return None
title = m.group(2).strip()
groups, rest = _groups(m.group(3))
effort, interactive, status, extra = None, False, None, []
for g in groups:
if g.startswith("Effort:"):
parts = [p.strip() for p in g[len("Effort:"):].split(",")]
effort, interactive = parts[0], "interactive" in parts[1:]
elif g.startswith(("in progress:", "blocked:")):
status = g.replace("(", "[").replace(")", "]")
else:
extra.append(f"({g})")
goal = re.sub(r"^\s*[—–-]\s*", "", rest).strip()
goal = " ".join(extra + ([goal] if goal else []))
slice_ = SLICE_RE.match(title)
if slice_:
base = tasks.make_id(_id_title(slice_.group(1)), set())
id = f"{base}-{int(slice_.group(2))}"
n = 2
while id in taken:
id, n = f"{base}-{int(slice_.group(2))}-{n}", n + 1
else:
id = tasks.make_id(_id_title(title), taken)
body, after, ref = _split_body(lines[1:])
item = tasks.Item(id=id, prio=int(m.group(1)), effort=effort, interactive=interactive,
status=status, text=_sentence(title, goal), body=body)
return _Old(title, item, after, ref)
def _awaiting(lines: list[str], taken: set[str]) -> _Old:
m = BULLET_RE.match(lines[0])
bold, rest = m.group(1), m.group(2).strip()
if bold:
title, text = bold.strip(), _sentence(bold, rest)
else:
title = rest.split(". ", 1)[0].rstrip(".")
text = rest
id = tasks.make_id(_id_title(title), taken, "a-")
body, _, _ = _split_body(lines[1:])
return _Old(title, tasks.Item(id=id, text=text, body=body), [], None)
def _is_new_item(line: str) -> bool:
m = re.match(r"^- \*\*([^*\s]+)\*\*", line)
return bool(m and tasks.ID_RE.match(m.group(1)))
def migrate(text: str, taken: set[str] | None = None) -> tuple[str, Report]:
taken = set(taken or ())
report = Report()
newline = "\r\n" if "\r\n" in text else "\n"
lines = text.replace("\r\n", "\n").split("\n")
while lines and not lines[-1].strip():
lines.pop()
taken |= {m.group(1) for l in lines if (m := re.match(r"^- \*\*([^*\s]+)\*\*", l)) and _is_new_item(l)}
out: list[object] = [] # str lines and _Old items, in order
key = None
i = 0
while i < len(lines):
line = lines[i]
if line.startswith("## "):
heading = line[3:].strip()
key = next((k for k, h in tasks.SECTIONS.items() if heading == h or heading.startswith(h + " (")), None)
if key and heading != tasks.SECTIONS[key]:
out += [f"## {tasks.SECTIONS[key]}", "", heading[len(tasks.SECTIONS[key]):].strip()]
else:
out.append(line)
i += 1
continue
old_task = key in tasks.TASK_KEYS and NUMBERED_RE.match(line)
old_bullet = key == "awaiting" and line.startswith("- ") and not _is_new_item(line)
if not (old_task or old_bullet):
out.append(line)
i += 1
continue
j = i + 1
while j < len(lines) and (not lines[j].strip() or lines[j][0].isspace()):
j += 1
block = lines[i:j]
while block and not block[-1].strip():
block.pop()
old = _task(block, taken) if old_task else _awaiting(block, taken)
if old is None:
report.notes.append(f"line {i + 1}: numbered item not understood, left as it is: {line}")
out += lines[i:j]
else:
taken.add(old.item.id)
report.ids[old.title] = old.item.id
out += [old, ""]
i = j
by_title = {t.lower().rstrip("."): id for t, id in report.ids.items()}
result: list[str] = []
for entry in out:
if isinstance(entry, str):
result.append(entry)
continue
item, ids = entry.item, []
for wanted in entry.after:
whole = by_title.get(wanted.lower().rstrip("."))
parts = [whole] if whole else [by_title.get(p.strip().lower().rstrip(".")) for p in wanted.split("; ")]
if not any(parts):
report.notes.append(f"{item.id}: dropped 'After: {wanted}' (no open task with that title: done)")
ids += [p for p in parts if p]
if ids:
item.body.append(f"{tasks.INDENT}- After: " + ", ".join(f"[[{a}]]" for a in ids))
if entry.refs:
item.body.append(f"{tasks.INDENT}Ref: {entry.refs}")
result += item.lines()
doc = tasks.parse("\n".join(result) + "\n")
for key, heading in tasks.SECTIONS.items():
if not any(s.key == key for s in doc.sections):
if doc.sections and doc.sections[-1].lines()[-1].strip():
doc.sections[-1].suffix.append("") if doc.sections[-1].key else doc.sections[-1].prefix.append("")
doc.sections.append(tasks.Section(heading, key, prefix=[""]))
doc.newline = newline
return tasks.render(doc), report