aboutsummaryrefslogtreecommitdiffziptar.gz
path: root/wflib/migrate.py
blob: 04f606c79a3be3ab818e7373c84ea03dc42e5418 (plain)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
"""One-time conversion of the numbered TASKS.md format to the id format.

Old:  `N. **[P1] Title** (Effort: 1h) (in progress: x) — goal.` + indented body,
      `After: <title>`, `Reference: …`, Awaiting bullets without ids.
New:  docs/design.md, section 3. Input already in the new format comes back unchanged.
"""
from __future__ import annotations

import re
from dataclasses import dataclass, field

from . import tasks

NUMBERED_RE = re.compile(r"^\d+\. ")
OLD_HEADER_RE = re.compile(r"^\d+\. (?:N\. )?\*\*\[P(\d)\] (.+?)\*\*(.*)$")
SLICE_RE = re.compile(r"^(.+?) (\d+)/\d+(?::|$)")
MD_LINK_RE = re.compile(r"\[[^\]]*\]\(([^)\s]+)\)")
AFTER_RE = re.compile(r"^(\s*)(?:-\s+)?After:\s*(.*)$")
REFERENCE_RE = re.compile(r"^\s*(?:-\s+)?(?:Reference|Ref):\s*(.*)$")
BULLET_RE = re.compile(r"^- (?:\*\*(.+?)\*\*:?\s*)?(.*)$")


@dataclass
class Report:
    ids: dict[str, str] = field(default_factory=dict)       # old title -> new id
    notes: list[str] = field(default_factory=list)


@dataclass
class _Old:
    title: str
    item: tasks.Item
    after: list[str]
    refs: str | None


def _groups(rest: str) -> tuple[list[str], str]:
    """Leading balanced '(…)' groups of `rest`, and what follows them."""
    out, i = [], 0
    while True:
        j = i
        while j < len(rest) and rest[j] == " ":
            j += 1
        if j >= len(rest) or rest[j] != "(":
            return out, rest[i:]
        depth, k = 0, j
        while k < len(rest):
            depth += rest[k] == "("
            depth -= rest[k] == ")"
            k += 1
            if depth == 0:
                break
        if depth:
            return out, rest[i:]
        out.append(rest[j + 1:k - 1])
        i = k


def _id_title(title: str) -> str:
    return re.sub(r"\s*\([^)]*\)", "", title).strip() or title


def _body(lines: list[str]) -> list[str]:
    """Re-indented body. A line one column deeper than the top level is a
    child when the line above it opens a list (ends with ':'), else a typo."""
    out: list[str] = []
    child = False
    for line in tasks.indent_body(lines):
        if re.match(r"^   \S", line):
            line = (2 * tasks.INDENT if child else tasks.INDENT) + line[3:]
        elif re.match(r"^  \S", line):
            child = line.rstrip().endswith(":")
        out.append(line)
    return out


def _is_path(part: str) -> bool:
    first = part.split(" ", 1)[0]
    return "/" in first or bool(re.search(r"\.\w+(#|$)", first))


def _refs(text: str) -> tuple[str | None, str | None]:
    """(Ref: value, leftover words that name no file)."""
    parts: list[str] = []
    for part in tasks.split_refs(MD_LINK_RE.sub(r"\1", text).replace(" · ", ", ").strip().rstrip(".")):
        if _is_path(part):
            parts.append(part)
        elif parts:
            last = parts[-1]
            parts[-1] = f"{last[:-1]}; {part})" if last.endswith(")") else f"{last} ({part})"
        else:
            return None, text.strip()
    return (", ".join(parts) or None), None


def _sentence(title: str, goal: str) -> str:
    title = title.strip().rstrip(".").replace(". ", ", ")
    return f"{title}. {goal.strip()}" if goal.strip() else f"{title}."


def _split_body(lines: list[str]) -> tuple[list[str], list[str], str | None]:
    """(body re-indented, After: titles, Ref: value)."""
    body, after, ref, notes = [], [], None, []
    for line in lines:
        if (m := AFTER_RE.match(line)) and not tasks.LINK_RE.search(line):
            after.append(m.group(2).strip())
        elif (m := REFERENCE_RE.match(line)):
            ref, words = _refs(m.group(1))
            if words:
                notes.append(f"{tasks.INDENT}- Reference: {words}")
        else:
            body.append(line)
    return _body(body) + notes, after, ref


def _task(lines: list[str], taken: set[str]) -> _Old | None:
    m = OLD_HEADER_RE.match(lines[0])
    if not m:
        return None
    title = m.group(2).strip()
    groups, rest = _groups(m.group(3))
    effort, interactive, status, extra = None, False, None, []
    for g in groups:
        if g.startswith("Effort:"):
            parts = [p.strip() for p in g[len("Effort:"):].split(",")]
            effort, interactive = parts[0], "interactive" in parts[1:]
        elif g.startswith(("in progress:", "blocked:")):
            status = g.replace("(", "[").replace(")", "]")
        else:
            extra.append(f"({g})")
    goal = re.sub(r"^\s*[—–-]\s*", "", rest).strip()
    goal = " ".join(extra + ([goal] if goal else []))
    slice_ = SLICE_RE.match(title)
    if slice_:
        base = tasks.make_id(_id_title(slice_.group(1)), set())
        id = f"{base}-{int(slice_.group(2))}"
        n = 2
        while id in taken:
            id, n = f"{base}-{int(slice_.group(2))}-{n}", n + 1
    else:
        id = tasks.make_id(_id_title(title), taken)
    body, after, ref = _split_body(lines[1:])
    item = tasks.Item(id=id, prio=int(m.group(1)), effort=effort, interactive=interactive,
                      status=status, text=_sentence(title, goal), body=body)
    return _Old(title, item, after, ref)


def _awaiting(lines: list[str], taken: set[str]) -> _Old:
    m = BULLET_RE.match(lines[0])
    bold, rest = m.group(1), m.group(2).strip()
    if bold:
        title, text = bold.strip(), _sentence(bold, rest)
    else:
        title = rest.split(". ", 1)[0].rstrip(".")
        text = rest
    id = tasks.make_id(_id_title(title), taken, "a-")
    body, _, _ = _split_body(lines[1:])
    return _Old(title, tasks.Item(id=id, text=text, body=body), [], None)


def _is_new_item(line: str) -> bool:
    m = re.match(r"^- \*\*([^*\s]+)\*\*", line)
    return bool(m and tasks.ID_RE.match(m.group(1)))


def migrate(text: str, taken: set[str] | None = None) -> tuple[str, Report]:
    taken = set(taken or ())
    report = Report()
    newline = "\r\n" if "\r\n" in text else "\n"
    lines = text.replace("\r\n", "\n").split("\n")
    while lines and not lines[-1].strip():
        lines.pop()
    taken |= {m.group(1) for l in lines if (m := re.match(r"^- \*\*([^*\s]+)\*\*", l)) and _is_new_item(l)}

    out: list[object] = []          # str lines and _Old items, in order
    key = None
    i = 0
    while i < len(lines):
        line = lines[i]
        if line.startswith("## "):
            heading = line[3:].strip()
            key = next((k for k, h in tasks.SECTIONS.items() if heading == h or heading.startswith(h + " (")), None)
            if key and heading != tasks.SECTIONS[key]:
                out += [f"## {tasks.SECTIONS[key]}", "", heading[len(tasks.SECTIONS[key]):].strip()]
            else:
                out.append(line)
            i += 1
            continue
        old_task = key in tasks.TASK_KEYS and NUMBERED_RE.match(line)
        old_bullet = key == "awaiting" and line.startswith("- ") and not _is_new_item(line)
        if not (old_task or old_bullet):
            out.append(line)
            i += 1
            continue
        j = i + 1
        while j < len(lines) and (not lines[j].strip() or lines[j][0].isspace()):
            j += 1
        block = lines[i:j]
        while block and not block[-1].strip():
            block.pop()
        old = _task(block, taken) if old_task else _awaiting(block, taken)
        if old is None:
            report.notes.append(f"line {i + 1}: numbered item not understood, left as it is: {line}")
            out += lines[i:j]
        else:
            taken.add(old.item.id)
            report.ids[old.title] = old.item.id
            out += [old, ""]
        i = j

    by_title = {t.lower().rstrip("."): id for t, id in report.ids.items()}
    result: list[str] = []
    for entry in out:
        if isinstance(entry, str):
            result.append(entry)
            continue
        item, ids = entry.item, []
        for wanted in entry.after:
            whole = by_title.get(wanted.lower().rstrip("."))
            parts = [whole] if whole else [by_title.get(p.strip().lower().rstrip(".")) for p in wanted.split("; ")]
            if not any(parts):
                report.notes.append(f"{item.id}: dropped 'After: {wanted}' (no open task with that title: done)")
            ids += [p for p in parts if p]
        if ids:
            item.body.append(f"{tasks.INDENT}- After: " + ", ".join(f"[[{a}]]" for a in ids))
        if entry.refs:
            item.body.append(f"{tasks.INDENT}Ref: {entry.refs}")
        result += item.lines()

    doc = tasks.parse("\n".join(result) + "\n")
    for key, heading in tasks.SECTIONS.items():
        if not any(s.key == key for s in doc.sections):
            if doc.sections and doc.sections[-1].lines()[-1].strip():
                doc.sections[-1].suffix.append("") if doc.sections[-1].key else doc.sections[-1].prefix.append("")
            doc.sections.append(tasks.Section(heading, key, prefix=[""]))
    doc.newline = newline
    return tasks.render(doc), report