workflow

git clone https://git.godosa.eu/workflow

master

raw ยท 3227 bytes

"""Ranked word search over open tasks, doc sections and the archive.

A word matches at the start of a word, any case ("guard" finds "Guards").
Rank: more of the words found first, then the weight of where they were
found, then open tasks before docs before archive. No index: read per call.
"""
from __future__ import annotations

import re
from dataclasses import dataclass

from . import refs, tasks

W_TITLE, W_HEADING, W_GOAL, W_ARCHIVE, W_TEXT = 5, 4, 3, 2, 1
KINDS = ("task", "doc", "archive")


@dataclass
class Hit:
    found: int          # how many of the words
    score: int
    kind: str
    where: str          # id, or path:line
    label: str
    line: str           # the best matching line


def _patterns(words: list[str]) -> list[re.Pattern]:
    return [re.compile(r"(?<![A-Za-z0-9])" + re.escape(w), re.I) for w in words if w.strip()]


def _rate(patterns: list[re.Pattern], parts: list[tuple[int, list[str]]]) -> tuple[int, int, str]:
    """parts = (weight, lines). Returns (words found, score, best line)."""
    found = score = 0
    best, best_weight = "", 0
    for p in patterns:
        hit = max(((w, l) for w, lines in parts for l in lines if p.search(l)), key=lambda x: x[0], default=None)
        if hit:
            found += 1
            score += hit[0]
            if hit[0] > best_weight:
                best_weight, best = hit
    return found, score, " ".join(best.split())[:120]


def search(words: list[str], doc: tasks.Doc, archive_text: str, docs: dict[str, str],
           kinds: set[str] | None = None, limit: int = 15, archive_name: str = "archive") -> list[Hit]:
    patterns = _patterns(words)
    if not patterns:
        return []
    kinds = kinds or set(KINDS)
    hits: list[Hit] = []

    def add(kind, where, label, parts, header=None):
        found, score, line = _rate(patterns, parts)
        if found:
            in_header = header and any(line == " ".join(l.split())[:120] for w, ls in parts[:2] for l in ls)
            hits.append(Hit(found, score, kind, where, label, header if in_header else line))

    if "task" in kinds:
        for item in doc.all_items():
            if item.error:
                add("task", item.id, "", [(W_TEXT, [item.raw, *item.body])])
            else:
                add("task", item.id, item.title,
                    [(W_TITLE, [item.id, item.title]), (W_GOAL, [item.goal]), (W_TEXT, item.body)],
                    header=item.text)
    if "doc" in kinds:
        for path, text in docs.items():
            lines = text.replace("\r\n", "\n").split("\n")
            starts = [s for s in refs.sections(text) if s.level]
            for n, s in enumerate(starts):
                end = starts[n + 1].start if n + 1 < len(starts) else len(lines)
                add("doc", f"{path}:{s.start + 1}", s.heading,
                    [(W_HEADING, [s.heading]), (W_TEXT, lines[s.start + 1:end])])
    if "archive" in kinds:
        for n, line in enumerate(archive_text.replace("\r\n", "\n").split("\n"), 1):
            if line.startswith("- "):
                add("archive", f"{archive_name}:{n}", "", [(W_ARCHIVE, [line[2:]])])
    hits.sort(key=lambda h: (-h.found, -h.score, KINDS.index(h.kind)))
    return hits[:limit]