diff --git a/CHANGELOG.md b/CHANGELOG.md index 1419762..811eaf3 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -41,7 +41,9 @@ An additive change to the public API; it leaves the emitted C++ unchanged. argument kinds and a one-line explanation; the templates render the `message` and `hint` byte for byte, which keep their text. Codes are never reused (`tests/fixtures/diagnostic_codes_pin.json`). See - [Diagnostic codes](docs/PUBLIC_CONTRACT.md#diagnostic-codes). + [Diagnostic codes](docs/PUBLIC_CONTRACT.md#diagnostic-codes). A code is + read off its text in time linear in the text: a crafted script's text cannot + stall the classification, which a backtracking regex let it do. - **First error in source order.** Since 1.1.0 the settings metadata visited every input's `defval`, `options`, `minval`, `maxval` and `step` ahead of the script body, so an error there (an unknown name in `minval`) was raised diff --git a/pineforge_codegen/diagnostic_codes.py b/pineforge_codegen/diagnostic_codes.py index b476a45..da19448 100644 --- a/pineforge_codegen/diagnostic_codes.py +++ b/pineforge_codegen/diagnostic_codes.py @@ -14,6 +14,16 @@ never guessed: a text no template renders gets the uncatalogued code of its severity (``PF-E0000`` / ``PF-W0000``), which the test suite refuses. +The match never backtracks over the text's characters: each literal segment +of a template goes to its leftmost place after the previous one, the last to +the text's end, and every argument is the text between its literals -- the +split a fullmatch of the template with lazy ``(.*?)`` arguments returns. Where +an argument is named twice (``{receiver}`` in the message and the hint) the +leftmost split can name it two values; then the later places of a literal are +tried in order, as the regex's backtracking tries them, within a fixed budget +of steps. The regex took seconds, then minutes, as a crafted message grew, +and a user's script spells argument text. + Rendering (:func:`render`) is the ICU MessageFormat subset the catalog uses: literal text with ICU apostrophe quoting (``''`` is one apostrophe, ``'{'`` a literal brace) and simple ``{name}`` arguments, a string argument @@ -158,47 +168,133 @@ def render_diagnostic(code: str, args: dict) -> tuple[str, str | None]: # Classification # --------------------------------------------------------------------------- -# Joins a message and its hint into one subject, so an argument both name is -# one value (a backreference). No template spells it. -_JOIN = "\x00" _CANONICAL_INT = re.compile(r"-?(?:0|[1-9][0-9]*)") _CANONICAL_FLOAT = re.compile(r"-?(?:0|[1-9][0-9]*)\.[0-9]+") +def _segments(parts: list) -> tuple[str, tuple[tuple[str, str], ...]]: + """A parsed template as its leading literal and ``(argument, literal after + it)`` pairs; the literal after an argument may be empty.""" + head = "" + pairs: list[list[str]] = [] + for part in parts: + if isinstance(part, str): + if pairs: + pairs[-1][1] += part + else: + head += part + else: + pairs.append([part[0], ""]) + return head, tuple((name, literal) for name, literal in pairs) + + +# Literal places a template whose argument is named twice may try, per +# classification: a script's text never needs more than a few, and a crafted +# one gets the uncatalogued code instead of a long search. +_SEARCH_BUDGET = 4096 + + +def _search(segments: list, texts: list[str], repeats: bool) -> dict[str, str] | None: + """Split ``texts`` (the message, then the hint) by their templates' + ``segments``: each argument ends at the leftmost place of the literal after + it, the last argument of a text at its end -- the split a lazy-regex + fullmatch returns. Without an argument named twice (``repeats``) that split + succeeds whenever any does, since a leftmost place leaves the rest the most + room: it is the only one tried, so the work is linear. With one, a split can + give the name two values; then the later places of a literal are tried in + order, as the regex backtracks, within ``_SEARCH_BUDGET`` places.""" + starts: list[int] = [] + ends: list[int] = [] + slots: list[tuple[int, str, str, bool]] = [] # text, argument, literal after it, last + for index, ((head, pairs), text) in enumerate(zip(segments, texts)): + if not text.startswith(head): + return None + if not pairs: + if len(text) != len(head): + return None + starts.append(len(head)) + ends.append(len(head)) + continue + tail = pairs[-1][1] + end = len(text) - len(tail) + if end < len(head) or not text.endswith(tail): + return None + starts.append(len(head)) + ends.append(end) + for at, (name, literal) in enumerate(pairs): + slots.append((index, name, literal, at == len(pairs) - 1)) + found: dict[str, str] = {} + budget = [_SEARCH_BUDGET] + + def step(slot: int, pos: int) -> bool: + if slot == len(slots): + return True + text_index, name, literal, last = slots[slot] + text, end = texts[text_index], ends[text_index] + following = slot + 1 + known = found.get(name) + if last: + value = text[pos:end] + resume = (starts[slots[following][0]] if following < len(slots) else end) + if known is not None: + return known == value and step(following, resume) + found[name] = value + if step(following, resume): + return True + del found[name] + return False + if known is not None: + stop = pos + len(known) + return (text.startswith(known, pos) and stop + len(literal) <= end + and text.startswith(literal, stop) and step(following, stop + len(literal))) + at = text.find(literal, pos, end) + while at >= 0: + budget[0] -= 1 + if budget[0] < 0: + return False + found[name] = text[pos:at] + if step(following, at + len(literal)): + return True + del found[name] + if not repeats: + return False + at = text.find(literal, at + 1, end) if at < end else -1 + return False + + first = starts[slots[0][0]] if slots else 0 + return found if step(0, first) else None + + class _Matcher: __slots__ = ("code", "has_hint", "prefix", "suffix", "specificity", - "_parts", "_regex", "_kinds") + "_message", "_hint", "_kinds", "_repeats") def __init__(self, code: str, entry: dict): self.code = code message = parse_template(entry["message"]) hint = entry.get("hint") + hint_parts = parse_template(hint) if hint is not None else [] self.has_hint = hint is not None - self._parts = message + ([_JOIN] + parse_template(hint) if hint is not None else []) + self._message = _segments(message) + self._hint = _segments(hint_parts) if hint is not None else None self.prefix = message[0] if message and isinstance(message[0], str) else "" self.suffix = message[-1] if message and isinstance(message[-1], str) else "" - self.specificity = sum(len(p) for p in self._parts if isinstance(p, str)) - self._regex = None + # The text a template spells itself; the hint's separator counted as one + # character, as the order of the catalog's codes has always assumed. + self.specificity = (sum(len(p) for p in message + hint_parts if isinstance(p, str)) + + (1 if hint is not None else 0)) self._kinds = {name: spec.get("kind") for name, spec in entry.get("args", {}).items()} + names = [part[0] for part in message + hint_parts if isinstance(part, tuple)] + self._repeats = len(names) != len(set(names)) - def match(self, subject: str) -> dict | None: - if self._regex is None: - pieces: list[str] = [] - seen: set[str] = set() - for part in self._parts: - if isinstance(part, str): - pieces.append(re.escape(part)) - elif part[0] in seen: - pieces.append(f"(?P={part[0]})") - else: - seen.add(part[0]) - pieces.append(f"(?P<{part[0]}>.*?)") - self._regex = re.compile("".join(pieces), re.DOTALL) - found = self._regex.fullmatch(subject) + def match(self, message: str, hint: str | None) -> dict | None: + segments = [self._message] + ([self._hint] if self._hint is not None else []) + texts = [message] + ([hint] if self._hint is not None else []) + found = _search(segments, texts, self._repeats) if found is None: return None args: dict[str, Any] = {} - for name, value in found.groupdict().items(): + for name, value in found.items(): if self._kinds.get(name) == "number": if _CANONICAL_INT.fullmatch(value): args[name] = int(value) @@ -230,13 +326,12 @@ def classify(severity: str, message: str, hint: str | None = None) -> tuple[str, ``severity`` is ``"error"`` or ``"warning"``. A text no catalog template renders gets ``PF-E0000`` / ``PF-W0000`` with its text as ``args``. """ - subject = message if hint is None else message + _JOIN + hint for matcher in _matchers().get(severity, ()): if matcher.has_hint != (hint is not None): continue if not message.startswith(matcher.prefix) or not message.endswith(matcher.suffix): continue - args = matcher.match(subject) + args = matcher.match(message, hint) if args is not None: return matcher.code, args args = {"message": message} diff --git a/tests/test_diagnostic_codes.py b/tests/test_diagnostic_codes.py index 15b664b..19e0366 100644 --- a/tests/test_diagnostic_codes.py +++ b/tests/test_diagnostic_codes.py @@ -158,6 +158,111 @@ def test_number_arguments_are_numbers(): assert render(entry["message"], diagnostic.args) == diagnostic.message +def _lazy_regex_patterns() -> list: + """Each template as a lazy-group regex over the message, a NUL and the + hint (the matcher before it became linear), most specific first.""" + import re as _re + patterns = [] + for code, entry in sorted(CATALOG.items(), key=lambda kv: (-_specificity(kv[1]), kv[0])): + if code in UNCATALOGUED.values(): + continue + parts = parse_template(entry["message"]) + if entry.get("hint") is not None: + parts = parts + ["\x00"] + parse_template(entry["hint"]) + pieces, seen = [], set() + for part in parts: + if isinstance(part, str): + pieces.append(_re.escape(part)) + elif part[0] in seen: + pieces.append(f"(?P={part[0]})") + else: + seen.add(part[0]) + pieces.append(f"(?P<{part[0]}>.*?)") + patterns.append((code, entry["severity"], entry.get("hint") is not None, + _re.compile("".join(pieces), _re.DOTALL))) + return patterns + + +def _lazy_regex_classify(patterns: list, severity: str, message: str, hint: str | None): + subject = message if hint is None else message + "\x00" + hint + for code, template_severity, has_hint, pattern in patterns: + if template_severity != severity or has_hint != (hint is not None): + continue + found = pattern.fullmatch(subject) + if found is not None: + return code, found.groupdict() + return UNCATALOGUED[severity], None + + +def _specificity(entry: dict) -> int: + parts = parse_template(entry["message"]) + if entry.get("hint") is not None: + parts = parts + ["\x00"] + parse_template(entry["hint"]) + return sum(len(part) for part in parts if isinstance(part, str)) + + +def _raw(args: dict) -> dict: + return {name: str(value) for name, value in args.items()} + + +def test_classification_is_the_lazy_regex_split(): + """Every template rendered with plain arguments and with arguments holding + its own literals reads back as a lazy-regex fullmatch reads it: the same + code and the same arguments, a name used twice included.""" + import random + rng = random.Random(20261004) + patterns = _lazy_regex_patterns() + checked = 0 + for code, entry in sorted(CATALOG.items()): + if code in UNCATALOGUED.values(): + continue + literals = [part for template in (entry["message"], entry.get("hint")) if template + for part in parse_template(template) if isinstance(part, str) and part] or ["x"] + for trial in range(4): + args = {} + for name in entry["args"]: + if trial < 2: + args[name] = rng.choice(["x", "a[1]", "ta.sma(close, 14)", "obj.field[1]", "it's"]) + else: + args[name] = "".join(rng.choice(["q", rng.choice(literals)[:rng.randint(1, 6)], + rng.choice(literals)]) for _ in range(2)) + message = render(entry["message"], args) + hint = render(entry.get("hint"), args) + expected_code, expected_args = _lazy_regex_classify(patterns, entry["severity"], message, hint) + got_code, got_args = classify(entry["severity"], message, hint) + assert got_code == expected_code, (code, message, hint) + if expected_args is not None: + assert _raw(got_args) == expected_args, (code, message, hint) + checked += 1 + assert checked > 1000 + + +def test_classification_is_linear_on_crafted_text(): + """Arguments flooded with the template's own separators and a hint no + template has: a lazy-regex fullmatch backtracked over every split (23 s + for 3.4 KB of one template's text); the split stays linear.""" + import time + worst = 0.0 + for code, entry in CATALOG.items(): + if code in UNCATALOGUED.values(): + continue + parts = parse_template(entry["message"]) + if sum(isinstance(part, tuple) for part in parts) < 3: + continue + flooded = [] + for index, part in enumerate(parts): + if isinstance(part, str): + flooded.append(part) + else: + after = next((p for p in parts[index + 1:] if isinstance(p, str)), "") + flooded.append((after or "z") * 2000) + started = time.monotonic() + classify(entry["severity"], "".join(flooded), + None if entry.get("hint") is None else "a hint no template has") + worst = max(worst, time.monotonic() - started) + assert worst < 2.0 + + # --------------------------------------------------------------------------- # Every emitted diagnostic renders back to its text # ---------------------------------------------------------------------------