Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 3 additions & 1 deletion CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -41,7 +41,9 @@ An additive change to the public API; it leaves the emitted C++ unchanged.
argument kinds and a one-line explanation; the templates render the
`message` and `hint` byte for byte, which keep their text. Codes are never
reused (`tests/fixtures/diagnostic_codes_pin.json`). See
[Diagnostic codes](docs/PUBLIC_CONTRACT.md#diagnostic-codes).
[Diagnostic codes](docs/PUBLIC_CONTRACT.md#diagnostic-codes). A code is
read off its text in time linear in the text: a crafted script's text cannot
stall the classification, which a backtracking regex let it do.
- **First error in source order.** Since 1.1.0 the settings metadata visited
every input's `defval`, `options`, `minval`, `maxval` and `step` ahead of
the script body, so an error there (an unknown name in `minval`) was raised
Expand Down
143 changes: 119 additions & 24 deletions pineforge_codegen/diagnostic_codes.py
Original file line number Diff line number Diff line change
Expand Up @@ -14,6 +14,16 @@
never guessed: a text no template renders gets the uncatalogued code of its
severity (``PF-E0000`` / ``PF-W0000``), which the test suite refuses.

The match never backtracks over the text's characters: each literal segment
of a template goes to its leftmost place after the previous one, the last to
the text's end, and every argument is the text between its literals -- the
split a fullmatch of the template with lazy ``(.*?)`` arguments returns. Where
an argument is named twice (``{receiver}`` in the message and the hint) the
leftmost split can name it two values; then the later places of a literal are
tried in order, as the regex's backtracking tries them, within a fixed budget
of steps. The regex took seconds, then minutes, as a crafted message grew,
and a user's script spells argument text.

Rendering (:func:`render`) is the ICU MessageFormat subset the catalog uses:
literal text with ICU apostrophe quoting (``''`` is one apostrophe, ``'{'``
a literal brace) and simple ``{name}`` arguments, a string argument
Expand Down Expand Up @@ -158,47 +168,133 @@ def render_diagnostic(code: str, args: dict) -> tuple[str, str | None]:
# Classification
# ---------------------------------------------------------------------------

# Joins a message and its hint into one subject, so an argument both name is
# one value (a backreference). No template spells it.
_JOIN = "\x00"
_CANONICAL_INT = re.compile(r"-?(?:0|[1-9][0-9]*)")
_CANONICAL_FLOAT = re.compile(r"-?(?:0|[1-9][0-9]*)\.[0-9]+")


def _segments(parts: list) -> tuple[str, tuple[tuple[str, str], ...]]:
"""A parsed template as its leading literal and ``(argument, literal after
it)`` pairs; the literal after an argument may be empty."""
head = ""
pairs: list[list[str]] = []
for part in parts:
if isinstance(part, str):
if pairs:
pairs[-1][1] += part
else:
head += part
else:
pairs.append([part[0], ""])
return head, tuple((name, literal) for name, literal in pairs)


# Literal places a template whose argument is named twice may try, per
# classification: a script's text never needs more than a few, and a crafted
# one gets the uncatalogued code instead of a long search.
_SEARCH_BUDGET = 4096


def _search(segments: list, texts: list[str], repeats: bool) -> dict[str, str] | None:
"""Split ``texts`` (the message, then the hint) by their templates'
``segments``: each argument ends at the leftmost place of the literal after
it, the last argument of a text at its end -- the split a lazy-regex
fullmatch returns. Without an argument named twice (``repeats``) that split
succeeds whenever any does, since a leftmost place leaves the rest the most
room: it is the only one tried, so the work is linear. With one, a split can
give the name two values; then the later places of a literal are tried in
order, as the regex backtracks, within ``_SEARCH_BUDGET`` places."""
starts: list[int] = []
ends: list[int] = []
slots: list[tuple[int, str, str, bool]] = [] # text, argument, literal after it, last
for index, ((head, pairs), text) in enumerate(zip(segments, texts)):
if not text.startswith(head):
return None
if not pairs:
if len(text) != len(head):
return None
starts.append(len(head))
ends.append(len(head))
continue
tail = pairs[-1][1]
end = len(text) - len(tail)
if end < len(head) or not text.endswith(tail):
return None
starts.append(len(head))
ends.append(end)
for at, (name, literal) in enumerate(pairs):
slots.append((index, name, literal, at == len(pairs) - 1))
found: dict[str, str] = {}
budget = [_SEARCH_BUDGET]

def step(slot: int, pos: int) -> bool:
if slot == len(slots):
return True
text_index, name, literal, last = slots[slot]
text, end = texts[text_index], ends[text_index]
following = slot + 1
known = found.get(name)
if last:
value = text[pos:end]
resume = (starts[slots[following][0]] if following < len(slots) else end)
if known is not None:
return known == value and step(following, resume)
found[name] = value
if step(following, resume):
return True
del found[name]
return False
if known is not None:
stop = pos + len(known)
return (text.startswith(known, pos) and stop + len(literal) <= end
and text.startswith(literal, stop) and step(following, stop + len(literal)))
at = text.find(literal, pos, end)
while at >= 0:
budget[0] -= 1
if budget[0] < 0:
return False
found[name] = text[pos:at]
if step(following, at + len(literal)):
return True
del found[name]
if not repeats:
return False
at = text.find(literal, at + 1, end) if at < end else -1
return False

first = starts[slots[0][0]] if slots else 0
return found if step(0, first) else None


class _Matcher:
__slots__ = ("code", "has_hint", "prefix", "suffix", "specificity",
"_parts", "_regex", "_kinds")
"_message", "_hint", "_kinds", "_repeats")

def __init__(self, code: str, entry: dict):
self.code = code
message = parse_template(entry["message"])
hint = entry.get("hint")
hint_parts = parse_template(hint) if hint is not None else []
self.has_hint = hint is not None
self._parts = message + ([_JOIN] + parse_template(hint) if hint is not None else [])
self._message = _segments(message)
self._hint = _segments(hint_parts) if hint is not None else None
self.prefix = message[0] if message and isinstance(message[0], str) else ""
self.suffix = message[-1] if message and isinstance(message[-1], str) else ""
self.specificity = sum(len(p) for p in self._parts if isinstance(p, str))
self._regex = None
# The text a template spells itself; the hint's separator counted as one
# character, as the order of the catalog's codes has always assumed.
self.specificity = (sum(len(p) for p in message + hint_parts if isinstance(p, str))
+ (1 if hint is not None else 0))
self._kinds = {name: spec.get("kind") for name, spec in entry.get("args", {}).items()}
names = [part[0] for part in message + hint_parts if isinstance(part, tuple)]
self._repeats = len(names) != len(set(names))

def match(self, subject: str) -> dict | None:
if self._regex is None:
pieces: list[str] = []
seen: set[str] = set()
for part in self._parts:
if isinstance(part, str):
pieces.append(re.escape(part))
elif part[0] in seen:
pieces.append(f"(?P={part[0]})")
else:
seen.add(part[0])
pieces.append(f"(?P<{part[0]}>.*?)")
self._regex = re.compile("".join(pieces), re.DOTALL)
found = self._regex.fullmatch(subject)
def match(self, message: str, hint: str | None) -> dict | None:
segments = [self._message] + ([self._hint] if self._hint is not None else [])
texts = [message] + ([hint] if self._hint is not None else [])
found = _search(segments, texts, self._repeats)
if found is None:
return None
args: dict[str, Any] = {}
for name, value in found.groupdict().items():
for name, value in found.items():
if self._kinds.get(name) == "number":
if _CANONICAL_INT.fullmatch(value):
args[name] = int(value)
Expand Down Expand Up @@ -230,13 +326,12 @@ def classify(severity: str, message: str, hint: str | None = None) -> tuple[str,
``severity`` is ``"error"`` or ``"warning"``. A text no catalog template
renders gets ``PF-E0000`` / ``PF-W0000`` with its text as ``args``.
"""
subject = message if hint is None else message + _JOIN + hint
for matcher in _matchers().get(severity, ()):
if matcher.has_hint != (hint is not None):
continue
if not message.startswith(matcher.prefix) or not message.endswith(matcher.suffix):
continue
args = matcher.match(subject)
args = matcher.match(message, hint)
if args is not None:
return matcher.code, args
args = {"message": message}
Expand Down
105 changes: 105 additions & 0 deletions tests/test_diagnostic_codes.py
Original file line number Diff line number Diff line change
Expand Up @@ -158,6 +158,111 @@ def test_number_arguments_are_numbers():
assert render(entry["message"], diagnostic.args) == diagnostic.message


def _lazy_regex_patterns() -> list:
"""Each template as a lazy-group regex over the message, a NUL and the
hint (the matcher before it became linear), most specific first."""
import re as _re
patterns = []
for code, entry in sorted(CATALOG.items(), key=lambda kv: (-_specificity(kv[1]), kv[0])):
if code in UNCATALOGUED.values():
continue
parts = parse_template(entry["message"])
if entry.get("hint") is not None:
parts = parts + ["\x00"] + parse_template(entry["hint"])
pieces, seen = [], set()
for part in parts:
if isinstance(part, str):
pieces.append(_re.escape(part))
elif part[0] in seen:
pieces.append(f"(?P={part[0]})")
else:
seen.add(part[0])
pieces.append(f"(?P<{part[0]}>.*?)")
patterns.append((code, entry["severity"], entry.get("hint") is not None,
_re.compile("".join(pieces), _re.DOTALL)))
return patterns


def _lazy_regex_classify(patterns: list, severity: str, message: str, hint: str | None):
subject = message if hint is None else message + "\x00" + hint
for code, template_severity, has_hint, pattern in patterns:
if template_severity != severity or has_hint != (hint is not None):
continue
found = pattern.fullmatch(subject)
if found is not None:
return code, found.groupdict()
return UNCATALOGUED[severity], None


def _specificity(entry: dict) -> int:
parts = parse_template(entry["message"])
if entry.get("hint") is not None:
parts = parts + ["\x00"] + parse_template(entry["hint"])
return sum(len(part) for part in parts if isinstance(part, str))


def _raw(args: dict) -> dict:
return {name: str(value) for name, value in args.items()}


def test_classification_is_the_lazy_regex_split():
"""Every template rendered with plain arguments and with arguments holding
its own literals reads back as a lazy-regex fullmatch reads it: the same
code and the same arguments, a name used twice included."""
import random
rng = random.Random(20261004)
patterns = _lazy_regex_patterns()
checked = 0
for code, entry in sorted(CATALOG.items()):
if code in UNCATALOGUED.values():
continue
literals = [part for template in (entry["message"], entry.get("hint")) if template
for part in parse_template(template) if isinstance(part, str) and part] or ["x"]
for trial in range(4):
args = {}
for name in entry["args"]:
if trial < 2:
args[name] = rng.choice(["x", "a[1]", "ta.sma(close, 14)", "obj.field[1]", "it's"])
else:
args[name] = "".join(rng.choice(["q", rng.choice(literals)[:rng.randint(1, 6)],
rng.choice(literals)]) for _ in range(2))
message = render(entry["message"], args)
hint = render(entry.get("hint"), args)
expected_code, expected_args = _lazy_regex_classify(patterns, entry["severity"], message, hint)
got_code, got_args = classify(entry["severity"], message, hint)
assert got_code == expected_code, (code, message, hint)
if expected_args is not None:
assert _raw(got_args) == expected_args, (code, message, hint)
checked += 1
assert checked > 1000


def test_classification_is_linear_on_crafted_text():
"""Arguments flooded with the template's own separators and a hint no
template has: a lazy-regex fullmatch backtracked over every split (23 s
for 3.4 KB of one template's text); the split stays linear."""
import time
worst = 0.0
for code, entry in CATALOG.items():
if code in UNCATALOGUED.values():
continue
parts = parse_template(entry["message"])
if sum(isinstance(part, tuple) for part in parts) < 3:
continue
flooded = []
for index, part in enumerate(parts):
if isinstance(part, str):
flooded.append(part)
else:
after = next((p for p in parts[index + 1:] if isinstance(p, str)), "")
flooded.append((after or "z") * 2000)
started = time.monotonic()
classify(entry["severity"], "".join(flooded),
None if entry.get("hint") is None else "a hint no template has")
worst = max(worst, time.monotonic() - started)
assert worst < 2.0


# ---------------------------------------------------------------------------
# Every emitted diagnostic renders back to its text
# ---------------------------------------------------------------------------
Expand Down
Loading