From b03b4b5341088af78af3df1c1654cd444ea2a91b Mon Sep 17 00:00:00 2001 From: Jothsnapraveena Date: Sat, 3 Oct 2026 23:57:56 -0500 Subject: [PATCH] bumped v0.70 --- CHANGELOG.md | 43 +++ README.md | 139 +++++++++- docs/docs/index.html | 165 +++++++++-- docs/index.html | 17 +- pyproject.toml | 2 +- src/agentic_evals/__init__.py | 8 +- src/agentic_evals/evaluators.py | 4 +- src/agentic_evals/gate.py | 46 ++- src/agentic_evals/models.py | 69 +++++ src/agentic_evals/runner.py | 125 ++++++++- src/agentic_evals/scorers/__init__.py | 2 + src/agentic_evals/scorers/rubric.py | 67 ++++- src/agentic_evals/simple.py | 2 +- .../analyze-an-eval-experiment/SKILL.md | 3 +- .../skills/choose-a-rubric-template/SKILL.md | 11 +- .../skills/define-a-release-gate/SKILL.md | 4 + .../skills/discover-failure-modes/SKILL.md | 8 +- .../skills/write-a-scorer/SKILL.md | 12 +- tests/test_gate.py | 65 +++++ tests/test_report_breakdowns.py | 262 ++++++++++++++++++ tests/test_scorers_rubric.py | 78 ++++++ tests/test_tool_expectations.py | 125 +++++++++ uv.lock | 2 +- voice-agent-evals/run_evals.py | 41 ++- voice-agent-evals/scenario_runner.py | 25 +- 25 files changed, 1228 insertions(+), 97 deletions(-) create mode 100644 tests/test_report_breakdowns.py create mode 100644 tests/test_tool_expectations.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 6ee961d..d762773 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,49 @@ All notable changes to this project will be documented here. This project follows [Semantic Versioning](https://semver.org/). +## 0.7.0 - 2026-10-03 + +### Added + +- `EvaluationSummary.metrics` and `EvaluationSummary.tags`: every report + now breaks its results down by metric (`MetricSummary`: checks passed, + failed and skipped, pass rate, average score) and by case tag + (`TagSummary`: case counts, pass rate, average score). +- `Score.metric`: a stable key that reports group by. It defaults to + `Score.name`; the implicit checks whose names carry a per-case detail + (`contains:`, `required_field:`, `required_tool:`, + `forbidden_tool:`, `tool_args:`) share the metric named by + their prefix. `Score.name` is unchanged. +- `Score.skip(name, explanation)` and `Score.skipped`: an evaluator can + record that a check did not apply to a case. A skipped score never + fails the case, is left out of `average_score` and pass rates, and is + counted in `MetricSummary.skipped`. `Eval()` ignores a skipped `Score` + returned by a scorer. +- `CaseEvaluation.tags` and `CaseEvaluation.metadata`, copied from the + `TestCase`, so a report can be filtered or regrouped on its own. +- `GateConfig.min_tag_pass_rate` and `GateConfig.min_metric_pass_rate`: + pass-rate floors for one tag or one metric, applied on top of + `min_pass_rate`. A tag or metric named in the config but absent from + the report fails the gate. Observed values are reported under + `tag_pass_rate:` and `metric_pass_rate:`. +- `TestCase.expected_tool_arguments`: expected argument values per tool. + Passes when one call to the tool carries every listed argument with an + equal value (no type coercion). Scored as `tool_arg_values:`. +- `TestCase.required_tool_order`: tools whose first calls must come in + the listed order. Scored as `tool_order`. +- `make_rubric(name, criteria, levels=..., with_reference=...)`: builds a + `RubricTemplate` from criteria written in plain language, with a + pass/fail scale by default or a caller-supplied graded scale. + +All of the above are additive: reports saved by earlier versions still +load, and `schema_version` stays `1.0`. + +### Fixed + +- `evaluate_suite` no longer raises `ZeroDivisionError` when no case + produces a score (for example, every evaluator returned an empty list); + `average_score` is reported as `0.0`. + ## 0.6.0 - 2026-10-03 ### Changed diff --git a/README.md b/README.md index 7c9f00d..1a87171 100644 --- a/README.md +++ b/README.md @@ -119,20 +119,24 @@ directly -- the engine `Eval()` above is a thin, opinionated front end for. arbitrary `attributes`), total latency, estimated cost, and metadata. Deliberately not tied to any specific instrumentation format — build one from whatever you already have. -- **`Score`** — a single named judgment (0-1 value, pass/fail, explanation). +- **`Score`** — a single named judgment (0-1 value, pass/fail, explanation), + with a stable `metric` key that reports group by. `Score.skip(...)` records + a check that did not apply to a case. - **`Evaluator`** — anything with a `.name` and an `.evaluate(context) -> list[Score]`. `CallableEvaluator` adapts a plain Python function; `LLMJudgeEvaluator` and `BusinessRuleEvaluator` are named convenience subclasses for readability/reporting. - **`TestCase`/`TestSuite`** — declarative expectations (exact match, substring, JSON Schema, required fields, required/forbidden tool calls, - required tool arguments, latency/cost/turn-count thresholds, or a named - custom evaluator) plus the cases that make up a suite. + tool arguments and their values, tool call order, latency/cost/turn-count + thresholds, or a named custom evaluator) plus the cases that make up a + suite. - **`evaluate_suite`** — runs a suite against supplied `EvaluationSample`s and returns an `EvaluationReport` (per-case scores plus a pass-rate/cost/ - latency summary). + latency summary, broken down by metric and by tag). - **`GateConfig`/`evaluate_gate`** — turn an `EvaluationReport` into a - pass/fail release decision on configurable thresholds. + pass/fail release decision on configurable thresholds, for the whole + suite or for a single tag or metric. ### TestSuite quickstart @@ -174,6 +178,87 @@ report = evaluate_suite(suite, [sample]) print(report.summary.pass_rate) # 1.0 ``` +### Breakdowns by metric and tag + +An overall pass rate hides where the failures are. Tag your cases, and the +report's summary breaks results down two ways: + +```python +suite = TestSuite( + name="support-answers", + version="1", + cases=[ + TestCase( + id="r1", + name="Refund status", + tags=["refunds"], + expected_contains=["refund"], + required_tools=["lookup_order"], + ), + TestCase(id="t1", name="Order tracking", tags=["tracking"], expected_contains=["shipped"]), + ], +) +report = evaluate_suite(suite, samples) + +for tag, stats in report.summary.tags.items(): + print(f"{tag}: {stats.passed_cases}/{stats.total_cases} cases passed") + +for metric, stats in report.summary.metrics.items(): + print(f"{metric}: {stats.passed}/{stats.total} checks passed") +``` + +- **`summary.tags`** — one `TagSummary` per tag: case counts, pass rate and + average score for the cases carrying that tag. +- **`summary.metrics`** — one `MetricSummary` per metric: how many checks + passed, failed or were skipped, plus pass rate and average score. Checks + such as `contains:refund` and `contains:shipped` roll up under the single + metric `contains`; the detailed name stays on `Score.name`. +- Each `CaseEvaluation` carries its case's `tags` and `metadata`, so a + report can be filtered or regrouped without the original suite. + +A custom evaluator can mark a check as not applicable instead of passing or +failing it. A skipped score never fails the case and stays out of averages +and pass rates; it is counted separately in `MetricSummary.skipped`: + +```python +def cites_order_id(context: EvaluationContext) -> Score: + order_id = context.case.metadata.get("order_id") + if order_id is None: + return Score.skip("cites_order_id", "Case has no order id to cite.") + cited = order_id in context.sample.output + return Score( + name="cites_order_id", + value=float(cited), + passed=cited, + explanation=f"Order id {order_id} cited: {cited}.", + ) +``` + +### Tool call expectations + +Beyond which tools were called, a case can state what they were called +with and in what order. Tool arguments are read from each span's +`attributes["tool_args"]`: + +```python +TestCase( + id="publish-draft", + name="Saves the draft before publishing it", + required_tools=["save_draft", "publish"], + forbidden_tools=["delete_document"], + required_tool_arguments={"publish": ["document_id"]}, # keys present + expected_tool_arguments={"publish": {"visibility": "internal"}}, # exact values + required_tool_order=["save_draft", "publish"], # first calls in order +) +``` + +- **`expected_tool_arguments`** passes when at least one call to the tool + carries every listed argument with an equal value. Values are compared + as given, without type coercion (`5` is not `"5"`). +- **`required_tool_order`** passes when each listed tool is first called + before the next one in the list. Calls to other tools in between are + fine; a listed tool that is never called fails the check. + ## LLM-as-judge ```python @@ -221,6 +306,20 @@ if not decision.passed: raise SystemExit(f"Release gate failed: {decision.reasons}") ``` +A gate can hold one slice of the report to a stricter bar than the suite +as a whole, keyed by case tag or by `Score.metric`: + +```python +GateConfig( + min_pass_rate=0.95, + min_tag_pass_rate={"safety": 1.0}, # every safety-tagged case must pass + min_metric_pass_rate={"forbidden_tool": 1.0}, # no forbidden tool call, anywhere +) +``` + +A tag or metric named in the config but missing from the report fails the +gate, so a renamed tag cannot quietly switch a check off. + Never fabricates a value it can't back up: `total_cost_usd` on a summary or gate decision stays `None` unless every case in scope has a known cost — an incomplete cost picture is reported as unavailable, not `$0.00`. @@ -237,7 +336,8 @@ hand-write a `CallableEvaluator` for common checks: - **`scorers.rubric`** (LLM-graded, provider-neutral): `RubricTemplate` + `LLMRubricEvaluator`, with built-in templates `FACTUALITY`, `CLOSED_QA`, `SUMMARY_QUALITY`, `BATTLE` (pairwise A/B), `MODERATION`, `TRANSLATION`, - `SECURITY`, `SQL_CORRECTNESS`, `POSSIBLE`, `PII_LEAKAGE`. Like + `SECURITY`, `SQL_CORRECTNESS`, `POSSIBLE`, `PII_LEAKAGE`, plus + `make_rubric()` to build one from your own criteria. Like `LLMJudgeEvaluator`, this package never calls a model itself -- you pass a `complete_fn: Callable[[str], str]`. - **`scorers.trajectory`** (reads the trace, not just the output text -- @@ -284,6 +384,33 @@ registry = default_registry() registry.register(LLMRubricEvaluator("factuality", FACTUALITY, complete_fn=call_your_model)) ``` +When no built-in template fits, describe the criterion in plain language +and `make_rubric()` builds the template — the prompt, the verdict letters +and their scores: + +```python +from agentic_evals import LLMRubricEvaluator, make_rubric + +concise = make_rubric("concise", "The answer is at most two sentences and has no preamble.") + +tone = make_rubric( + "tone", + "The reply is courteous and does not blame the reader.", + levels=[ # best first; scores in [0, 1] + ("Courteous throughout.", 1.0), + ("Neutral: neither courteous nor rude.", 0.5), + ("Rude, dismissive, or blames the reader.", 0.0), + ], +) + +registry.register(LLMRubricEvaluator("concise", concise, complete_fn=call_your_model)) +registry.register(LLMRubricEvaluator("tone", tone, complete_fn=call_your_model)) +``` + +The default scale is pass/fail. Pass `with_reference=True` to show the +judge a reference next to the output; it is read from +`EvaluatorConfig.config["reference"]`, as with the built-in templates. + ## Live targets Point a suite at a real running system (a trusted Python callable, or an diff --git a/docs/docs/index.html b/docs/docs/index.html index 96ee1bc..e83a9ab 100644 --- a/docs/docs/index.html +++ b/docs/docs/index.html @@ -294,7 +294,7 @@

What is agentic-evals?

agentic-evals returns:

  • Per-case scores — individual judgments with explanations
  • -
  • Summary metrics — pass rate, average score, cost, latency
  • +
  • Summary metrics — pass rate, average score, cost, latency, broken down by metric and by tag
  • Pass/fail gates — CI-ready exit codes for release decisions
@@ -314,7 +314,7 @@

Your first eval in 60 seconds

{"input": "France", "expected": "Paris"}, {"input": "Japan", "expected": "Tokyo"}, ], - task=lambda x: my_agent(x["input"]), + task=my_agent, scores=[equals], ) @@ -403,7 +403,17 @@

TestSuite and TestCase

required_tool_arguments dict[str, list[str]] - Specific arguments a tool must receive + Argument names a tool call must include + + + expected_tool_arguments + dict[str, dict[str, Any]] + Argument values a tool call must carry (compared without type coercion) + + + required_tool_order + list[str] + Tools whose first calls must come in this order max_latency_ms @@ -425,6 +435,16 @@

TestSuite and TestCase

list[EvaluatorConfig] Custom evaluators to run + + tags + list[str] + Labels for grouping results in the report (e.g., "refunds") + + + metadata + dict[str, Any] + Free-form data for custom evaluators; copied onto the case result + @@ -440,10 +460,23 @@

TestSuite and TestCase

expected_contains=["Your refund", "processing"], required_tools=["lookup_order"], max_latency_ms=1000, + tags=["refunds"], ), ], ) +

Tool expectations can also state what a tool was called with, and in what order. Arguments are read from each span's attributes["tool_args"]:

+ +
TestCase(
+    id="publish-draft",
+    name="Saves the draft before publishing it",
+    required_tools=["save_draft", "publish"],
+    expected_tool_arguments={"publish": {"visibility": "internal"}},
+    required_tool_order=["save_draft", "publish"],
+)
+ +

expected_tool_arguments passes when one call to the tool carries every listed value. required_tool_order passes when each listed tool is first called before the next; calls to other tools in between are fine.

+

Evaluators

An evaluator is anything with a .name property and an .evaluate(context) -> list[Score] method.

@@ -458,10 +491,12 @@

Scores and Reports

Each evaluation produces a Score:

class Score(BaseModel):
-    name: str                    # e.g., "factuality", "exact_match"
+    name: str                    # e.g., "factuality", "contains:refund"
+    metric: str                  # Grouping key; defaults to name ("contains")
     value: float                 # 0.0 to 1.0
     passed: bool                 # True if value >= threshold
     required: bool = True
+    skipped: bool = False        # True when the check did not apply
     explanation: str             # Why it passed/failed
     evaluator_type: str = "deterministic"  # or "llm_judge"
     metadata: dict[str, Any] = {}
@@ -474,6 +509,56 @@

Scores and Reports

created_at: datetime summary: EvaluationSummary # Pass rate, avg score, cost, latency cases: list[CaseEvaluation] # Per-case details + +

Breakdowns by metric and tag

+

An overall pass rate does not show where the failures are. The summary breaks results down two ways:

+ +
report = evaluate_suite(suite, samples)
+
+for tag, stats in report.summary.tags.items():
+    print(f"{tag}: {stats.passed_cases}/{stats.total_cases} cases passed")
+
+for metric, stats in report.summary.metrics.items():
+    print(f"{metric}: {stats.passed}/{stats.total} checks passed")
+ + + + + + + + + + + + + + + + + + + + + +
FieldTypeContents
summary.tagsdict[str, TagSummary]Per tag: total_cases, passed_cases, failed_cases, pass_rate, average_score
summary.metricsdict[str, MetricSummary]Per metric: total, passed, failed, skipped, pass_rate, average_score
+ +

Checks whose names carry a per-case detail share one metric: contains:refund and contains:shipped both roll up under contains. Each CaseEvaluation also carries its case's tags and metadata, so a saved report can be filtered or regrouped on its own.

+ +

Skipped checks

+

When a check does not apply to a case, return Score.skip(...) instead of a pass. A skipped score never fails the case and stays out of averages and pass rates; it is counted in MetricSummary.skipped.

+ +
def cites_order_id(context: EvaluationContext) -> Score:
+    order_id = context.case.metadata.get("order_id")
+    if order_id is None:
+        return Score.skip("cites_order_id", "Case has no order id to cite.")
+    cited = order_id in context.sample.output
+    return Score(
+        name="cites_order_id",
+        value=float(cited),
+        passed=cited,
+        explanation=f"Order id {order_id} cited: {cited}.",
+    )
@@ -515,7 +600,7 @@

Eval() Quickstart

task callable - Function to evaluate, takes input dict, returns string + Function under test; called with each row's input value, returns the output scores @@ -551,7 +636,7 @@

TestSuite API

) -> EvaluationReport -

Returns a full EvaluationReport with per-case scores, summary metrics, pass rates, cost/latency.

+

Returns a full EvaluationReport with per-case scores, summary metrics, pass rates, cost/latency, and breakdowns by metric and tag.

Scorers Library

Pre-built evaluators organized by type:

@@ -579,6 +664,27 @@

Rubric Scorers (LLM-Graded)

) ) +

When no built-in template fits, describe the criterion in plain language. make_rubric() builds the prompt, the verdict letters and their scores:

+ +
from agentic_evals import LLMRubricEvaluator, make_rubric
+
+concise = make_rubric("concise", "The answer is at most two sentences and has no preamble.")
+
+tone = make_rubric(
+    "tone",
+    "The reply is courteous and does not blame the reader.",
+    levels=[                       # best first; scores from 0 to 1
+        ("Courteous throughout.", 1.0),
+        ("Neutral: neither courteous nor rude.", 0.5),
+        ("Rude, dismissive, or blames the reader.", 0.0),
+    ],
+)
+
+registry.register(LLMRubricEvaluator("concise", concise, complete_fn=your_model_call))
+registry.register(LLMRubricEvaluator("tone", tone, complete_fn=your_model_call))
+ +

The default scale is pass/fail. Pass with_reference=True to show the judge a reference next to the output.

+

Trajectory Scorers (Trace-Aware)

from agentic_evals import scorers
 
@@ -612,6 +718,16 @@ 

Release Gates

print(f"Release blocked: {decision.reasons}") exit(1)
+

A gate can hold one slice of the report to a stricter bar than the suite as a whole, keyed by case tag or by Score.metric:

+ +
GateConfig(
+    min_pass_rate=0.95,
+    min_tag_pass_rate={"safety": 1.0},             # Every safety-tagged case passes
+    min_metric_pass_rate={"forbidden_tool": 1.0},  # No forbidden tool call, anywhere
+)
+ +

A tag or metric named in the config but missing from the report fails the gate, so a renamed tag cannot quietly switch a check off.

+

Live Targets

Run a suite against a real system (trusted code or HTTP endpoint) instead of pre-recorded samples:

@@ -687,24 +803,35 @@

Eval Packs

Bundle a TestSuite with the scorers it needs into one shareable YAML/JSON file:

name: my-eval-pack
-version: 1.0
+description: Checks answers for factual consistency.
 required_scorers:
-  - levenshtein_similarity
   - factuality
 
-cases:
-  - id: case-1
-    name: Check if output is factual
-    expected_output: Paris
-    evaluators:
-      - name: factuality
-        threshold: 0.9
+suite: + name: my-eval-pack + version: "1" + cases: + - id: case-1 + name: Check if output is factual + input: What is the capital of France? + evaluators: + - name: factuality + threshold: 0.8 + config: + reference: Paris is the capital of France.

Load and use it:

-
from agentic_evals import load_pack, evaluate_suite, default_registry
+          
from pathlib import Path
+
+from agentic_evals import (
+    FACTUALITY, LLMRubricEvaluator, default_registry, evaluate_suite, load_pack,
+)
+
+registry = default_registry()
+registry.register(LLMRubricEvaluator("factuality", FACTUALITY, complete_fn=your_model_call))
 
-pack = load_pack("path/to/pack.yaml")
-report = evaluate_suite(pack.to_suite(), samples, registry=default_registry())
+pack = load_pack(Path("path/to/pack.yaml"), registry=registry) +report = evaluate_suite(pack.to_suite(), samples, registry=registry)

Built-in packs:

    @@ -745,7 +872,7 @@

    Next Steps

    diff --git a/docs/index.html b/docs/index.html index fa2a168..8aaf652 100644 --- a/docs/index.html +++ b/docs/index.html @@ -407,7 +407,7 @@
    -

    Agentic Evals · v0.6.0

    +

    Agentic Evals · v0.7.0

    Know your agent is ready before it ships.

    Score LLM and agent outputs with deterministic checks, LLM-as-judge rubrics and @@ -467,17 +467,17 @@

    Deterministic scoring

    02

    LLM-as-judge

    -

    Rubric scoring with ten built-in templates, from factuality to SQL correctness. You supply complete_fn — any provider works.

    +

    Rubric scoring with ten built-in templates, from factuality to SQL correctness, or your own criteria in plain language. You supply complete_fn — any provider works.

    03

    Trace-aware

    -

    Score the path, not just the answer: tool-call precision and recall, redundant calls, trajectory efficiency, latency, cost and turns.

    +

    Score the path, not just the answer: which tools were called, with what arguments and in what order, plus redundant calls, latency, cost and turns.

    04

    Release gates

    -

    Set thresholds on pass rate, average score, latency and total cost. The CLI exits 0 or 1, so CI can block the merge.

    +

    Set thresholds on pass rate, average score, latency and total cost, for the whole suite or for a single tag or metric. The CLI exits 0 or 1, so CI can block the merge.

    05 @@ -552,13 +552,18 @@

    From first eval to CI gate.

    required_output_fields=["status"], max_latency_ms=2000, max_cost_usd=0.02, + tags=["refunds"], ), ], ) report = run_live_suite(suite, PythonTarget(callable_path="my_app:run_case")) +for tag, stats in report.summary.tags.items(): + print(f"{tag}: {stats.passed_cases}/{stats.total_cases} cases passed") + decision = evaluate_gate(report, GateConfig( min_pass_rate=0.95, + min_tag_pass_rate={"refunds": 1.0}, max_average_latency_ms=1500, max_total_cost_usd=0.25, )) @@ -566,7 +571,7 @@

    From first eval to CI gate.

    raise SystemExit(f"Release gate failed: {decision.reasons}")
    Declarative suites - Expectations cover tool calls, output fields, JSON Schema, latency, cost and turn count. Run against a live Python or HTTP target, then gate the release on the report. + Expectations cover tool calls, output fields, JSON Schema, latency, cost and turn count. Run against a live Python or HTTP target, read the results by tag and by metric, then gate the release on the report.
    @@ -628,7 +633,7 @@

    Where teams put it to work.

    QA

    Answer quality

    -

    Grade answers against references with closed-QA and factuality rubrics, backed by deterministic checks.

    +

    Grade answers against references with closed-QA and factuality rubrics, backed by deterministic checks. Tag cases to see pass rates per scenario.

    Tools diff --git a/pyproject.toml b/pyproject.toml index b1051bc..95db401 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "hatchling.build" [project] name = "agentic-evals" -version = "0.6.0" +version = "0.7.0" description = "A standalone, framework-agnostic evaluation and scoring engine for LLM/agent outputs — extracted from AgenticLens's evaluation module." readme = "README.md" license = { file = "LICENSE" } diff --git a/src/agentic_evals/__init__.py b/src/agentic_evals/__init__.py index ae865c5..0ac0e7a 100644 --- a/src/agentic_evals/__init__.py +++ b/src/agentic_evals/__init__.py @@ -24,8 +24,10 @@ EvaluatorConfig, HTTPTarget, LiveTarget, + MetricSummary, PythonTarget, Score, + TagSummary, TestCase, TestSuite, ) @@ -58,6 +60,7 @@ exact_match, json_diff, levenshtein_similarity, + make_rubric, no_redundant_tool_calls, numeric_diff, numeric_range, @@ -81,7 +84,7 @@ numeric_close, ) -__version__ = "0.6.0" +__version__ = "0.7.0" __all__ = [ "BATTLE", @@ -117,9 +120,11 @@ "LLMJudgeEvaluator", "LLMRubricEvaluator", "LiveTarget", + "MetricSummary", "PythonTarget", "RubricTemplate", "Score", + "TagSummary", "TestCase", "TestSuite", "__version__", @@ -143,6 +148,7 @@ "load_pack", "load_samples", "load_suite", + "make_rubric", "matches", "no_redundant_tool_calls", "numeric_close", diff --git a/src/agentic_evals/evaluators.py b/src/agentic_evals/evaluators.py index 6d77f49..a94c401 100644 --- a/src/agentic_evals/evaluators.py +++ b/src/agentic_evals/evaluators.py @@ -52,7 +52,9 @@ def evaluate(self, context: EvaluationContext) -> list[Score]: scores = result if isinstance(result, list) else [result] return [ score.model_copy( - update={ + update={"evaluator_type": self._evaluator_type} + if score.skipped + else { "evaluator_type": self._evaluator_type, "passed": score.value >= context.config.threshold, "required": context.config.required, diff --git a/src/agentic_evals/gate.py b/src/agentic_evals/gate.py index 82e0180..b34bef9 100644 --- a/src/agentic_evals/gate.py +++ b/src/agentic_evals/gate.py @@ -1,7 +1,11 @@ +from typing import Annotated + from pydantic import BaseModel, Field from agentic_evals.models import EvaluationReport +PassRate = Annotated[float, Field(ge=0, le=1)] + class GateConfig(BaseModel): """Release thresholds. The default is strict: every case must pass. @@ -10,11 +14,19 @@ class GateConfig(BaseModel): when left `None` -- so `min_pass_rate` alone decides how many failures are tolerated, and a graded score below 1.0 on a passing case does not fail the gate unless an average-score floor is set explicitly. + + `min_tag_pass_rate` and `min_metric_pass_rate` set a floor for one + slice of the report, keyed by tag or by `Score.metric`. They apply on + top of `min_pass_rate`, so a single tag or metric can be held to a + stricter bar than the suite as a whole. A tag or metric named here but + absent from the report fails the gate rather than passing unchecked. """ min_pass_rate: float = Field(default=1.0, ge=0, le=1) min_average_score: float | None = Field(default=None, ge=0, le=1) max_failed_cases: int | None = Field(default=None, ge=0) + min_tag_pass_rate: dict[str, PassRate] = Field(default_factory=dict) + min_metric_pass_rate: dict[str, PassRate] = Field(default_factory=dict) max_average_latency_ms: float | None = Field(default=None, gt=0) max_total_cost_usd: float | None = Field(default=None, ge=0) @@ -56,14 +68,26 @@ def evaluate_gate(report: EvaluationReport, config: GateConfig) -> GateDecision: f"Total cost ${summary.total_cost_usd:.6f} exceeds " f"${config.max_total_cost_usd:.6f}." ) - return GateDecision( - passed=not reasons, - reasons=reasons, - observed={ - "pass_rate": summary.pass_rate, - "average_score": summary.average_score, - "failed_cases": summary.failed_cases, - "average_latency_ms": summary.average_latency_ms, - "total_cost_usd": summary.total_cost_usd, - }, - ) + observed: dict[str, float | int | None] = { + "pass_rate": summary.pass_rate, + "average_score": summary.average_score, + "failed_cases": summary.failed_cases, + "average_latency_ms": summary.average_latency_ms, + "total_cost_usd": summary.total_cost_usd, + } + for tag, minimum in config.min_tag_pass_rate.items(): + tagged = summary.tags.get(tag) + observed[f"tag_pass_rate:{tag}"] = tagged.pass_rate if tagged else None + if tagged is None: + reasons.append(f"Tag {tag!r} has no cases in the report.") + elif tagged.pass_rate < minimum: + reasons.append(f"Tag {tag!r} pass rate {tagged.pass_rate:.1%} is below {minimum:.1%}.") + for metric, minimum in config.min_metric_pass_rate.items(): + measured = summary.metrics.get(metric) + rate = measured.pass_rate if measured else None + observed[f"metric_pass_rate:{metric}"] = rate + if rate is None: + reasons.append(f"Metric {metric!r} has no evaluated scores in the report.") + elif rate < minimum: + reasons.append(f"Metric {metric!r} pass rate {rate:.1%} is below {minimum:.1%}.") + return GateDecision(passed=not reasons, reasons=reasons, observed=observed) diff --git a/src/agentic_evals/models.py b/src/agentic_evals/models.py index 5e4b427..11f6c06 100644 --- a/src/agentic_evals/models.py +++ b/src/agentic_evals/models.py @@ -46,6 +46,8 @@ class TestCase(BaseModel): required_tools: list[str] = Field(default_factory=list) forbidden_tools: list[str] = Field(default_factory=list) required_tool_arguments: dict[str, list[str]] = Field(default_factory=dict) + expected_tool_arguments: dict[str, dict[str, Any]] = Field(default_factory=dict) + required_tool_order: list[str] = Field(default_factory=list) max_latency_ms: float | None = Field(default=None, gt=0) max_cost_usd: float | None = Field(default=None, ge=0) max_turns: int | None = Field(default=None, gt=0) @@ -64,6 +66,8 @@ def require_expectation(self) -> "TestCase": self.required_tools, self.forbidden_tools, self.required_tool_arguments, + self.expected_tool_arguments, + self.required_tool_order, self.max_latency_ms is not None, self.max_cost_usd is not None, self.max_turns is not None, @@ -98,14 +102,54 @@ class EvaluationSample(BaseModel): class Score(BaseModel): + """A single named judgment. + + `metric` is the stable key reports group by. It defaults to `name`; + checks whose name carries a per-case detail (`contains:42`, + `required_tool:lookup`) share one metric (`contains`, `required_tool`). + + A `skipped` score records that a check did not apply to a case. It + never fails the case and is left out of averages and pass rates -- + build one with `Score.skip(...)`. + """ + name: str + metric: str = "" value: float = Field(ge=0, le=1) passed: bool required: bool = True + skipped: bool = False explanation: str evaluator_type: str = "deterministic" metadata: dict[str, Any] = Field(default_factory=dict) + @model_validator(mode="after") + def default_metric_to_name(self) -> "Score": + if not self.metric: + self.metric = self.name + return self + + @classmethod + def skip( + cls, + name: str, + explanation: str, + *, + metric: str = "", + metadata: dict[str, Any] | None = None, + ) -> "Score": + """A score for a check that did not apply; `explanation` says why.""" + return cls( + name=name, + metric=metric, + value=0.0, + passed=False, + required=False, + skipped=True, + explanation=explanation, + metadata=metadata or {}, + ) + class CaseEvaluation(BaseModel): case_id: str @@ -116,6 +160,29 @@ class CaseEvaluation(BaseModel): trace_id: str latency_ms: float cost_usd: float | None = None + tags: list[str] = Field(default_factory=list) + metadata: dict[str, Any] = Field(default_factory=dict) + + +class MetricSummary(BaseModel): + """How one metric did across the suite. Skipped scores are counted apart.""" + + total: int + passed: int + failed: int + skipped: int = 0 + pass_rate: float | None = Field(default=None, ge=0, le=1) + average_score: float | None = Field(default=None, ge=0, le=1) + + +class TagSummary(BaseModel): + """How the cases carrying one tag did.""" + + total_cases: int + passed_cases: int + failed_cases: int + pass_rate: float = Field(ge=0, le=1) + average_score: float | None = Field(default=None, ge=0, le=1) class EvaluationSummary(BaseModel): @@ -126,6 +193,8 @@ class EvaluationSummary(BaseModel): average_score: float = Field(ge=0, le=1) total_cost_usd: float | None = None average_latency_ms: float + metrics: dict[str, MetricSummary] = Field(default_factory=dict) + tags: dict[str, TagSummary] = Field(default_factory=dict) class EvaluationReport(BaseModel): diff --git a/src/agentic_evals/runner.py b/src/agentic_evals/runner.py index b303901..ceb11d0 100644 --- a/src/agentic_evals/runner.py +++ b/src/agentic_evals/runner.py @@ -20,8 +20,10 @@ EvaluationSummary, HTTPTarget, LiveTarget, + MetricSummary, PythonTarget, Score, + TagSummary, TestCase, TestSuite, ) @@ -113,6 +115,34 @@ def _graded_against_expected_output(case: TestCase) -> bool: ) +def _score_tool_order(required_order: list[str], sample: EvaluationSample) -> Score: + """Each tool in `required_order` is first called before the next one is. + + Calls to other tools in between are fine; what fails is a listed tool + that was never called, or one whose first call came ahead of a tool + listed before it. + """ + called = [span.tool_name for span in sample.trace.spans if span.tool_name] + first_call = {tool: called.index(tool) for tool in required_order if tool in called} + missing = [tool for tool in required_order if tool not in first_call] + out_of_order = [ + (earlier, later) + for earlier, later in zip(required_order, required_order[1:], strict=False) + if earlier in first_call and later in first_call and first_call[earlier] > first_call[later] + ] + passed = not missing and not out_of_order + if passed: + explanation = f"Tools were first called in the required order {required_order}." + elif missing: + explanation = f"Required order {required_order} not met: {missing} never called." + else: + earlier, later = out_of_order[0] + explanation = ( + f"Required order {required_order} not met: {later!r} was called before {earlier!r}." + ) + return Score(name="tool_order", value=float(passed), passed=passed, explanation=explanation) + + def _score_case( case: TestCase, sample: EvaluationSample, @@ -137,6 +167,7 @@ def _score_case( scores.append( Score( name=f"contains:{expected}", + metric="contains", value=float(passed), passed=passed, explanation=f"Output contains required text: {expected!r}." @@ -180,6 +211,7 @@ def _score_case( scores.append( Score( name=f"required_field:{field_path}", + metric="required_field", value=float(exists), passed=exists, explanation=f"Output contains required field {field_path!r}." @@ -199,6 +231,7 @@ def _score_case( scores.append( Score( name=f"required_tool:{tool}", + metric="required_tool", value=float(passed), passed=passed, explanation=f"Required tool {tool!r} was called." @@ -211,6 +244,7 @@ def _score_case( scores.append( Score( name=f"forbidden_tool:{tool}", + metric="forbidden_tool", value=float(passed), passed=passed, explanation=f"Forbidden tool {tool!r} was not called." @@ -229,6 +263,7 @@ def _score_case( scores.append( Score( name=f"tool_args:{tool_name}", + metric="tool_args", value=float(args_present), passed=args_present, explanation=f"Tool {tool_name!r} included required arguments {required_args}." @@ -236,6 +271,36 @@ def _score_case( else f"Tool {tool_name!r} did not include required arguments {required_args}.", ) ) + for tool_name, expected_args in case.expected_tool_arguments.items(): + calls = [ + span.attributes.get("tool_args") + for span in sample.trace.spans + if span.tool_name == tool_name + ] + values_match = any( + isinstance(tool_args, dict) + and all( + arg in tool_args and tool_args[arg] == value for arg, value in expected_args.items() + ) + for tool_args in calls + ) + scores.append( + Score( + name=f"tool_arg_values:{tool_name}", + metric="tool_arg_values", + value=float(values_match), + passed=values_match, + explanation=f"Tool {tool_name!r} was called with {expected_args}." + if values_match + else ( + f"Tool {tool_name!r} was called with {calls}, not {expected_args}." + if calls + else f"Tool {tool_name!r} was not called, so {expected_args} was not passed." + ), + ) + ) + if case.required_tool_order: + scores.append(_score_tool_order(case.required_tool_order, sample)) if case.max_latency_ms is not None: passed = sample.trace.total_latency_ms <= case.max_latency_ms scores.append( @@ -371,6 +436,54 @@ def run_live_suite( return evaluate_suite(suite, samples, registry=registry) +def _mean(values: list[float]) -> float | None: + return sum(values) / len(values) if values else None + + +def _score_values(results: list[CaseEvaluation]) -> list[float]: + """Values of every score that was actually evaluated (skipped ones excluded).""" + return [score.value for result in results for score in result.scores if not score.skipped] + + +def _summarize_metrics(results: list[CaseEvaluation]) -> dict[str, MetricSummary]: + by_metric: dict[str, list[Score]] = {} + for result in results: + for score in result.scores: + by_metric.setdefault(score.metric, []).append(score) + summaries: dict[str, MetricSummary] = {} + for metric in sorted(by_metric): + scored = [score for score in by_metric[metric] if not score.skipped] + passed = sum(score.passed for score in scored) + summaries[metric] = MetricSummary( + total=len(scored), + passed=passed, + failed=len(scored) - passed, + skipped=len(by_metric[metric]) - len(scored), + pass_rate=passed / len(scored) if scored else None, + average_score=_mean([score.value for score in scored]), + ) + return summaries + + +def _summarize_tags(results: list[CaseEvaluation]) -> dict[str, TagSummary]: + by_tag: dict[str, list[CaseEvaluation]] = {} + for result in results: + for tag in dict.fromkeys(result.tags): + by_tag.setdefault(tag, []).append(result) + summaries: dict[str, TagSummary] = {} + for tag in sorted(by_tag): + tagged = by_tag[tag] + passed = sum(result.passed for result in tagged) + summaries[tag] = TagSummary( + total_cases=len(tagged), + passed_cases=passed, + failed_cases=len(tagged) - passed, + pass_rate=passed / len(tagged), + average_score=_mean(_score_values(tagged)), + ) + return summaries + + def evaluate_suite( suite: TestSuite, samples: list[EvaluationSample], @@ -412,6 +525,8 @@ def evaluate_suite( output="", trace_id="", latency_ms=0, + tags=case.tags, + metadata=case.metadata, ) ) continue @@ -420,16 +535,18 @@ def evaluate_suite( CaseEvaluation( case_id=case.id, case_name=case.name, - passed=all(score.passed or not score.required for score in scores), + passed=all(score.passed or not score.required or score.skipped for score in scores), scores=scores, output=sample.output, trace_id=sample.trace.trace_id, latency_ms=sample.trace.total_latency_ms, cost_usd=sample.trace.estimated_cost_usd, + tags=case.tags, + metadata=case.metadata, ) ) passed = sum(result.passed for result in results) - all_scores = [score.value for result in results for score in result.scores] + all_scores = _score_values(results) costs = [result.cost_usd for result in results if result.cost_usd is not None] return EvaluationReport( suite_name=suite.name, @@ -439,9 +556,11 @@ def evaluate_suite( passed_cases=passed, failed_cases=len(results) - passed, pass_rate=passed / len(results), - average_score=sum(all_scores) / len(all_scores), + average_score=_mean(all_scores) or 0.0, total_cost_usd=sum(costs) if len(costs) == len(results) else None, average_latency_ms=sum(result.latency_ms for result in results) / len(results), + metrics=_summarize_metrics(results), + tags=_summarize_tags(results), ), cases=results, ) diff --git a/src/agentic_evals/scorers/__init__.py b/src/agentic_evals/scorers/__init__.py index 79eb117..79fcb4e 100644 --- a/src/agentic_evals/scorers/__init__.py +++ b/src/agentic_evals/scorers/__init__.py @@ -30,6 +30,7 @@ TRANSLATION, LLMRubricEvaluator, RubricTemplate, + make_rubric, ) from agentic_evals.scorers.text import ( contains_all, @@ -74,6 +75,7 @@ "exact_match", "json_diff", "levenshtein_similarity", + "make_rubric", "no_redundant_tool_calls", "numeric_diff", "numeric_range", diff --git a/src/agentic_evals/scorers/rubric.py b/src/agentic_evals/scorers/rubric.py index 91680c7..a0325fe 100644 --- a/src/agentic_evals/scorers/rubric.py +++ b/src/agentic_evals/scorers/rubric.py @@ -7,8 +7,9 @@ """ import re -from collections.abc import Callable +from collections.abc import Callable, Sequence from dataclasses import dataclass +from string import ascii_uppercase from typing import Any from agentic_evals.evaluators import CallableEvaluator, EvaluationContext @@ -273,6 +274,70 @@ def verdict(found: str) -> tuple[str, float]: ) +_DEFAULT_LEVELS: tuple[tuple[str, float], ...] = ( + ("The output meets the criteria.", 1.0), + ("The output does not meet the criteria.", 0.0), +) + + +def _literal(text: str) -> str: + """Escape braces so caller-supplied text survives `str.format` in `render`.""" + return text.replace("{", "{{").replace("}", "}}") + + +def make_rubric( + name: str, + criteria: str, + *, + levels: Sequence[tuple[str, float]] | None = None, + with_reference: bool = False, +) -> RubricTemplate: + """Build a `RubricTemplate` from criteria written in plain language. + + `criteria` says what a good output looks like. `levels` is the grading + scale, best first: `(description, score)` pairs with scores in [0, 1]. + Each level is assigned a verdict letter in order (A, B, C, ...), the + same form the built-in templates use. The default scale is pass/fail. + + Set `with_reference=True` to show the judge a reference alongside the + output; `LLMRubricEvaluator` reads it from + `EvaluatorConfig.config["reference"]`. + """ + if not criteria.strip(): + raise ValueError("make_rubric requires non-empty criteria") + scale = list(levels) if levels is not None else list(_DEFAULT_LEVELS) + if not 2 <= len(scale) <= len(ascii_uppercase): + raise ValueError(f"make_rubric requires between 2 and {len(ascii_uppercase)} levels") + for description, score in scale: + if not description.strip(): + raise ValueError("every rubric level needs a description") + if not 0.0 <= score <= 1.0: + raise ValueError(f"rubric level scores must be within [0, 1], got {score}") + letters = ascii_uppercase[: len(scale)] + options = "\n".join( + f"({letter}) {_literal(description.strip())}" + for letter, (description, _) in zip(letters, scale, strict=True) + ) + reference = "Reference: {expected}\n" if with_reference else "" + prompt_template = ( + "You are grading an output against the criteria below. Grade using " + "exactly one letter:\n" + f"{options}\n\n" + f"Criteria: {_literal(criteria.strip())}\n\n" + "Input: {input}\n" + f"{reference}" + "Output: {output}\n\n" + "Respond with only the letter." + ) + return RubricTemplate( + name=name, + prompt_template=prompt_template, + verdict_scores={ + letter: float(score) for letter, (_, score) in zip(letters, scale, strict=True) + }, + ) + + class LLMRubricEvaluator(CallableEvaluator): """Grade a sample against a `RubricTemplate` using a caller-supplied model call. diff --git a/src/agentic_evals/simple.py b/src/agentic_evals/simple.py index 835316f..1dd56cc 100644 --- a/src/agentic_evals/simple.py +++ b/src/agentic_evals/simple.py @@ -139,7 +139,7 @@ def _call_score_fn(fn: ScoreFn, *, input: Any, output: Any, expected: Any) -> An def _normalize_score(raw: Any, *, default_name: str) -> dict[str, float]: if isinstance(raw, Score): - return {raw.name or default_name: raw.value} + return {} if raw.skipped else {raw.name or default_name: raw.value} if isinstance(raw, bool): return {default_name: 1.0 if raw else 0.0} if isinstance(raw, int | float): diff --git a/src/agentic_evals/skills/analyze-an-eval-experiment/SKILL.md b/src/agentic_evals/skills/analyze-an-eval-experiment/SKILL.md index 375bd4f..0fca9fb 100644 --- a/src/agentic_evals/skills/analyze-an-eval-experiment/SKILL.md +++ b/src/agentic_evals/skills/analyze-an-eval-experiment/SKILL.md @@ -22,7 +22,8 @@ influences a real decision. flipping; on a 2,000-case suite it's sixty. Treat the first as noise until `build-an-eval-dataset`/`size-a-test-suite` gives you enough cases to trust a difference that small. -3. Break results down by the tags set in `build-an-eval-dataset` — a small +3. Break results down by the tags set in `build-an-eval-dataset` (compare + `report.summary.tags` between the two reports) — a small overall improvement that's actually a large improvement on one scenario and a regression on another is a different finding (and a different decision) than a uniform small gain. diff --git a/src/agentic_evals/skills/choose-a-rubric-template/SKILL.md b/src/agentic_evals/skills/choose-a-rubric-template/SKILL.md index ee800a1..b5dd508 100644 --- a/src/agentic_evals/skills/choose-a-rubric-template/SKILL.md +++ b/src/agentic_evals/skills/choose-a-rubric-template/SKILL.md @@ -32,10 +32,13 @@ or whether none of them do and you actually need a custom one. 2. If two templates seem to both apply (e.g. `SECURITY` and `MODERATION` on the same output), register both as separate evaluators rather than picking one — they grade different things and a report should show both. -3. If nothing fits, write a new `RubricTemplate` next to the existing ones - in `rubric.py` rather than stuffing an unrelated criterion into an - existing template's prompt — see `write-a-scorer` for the general rule - on one-criterion-per-scorer. +3. If nothing fits, build a template with `make_rubric(name, criteria)` + rather than stuffing an unrelated criterion into an existing + template's prompt — see `write-a-scorer` for the general rule on + one-criterion-per-scorer. State the criterion as something a reader + could check against the output alone, and give `levels` a graded scale + only when the middle grades are distinct enough for a judge to tell + apart. ## Avoid diff --git a/src/agentic_evals/skills/define-a-release-gate/SKILL.md b/src/agentic_evals/skills/define-a-release-gate/SKILL.md index 7f993ff..365317b 100644 --- a/src/agentic_evals/skills/define-a-release-gate/SKILL.md +++ b/src/agentic_evals/skills/define-a-release-gate/SKILL.md @@ -28,6 +28,10 @@ just accept the defaults. 4. Treat a gate failure on cost/latency as at least as actionable as a pass-rate failure — regressions there are often the first sign of a prompt or retry-loop change before quality visibly degrades. +5. When some cases matter more than the overall rate allows for, set + `min_tag_pass_rate` or `min_metric_pass_rate` for that slice instead + of raising `min_pass_rate` for everything — for example a tag that + must pass in full while the suite as a whole tolerates a few misses. ## Avoid diff --git a/src/agentic_evals/skills/discover-failure-modes/SKILL.md b/src/agentic_evals/skills/discover-failure-modes/SKILL.md index 6dc86f5..9bfc78b 100644 --- a/src/agentic_evals/skills/discover-failure-modes/SKILL.md +++ b/src/agentic_evals/skills/discover-failure-modes/SKILL.md @@ -14,11 +14,13 @@ filing a single vague bug ("agent sometimes gets things wrong"). ## Do 1. Group failing `CaseEvaluation`s by which scorer failed them - (`Score.name`, `passed=False`), not just by case — a suite where ten + (`Score.metric`, `passed=False`), not just by case — a suite where ten cases fail `tool_call_precision` and two fail `numeric_diff` has two - distinct problems, not twelve. + distinct problems, not twelve. `report.summary.metrics` already holds + the per-metric pass and fail counts; start there. 2. Cross-reference failures against the tags set in `build-an-eval-dataset` - — a failure mode concentrated in one scenario tag points at a specific + (`report.summary.tags`, and `CaseEvaluation.tags` on each case) — a + failure mode concentrated in one scenario tag points at a specific fix; one spread evenly across all tags points at something systemic (a prompt regression, a shared tool wrapper). 3. Read `Score.explanation` on every failure in a candidate cluster before diff --git a/src/agentic_evals/skills/write-a-scorer/SKILL.md b/src/agentic_evals/skills/write-a-scorer/SKILL.md index 971d634..18dc7ef 100644 --- a/src/agentic_evals/skills/write-a-scorer/SKILL.md +++ b/src/agentic_evals/skills/write-a-scorer/SKILL.md @@ -33,6 +33,10 @@ aren't sure whether to reach for a built-in scorer, write a custom the final text. 6. If none of the above fit, write a plain `CallableEvaluator(name, fn)` — `fn` takes an `EvaluationContext` and returns a `Score` or `list[Score]`. +7. When the criterion does not apply to a case (no order id to check, no + tool expected), return `Score.skip(name, reason)` rather than a pass — + a skipped score never fails the case and stays out of the averages, + so the metric's pass rate reflects only the cases it actually judged. ## Avoid @@ -47,9 +51,11 @@ aren't sure whether to reach for a built-in scorer, write a custom ## Check -- The scorer raises a clear `ValueError` when its required input - (`expected_output`, `case.input`, a trace with tool calls) is missing, - rather than silently returning a meaningless score. +- The scorer raises a clear `ValueError` when its configuration + (`expected_output`, `case.input`, an `allowed_tools` list) is missing, + rather than silently returning a meaningless score. What the agent did + or failed to do — an empty trace, a wrong answer — is a result to + score, not an error to raise. - `Score.value` stays in `[0, 1]`; a hard pass/fail scorer returns exactly `0.0` or `1.0`, a graded one returns a genuine gradient. - You've tested the scorer directly against at least one passing and one diff --git a/tests/test_gate.py b/tests/test_gate.py index 64882ae..f6dbddb 100644 --- a/tests/test_gate.py +++ b/tests/test_gate.py @@ -1,3 +1,5 @@ +import pytest + from agentic_evals import ( EvalTrace, EvaluationReport, @@ -107,3 +109,66 @@ def test_gate_passes_when_thresholds_are_met() -> None: decision = evaluate_gate(report, GateConfig()) assert decision.passed assert decision.reasons == [] + + +def _tagged_report() -> EvaluationReport: + suite = TestSuite( + name="release", + version="1", + cases=[ + TestCase(id="s1", name="s1", expected_contains=["no"], tags=["safety"]), + TestCase(id="s2", name="s2", expected_contains=["no"], tags=["safety"]), + TestCase(id="q1", name="q1", expected_output="ok", tags=["quality"]), + TestCase(id="q2", name="q2", expected_output="ok", tags=["quality"]), + ], + ) + outputs = {"s1": "no", "s2": "no", "q1": "ok", "q2": "wrong"} + samples = [ + EvaluationSample(case_id=case_id, output=output, trace=make_trace()) + for case_id, output in outputs.items() + ] + return evaluate_suite(suite, samples) + + +def test_gate_holds_one_tag_to_a_stricter_bar_than_the_suite() -> None: + report = _tagged_report() + base = {"min_pass_rate": 0.75} + + assert evaluate_gate(report, GateConfig(**base, min_tag_pass_rate={"safety": 1.0})).passed + + decision = evaluate_gate(report, GateConfig(**base, min_tag_pass_rate={"quality": 1.0})) + assert decision.reasons == ["Tag 'quality' pass rate 50.0% is below 100.0%."] + assert decision.observed["tag_pass_rate:quality"] == 0.5 + + +def test_gate_checks_metric_pass_rates() -> None: + report = _tagged_report() + base = {"min_pass_rate": 0.75} + + assert evaluate_gate(report, GateConfig(**base, min_metric_pass_rate={"contains": 1.0})).passed + + decision = evaluate_gate(report, GateConfig(**base, min_metric_pass_rate={"exact_match": 0.9})) + assert decision.reasons == ["Metric 'exact_match' pass rate 50.0% is below 90.0%."] + assert decision.observed["metric_pass_rate:exact_match"] == 0.5 + + +def test_gate_fails_on_a_tag_or_metric_missing_from_the_report() -> None: + decision = evaluate_gate( + _tagged_report(), + GateConfig( + min_pass_rate=0.75, + min_tag_pass_rate={"billing": 0.5}, + min_metric_pass_rate={"groundedness": 0.5}, + ), + ) + + assert decision.reasons == [ + "Tag 'billing' has no cases in the report.", + "Metric 'groundedness' has no evaluated scores in the report.", + ] + assert decision.observed["tag_pass_rate:billing"] is None + + +def test_gate_rejects_an_out_of_range_slice_threshold() -> None: + with pytest.raises(ValueError): + GateConfig(min_tag_pass_rate={"safety": 1.5}) diff --git a/tests/test_report_breakdowns.py b/tests/test_report_breakdowns.py new file mode 100644 index 0000000..89e66c0 --- /dev/null +++ b/tests/test_report_breakdowns.py @@ -0,0 +1,262 @@ +from agentic_evals import ( + BusinessRuleEvaluator, + Eval, + EvalSpan, + EvalTrace, + EvaluationContext, + EvaluationReport, + EvaluationSample, + EvaluatorConfig, + EvaluatorRegistry, + Score, + TestCase, + TestSuite, + evaluate_suite, +) + + +def _sample(case_id: str, output: str, *tools: str) -> EvaluationSample: + trace = EvalTrace(trace_id=case_id, spans=[EvalSpan(tool_name=tool) for tool in tools]) + return EvaluationSample(case_id=case_id, output=output, trace=trace) + + +def _report() -> EvaluationReport: + suite = TestSuite( + name="support", + version="1", + cases=[ + TestCase( + id="refund-ok", + name="refund ok", + expected_contains=["refund", "5 days"], + required_tools=["lookup_order"], + tags=["refunds", "happy-path"], + metadata={"owner": "payments"}, + ), + TestCase( + id="refund-no-lookup", + name="refund without lookup", + expected_contains=["refund"], + required_tools=["lookup_order"], + tags=["refunds"], + ), + TestCase( + id="tracking", + name="tracking", + expected_contains=["shipped"], + tags=["tracking", "happy-path"], + ), + TestCase(id="untagged", name="untagged", expected_contains=["hello"]), + ], + ) + samples = [ + _sample("refund-ok", "Your refund arrives in 5 days.", "lookup_order"), + _sample("refund-no-lookup", "Your refund is on its way."), + _sample("tracking", "Your order has shipped."), + _sample("untagged", "goodbye"), + ] + return evaluate_suite(suite, samples) + + +def test_implicit_checks_share_a_metric_and_keep_their_detailed_name() -> None: + scores = _report().cases[0].scores + + assert [(score.name, score.metric) for score in scores] == [ + ("contains:refund", "contains"), + ("contains:5 days", "contains"), + ("required_tool:lookup_order", "required_tool"), + ] + + +def test_metric_defaults_to_the_score_name() -> None: + score = Score(name="groundedness", value=1.0, passed=True, explanation="ok") + assert score.metric == "groundedness" + + +def test_summary_breaks_results_down_by_metric() -> None: + metrics = _report().summary.metrics + + assert list(metrics) == ["contains", "required_tool"] + assert metrics["contains"].model_dump() == { + "total": 5, + "passed": 4, + "failed": 1, + "skipped": 0, + "pass_rate": 0.8, + "average_score": 0.8, + } + assert metrics["required_tool"].pass_rate == 0.5 + + +def test_summary_breaks_results_down_by_tag() -> None: + tags = _report().summary.tags + + assert list(tags) == ["happy-path", "refunds", "tracking"] + assert tags["refunds"].model_dump() == { + "total_cases": 2, + "passed_cases": 1, + "failed_cases": 1, + "pass_rate": 0.5, + "average_score": 0.8, + } + assert tags["happy-path"].pass_rate == 1.0 + assert tags["tracking"].total_cases == 1 + + +def test_case_results_carry_tags_and_metadata() -> None: + report = _report() + + assert report.cases[0].tags == ["refunds", "happy-path"] + assert report.cases[0].metadata == {"owner": "payments"} + assert report.cases[3].tags == [] + + +def test_missing_sample_still_counts_toward_its_tags() -> None: + suite = TestSuite( + name="support", + version="1", + cases=[TestCase(id="c", name="c", expected_contains=["x"], tags=["refunds"])], + ) + report = evaluate_suite(suite, []) + + assert report.cases[0].tags == ["refunds"] + assert report.summary.tags["refunds"].failed_cases == 1 + assert report.summary.metrics["sample_available"].failed == 1 + + +def _order_id_rule(context: EvaluationContext) -> Score: + order_id = context.case.metadata.get("order_id") + if order_id is None: + return Score.skip("cites_order_id", "Case has no order id to cite.") + cited = order_id in context.sample.output + return Score( + name="cites_order_id", + value=float(cited), + passed=cited, + explanation=f"Output {'cites' if cited else 'does not cite'} {order_id}.", + ) + + +def _skip_report() -> EvaluationReport: + registry = EvaluatorRegistry() + registry.register(BusinessRuleEvaluator("cites_order_id", _order_id_rule)) + evaluators = [EvaluatorConfig(name="cites_order_id")] + suite = TestSuite( + name="support", + version="1", + cases=[ + TestCase( + id="with-order", + name="with order", + evaluators=evaluators, + metadata={"order_id": "ORD-1"}, + ), + TestCase(id="no-order", name="no order", evaluators=evaluators), + ], + ) + samples = [_sample("with-order", "ORD-1 has shipped."), _sample("no-order", "Hello!")] + return evaluate_suite(suite, samples, registry=registry) + + +def test_skipped_score_does_not_fail_the_case() -> None: + report = _skip_report() + skipped = report.cases[1].scores[0] + + assert skipped.skipped is True + assert skipped.explanation == "Case has no order id to cite." + assert skipped.evaluator_type == "business_rule" + assert report.cases[1].passed + assert report.summary.pass_rate == 1.0 + + +def test_skipped_scores_are_left_out_of_averages_and_counted_apart() -> None: + report = _skip_report() + metric = report.summary.metrics["cites_order_id"] + + assert report.summary.average_score == 1.0 + assert (metric.total, metric.passed, metric.skipped) == (1, 1, 1) + assert metric.pass_rate == 1.0 + + +def test_metric_with_only_skipped_scores_has_no_pass_rate() -> None: + registry = EvaluatorRegistry() + registry.register(BusinessRuleEvaluator("cites_order_id", _order_id_rule)) + suite = TestSuite( + name="support", + version="1", + cases=[ + TestCase( + id="no-order", + name="no order", + evaluators=[EvaluatorConfig(name="cites_order_id")], + tags=["greeting"], + ) + ], + ) + report = evaluate_suite(suite, [_sample("no-order", "Hello!")], registry=registry) + metric = report.summary.metrics["cites_order_id"] + + assert (metric.total, metric.skipped) == (0, 1) + assert metric.pass_rate is None + assert metric.average_score is None + assert report.summary.tags["greeting"].average_score is None + assert report.summary.average_score == 0.0 + + +def test_eval_ignores_a_skipped_score() -> None: + def maybe(output: str) -> Score: + return Score.skip("maybe", "Not applicable.") + + result = Eval("skips", data=[{"input": "x"}], task=str, scores=[maybe], print_results=False) + + assert result.results[0].scores == {} + assert bool(result) + + +def test_report_round_trips_through_json() -> None: + report = _report() + restored = EvaluationReport.model_validate_json(report.model_dump_json()) + + assert restored.summary.tags == report.summary.tags + assert restored.summary.metrics == report.summary.metrics + assert restored.cases[0].scores[0].metric == "contains" + + +def test_reports_without_the_new_fields_still_load() -> None: + legacy = _report().model_dump(mode="json") + del legacy["summary"]["metrics"], legacy["summary"]["tags"] + for case in legacy["cases"]: + del case["tags"], case["metadata"] + for score in case["scores"]: + del score["metric"], score["skipped"] + + restored = EvaluationReport.model_validate(legacy) + + assert restored.summary.metrics == {} + assert restored.cases[0].scores[0].metric == "contains:refund" + assert restored.cases[0].scores[0].skipped is False + + +def test_suite_without_tags_has_an_empty_tag_breakdown() -> None: + suite = TestSuite( + name="s", version="1", cases=[TestCase(id="c", name="c", expected_contains=["x"])] + ) + report = evaluate_suite(suite, [_sample("c", "x")]) + + assert report.summary.tags == {} + assert list(report.summary.metrics) == ["contains"] + + +def test_suite_where_no_evaluator_returns_a_score_reports_zero_average() -> None: + registry = EvaluatorRegistry() + registry.register(BusinessRuleEvaluator("silent", lambda context: [])) + suite = TestSuite( + name="s", + version="1", + cases=[TestCase(id="c", name="c", evaluators=[EvaluatorConfig(name="silent")])], + ) + report = evaluate_suite(suite, [_sample("c", "x")], registry=registry) + + assert report.summary.average_score == 0.0 + assert report.summary.metrics == {} + assert report.cases[0].passed diff --git a/tests/test_scorers_rubric.py b/tests/test_scorers_rubric.py index ac919c6..5b1cdf4 100644 --- a/tests/test_scorers_rubric.py +++ b/tests/test_scorers_rubric.py @@ -15,6 +15,7 @@ EvaluatorConfig, LLMRubricEvaluator, TestCase, + make_rubric, ) @@ -182,3 +183,80 @@ def test_llm_rubric_evaluator_works_with_new_templates() -> None: assert score.name == "security" assert score.value == 1.0 assert score.passed is True + + +def test_make_rubric_defaults_to_a_pass_fail_scale() -> None: + rubric = make_rubric("concise", "The answer is at most two sentences.") + + assert rubric.name == "concise" + assert rubric.verdict_scores == {"A": 1.0, "B": 0.0} + + prompt = rubric.render(output="Short answer.", expected=None, input="Summarise this.") + assert "Criteria: The answer is at most two sentences." in prompt + assert "(A) The output meets the criteria." in prompt + assert "Input: Summarise this." in prompt + assert "Output: Short answer." in prompt + assert "Reference" not in prompt + + +def test_make_rubric_accepts_a_graded_scale() -> None: + rubric = make_rubric( + "tone", + "The reply is courteous.", + levels=[ + ("Courteous throughout.", 1.0), + ("Neutral, neither courteous nor rude.", 0.5), + ("Rude or dismissive.", 0.0), + ], + ) + + assert rubric.verdict_scores == {"A": 1.0, "B": 0.5, "C": 0.0} + assert "(B) Neutral, neither courteous nor rude." in rubric.prompt_template + assert rubric.parse_verdict("(B)") == ("B", 0.5) + + +def test_make_rubric_shows_the_reference_only_when_asked() -> None: + rubric = make_rubric("matches", "Agrees with the reference.", with_reference=True) + prompt = rubric.render(output="4", expected="four", input="2 + 2") + + assert "Reference: four" in prompt + + +def test_make_rubric_keeps_braces_in_criteria_literal() -> None: + rubric = make_rubric("json", 'The output is JSON shaped like {"id": int}.') + prompt = rubric.render(output="{}", expected=None, input=None) + + assert 'shaped like {"id": int}.' in prompt + + +@pytest.mark.parametrize( + ("criteria", "levels", "message"), + [ + (" ", None, "non-empty criteria"), + ("ok", [("only one", 1.0)], "between 2 and 26 levels"), + ("ok", [("good", 1.5), ("bad", 0.0)], r"within \[0, 1\]"), + ("ok", [("good", 1.0), (" ", 0.0)], "needs a description"), + ], +) +def test_make_rubric_rejects_an_unusable_definition( + criteria: str, levels: list[tuple[str, float]] | None, message: str +) -> None: + with pytest.raises(ValueError, match=message): + make_rubric("r", criteria, levels=levels) + + +def test_make_rubric_template_runs_through_the_evaluator() -> None: + prompts: list[str] = [] + + def complete(prompt: str) -> str: + prompts.append(prompt) + return "B" + + rubric = make_rubric("concise", "The answer is at most two sentences.") + evaluator = LLMRubricEvaluator("concise", rubric, complete_fn=complete) + score = evaluator.evaluate(_context(output="A very long answer...", input_="Summarise."))[0] + + assert score.name == "concise" + assert score.value == 0.0 + assert score.passed is False + assert "Output: A very long answer..." in prompts[0] diff --git a/tests/test_tool_expectations.py b/tests/test_tool_expectations.py new file mode 100644 index 0000000..fc33158 --- /dev/null +++ b/tests/test_tool_expectations.py @@ -0,0 +1,125 @@ +import pytest + +from agentic_evals import ( + EvalSpan, + EvalTrace, + EvaluationSample, + Score, + TestCase, + TestSuite, + evaluate_suite, +) + + +def _call(tool: str, **tool_args: object) -> EvalSpan: + return EvalSpan(tool_name=tool, attributes={"tool_args": tool_args}) + + +def _scores(case: TestCase, *spans: EvalSpan) -> dict[str, Score]: + suite = TestSuite(name="tools", version="1", cases=[case]) + sample = EvaluationSample(case_id=case.id, output="done", trace=EvalTrace(spans=list(spans))) + return {score.name: score for score in evaluate_suite(suite, [sample]).cases[0].scores} + + +def _values_case(**expected: dict[str, object]) -> TestCase: + return TestCase(id="c", name="c", expected_tool_arguments=expected) + + +def test_expected_tool_arguments_pass_when_a_call_carries_the_values() -> None: + case = _values_case(search={"query": "invoice 88", "limit": 5}) + score = _scores(case, _call("search", query="invoice 88", limit=5, lang="en"))[ + "tool_arg_values:search" + ] + + assert score.passed + assert score.metric == "tool_arg_values" + + +def test_expected_tool_arguments_fail_on_a_wrong_value() -> None: + case = _values_case(search={"query": "invoice 88"}) + score = _scores(case, _call("search", query="invoice 89"))["tool_arg_values:search"] + + assert not score.passed + assert "invoice 89" in score.explanation + + +def test_expected_tool_arguments_do_not_coerce_types() -> None: + case = _values_case(search={"limit": 5}) + assert not _scores(case, _call("search", limit="5"))["tool_arg_values:search"].passed + + +def test_expected_tool_arguments_match_any_call_to_the_tool() -> None: + case = _values_case(search={"query": "b"}) + spans = (_call("search", query="a"), _call("search", query="b")) + + assert _scores(case, *spans)["tool_arg_values:search"].passed + + +def test_expected_tool_arguments_need_one_call_with_every_value() -> None: + case = _values_case(search={"query": "a", "limit": 5}) + spans = (_call("search", query="a", limit=1), _call("search", query="z", limit=5)) + + assert not _scores(case, *spans)["tool_arg_values:search"].passed + + +def test_expected_tool_arguments_fail_when_the_tool_was_never_called() -> None: + case = _values_case(search={"query": "a"}) + score = _scores(case, _call("fetch", url="x"))["tool_arg_values:search"] + + assert not score.passed + assert "was not called" in score.explanation + + +def test_expected_tool_arguments_compare_nested_values() -> None: + case = _values_case(write={"record": {"id": 7, "tags": ["a", "b"]}}) + + assert _scores(case, _call("write", record={"id": 7, "tags": ["a", "b"]}))[ + "tool_arg_values:write" + ].passed + assert not _scores(case, _call("write", record={"id": 7, "tags": ["b", "a"]}))[ + "tool_arg_values:write" + ].passed + + +def _order_case(*order: str) -> TestCase: + return TestCase(id="c", name="c", required_tool_order=list(order)) + + +@pytest.mark.parametrize( + "called", + [ + ["authenticate", "read", "write"], + ["authenticate", "log", "read", "log", "write"], + ["authenticate", "read", "read", "write", "authenticate"], + ], +) +def test_required_tool_order_passes_when_first_calls_are_in_order(called: list[str]) -> None: + case = _order_case("authenticate", "read", "write") + assert _scores(case, *(_call(tool) for tool in called))["tool_order"].passed + + +def test_required_tool_order_fails_when_a_later_tool_is_called_first() -> None: + case = _order_case("authenticate", "write") + score = _scores(case, _call("write"), _call("authenticate"), _call("write"))["tool_order"] + + assert not score.passed + assert "'write' was called before 'authenticate'" in score.explanation + + +def test_required_tool_order_fails_when_a_listed_tool_is_missing() -> None: + case = _order_case("authenticate", "read", "write") + score = _scores(case, _call("authenticate"), _call("write"))["tool_order"] + + assert not score.passed + assert "['read'] never called" in score.explanation + + +def test_required_tool_order_fails_on_an_empty_trace() -> None: + assert not _scores(_order_case("authenticate", "write"))["tool_order"].passed + + +def test_tool_expectations_alone_satisfy_the_expectation_requirement() -> None: + assert _values_case(search={"query": "a"}).expected_tool_arguments + assert _order_case("a", "b").required_tool_order + with pytest.raises(ValueError, match="at least one expectation"): + TestCase(id="c", name="c") diff --git a/uv.lock b/uv.lock index 63f7275..f6e22af 100644 --- a/uv.lock +++ b/uv.lock @@ -9,7 +9,7 @@ resolution-markers = [ [[package]] name = "agentic-evals" -version = "0.6.0" +version = "0.7.0" source = { editable = "." } dependencies = [ { name = "jsonschema" }, diff --git a/voice-agent-evals/run_evals.py b/voice-agent-evals/run_evals.py index b5f06e2..d715e13 100644 --- a/voice-agent-evals/run_evals.py +++ b/voice-agent-evals/run_evals.py @@ -25,8 +25,6 @@ MAX_LATENCY_MS = 5000.0 -CAT_BY_ID = {tc["id"]: tc["category"] for tc in VOICE_TEST_CASES} - # Evaluators moved to evaluators.py (single source of truth). registry = build_registry() @@ -59,6 +57,9 @@ def build_test_suite() -> TestSuite: required_tool_arguments={tc["expected_tool"]: list(tc["expected_arguments"].keys())} if "expected_arguments" in tc else {}, + expected_tool_arguments={tc["expected_tool"]: tc["expected_arguments"]} + if "expected_arguments" in tc + else {}, max_latency_ms=MAX_LATENCY_MS, evaluators=evaluators, metadata={ @@ -135,7 +136,7 @@ def print_scorecard(report: EvaluationReport) -> None: for case_eval in report.cases: status = "PASS" if case_eval.passed else "FAIL" - category = CAT_BY_ID.get(case_eval.case_id, "unknown") + category = case_eval.tags[0] if case_eval.tags else "unknown" print(f"{case_eval.case_id:20} {category:18} {status:5} {case_eval.latency_ms:.2f} ms") @@ -148,28 +149,22 @@ def print_scorecard(report: EvaluationReport) -> None: print("\nMETRIC PASS RATES:") print("-" * 70) - metric_scores: dict[str, list[float]] = {} - metric_passed: dict[str, int] = {} - metric_total: dict[str, int] = {} + for name, metric in report.summary.metrics.items(): + if metric.pass_rate is None or metric.average_score is None: + continue + print( + f" {name:30} avg={metric.average_score:.3f} " + f"pass={metric.passed}/{metric.total} ({metric.pass_rate:.1%})" + ) - for case_eval in report.cases: - for score in case_eval.scores: - name = score.name - if name not in metric_scores: - metric_scores[name] = [] - metric_passed[name] = 0 - metric_total[name] = 0 - metric_scores[name].append(score.value) - metric_total[name] += 1 - if score.passed: - metric_passed[name] += 1 - - for name in sorted(metric_scores.keys()): - avg_score = sum(metric_scores[name]) / len(metric_scores[name]) - pass_rate = (metric_passed[name] / metric_total[name]) * 100 + print("-" * 70) + + print("\nCATEGORY PASS RATES:") + print("-" * 70) + + for tag, tagged in report.summary.tags.items(): print( - f" {name:30} avg={avg_score:.3f} " - f"pass={metric_passed[name]}/{metric_total[name]} ({pass_rate:.1f}%)" + f" {tag:30} pass={tagged.passed_cases}/{tagged.total_cases} ({tagged.pass_rate:.1%})" ) print("-" * 70) diff --git a/voice-agent-evals/scenario_runner.py b/voice-agent-evals/scenario_runner.py index 6e9e115..44b6d0d 100644 --- a/voice-agent-evals/scenario_runner.py +++ b/voice-agent-evals/scenario_runner.py @@ -32,18 +32,19 @@ "reliability": "reliability", } +# Built-in checks, keyed by `Score.metric`, that back a differently named rule. +_RULE_FOR_BUILTIN_METRIC = { + "required_tool": "tool_selection", + "contains": "response_quality", + "latency_threshold": "latency", +} + -def score_to_rule(name: str) -> str | None: - """Map a score name back to its rule key (None = unmapped).""" - if name in _EVALUATOR_FOR_RULE: - return name - if name.startswith("required_tool:"): - return "tool_selection" - if name.startswith("contains:"): - return "response_quality" - if name == "latency_threshold": - return "latency" - return None +def score_to_rule(metric: str) -> str | None: + """Map a score's metric back to its rule key (None = unmapped).""" + if metric in _EVALUATOR_FOR_RULE: + return metric + return _RULE_FOR_BUILTIN_METRIC.get(metric) # ----------------------------- @@ -171,7 +172,7 @@ def _metrics( """Aggregate case scores into one entry per rule key, in display order.""" by_rule: dict[str, list[Score]] = {} for score in case_eval.scores: - key = score_to_rule(score.name) + key = score_to_rule(score.metric) if key is not None: by_rule.setdefault(key, []).append(score)