Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
16 commits
Select commit Hold shift + click to select a range
dfc632a
fix(evaluation): NOT_EVALUATED metric no longer masked by a passing one
gaurav-gandhi-2411 Aug 11, 2026
49f1b4f
Merge branch 'main' into fix/local-eval-service-not-evaluated-verdict
gaurav-gandhi-2411 Aug 12, 2026
9df411f
Merge branch 'main' into fix/local-eval-service-not-evaluated-verdict
gaurav-gandhi-2411 Aug 12, 2026
e37af54
Merge branch 'main' into fix/local-eval-service-not-evaluated-verdict
gaurav-gandhi-2411 Aug 12, 2026
d7033fa
Merge branch 'main' into fix/local-eval-service-not-evaluated-verdict
gaurav-gandhi-2411 Aug 13, 2026
6254e7b
Merge branch 'main' into fix/local-eval-service-not-evaluated-verdict
gaurav-gandhi-2411 Aug 13, 2026
ee88b60
Merge branch 'main' into fix/local-eval-service-not-evaluated-verdict
gaurav-gandhi-2411 Aug 13, 2026
2597564
Merge branch 'main' into fix/local-eval-service-not-evaluated-verdict
gaurav-gandhi-2411 Aug 13, 2026
5589656
Merge branch 'main' into fix/local-eval-service-not-evaluated-verdict
gaurav-gandhi-2411 Aug 13, 2026
6a557f4
Merge branch 'main' into fix/local-eval-service-not-evaluated-verdict
gaurav-gandhi-2411 Aug 13, 2026
eec1a37
Merge branch 'main' into fix/local-eval-service-not-evaluated-verdict
gaurav-gandhi-2411 Aug 13, 2026
bc7512c
Merge branch 'main' into fix/local-eval-service-not-evaluated-verdict
gaurav-gandhi-2411 Aug 14, 2026
294d309
Merge branch 'main' into fix/local-eval-service-not-evaluated-verdict
gaurav-gandhi-2411 Aug 21, 2026
143e9bf
Merge branch 'main' into fix/local-eval-service-not-evaluated-verdict
gaurav-gandhi-2411 Aug 21, 2026
fe33ff9
Merge branch 'main' into fix/local-eval-service-not-evaluated-verdict
gaurav-gandhi-2411 Aug 21, 2026
9deddd0
Merge branch 'main' into fix/local-eval-service-not-evaluated-verdict
gaurav-gandhi-2411 Aug 22, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
9 changes: 9 additions & 0 deletions src/google/adk/evaluation/local_eval_service.py
Original file line number Diff line number Diff line change
Expand Up @@ -495,6 +495,7 @@ def _generate_final_eval_status(
self, overall_eval_metric_results: list[EvalMetricResult]
) -> EvalStatus:
final_eval_status = EvalStatus.NOT_EVALUATED
has_not_evaluated_metric = False
# Go over all the eval statuses and mark the final eval status as
# passed if all of them pass; otherwise, mark the final eval status to
# failed.
Expand All @@ -503,13 +504,21 @@ def _generate_final_eval_status(
if overall_eval_status == EvalStatus.PASSED:
final_eval_status = EvalStatus.PASSED
elif overall_eval_status == EvalStatus.NOT_EVALUATED:
has_not_evaluated_metric = True
continue
elif overall_eval_status == EvalStatus.FAILED:
final_eval_status = EvalStatus.FAILED
break
else:
raise ValueError(f"Unknown eval status: {overall_eval_status}.")

# A metric that never produced a verdict (e.g. a judge call that crashed)
# must not be silently masked by another metric that happened to pass --
# only report PASSED when every requested metric was actually evaluated.
# A genuine FAILED is real evidence and still takes precedence.
if has_not_evaluated_metric and final_eval_status == EvalStatus.PASSED:
final_eval_status = EvalStatus.NOT_EVALUATED

return final_eval_status

async def _perform_inference_single_eval_item(
Expand Down
90 changes: 90 additions & 0 deletions tests/unittests/evaluation/test_local_eval_service.py
Original file line number Diff line number Diff line change
Expand Up @@ -629,6 +629,96 @@ def test_generate_final_eval_status_doesn_t_throw_on(eval_service):
eval_service._generate_final_eval_status([eval_metric_result])


def test_generate_final_eval_status_not_evaluated_then_passed_is_not_evaluated(
eval_service,
):
"""A metric that never produced a verdict must not be masked by a later PASSED.

If metric_1 crashed (NOT_EVALUATED) and metric_2 passed, the eval case did
not actually pass every requested metric -- it should be reported as
NOT_EVALUATED, not PASSED.
"""
results = [
EvalMetricResult(
metric_name="metric1",
threshold=0.5,
eval_status=EvalStatus.NOT_EVALUATED,
),
EvalMetricResult(
metric_name="metric2", threshold=0.5, eval_status=EvalStatus.PASSED
),
]

assert (
eval_service._generate_final_eval_status(results)
== EvalStatus.NOT_EVALUATED
)


def test_generate_final_eval_status_passed_then_not_evaluated_is_not_evaluated(
eval_service,
):
"""Same as above with the metrics in the opposite order.

Both orderings of [PASSED, NOT_EVALUATED] must produce the same final
status. Prior to this fix they didn't: this ordering silently returned
PASSED, which is exactly the evidence that the old behavior was a bug
and not intended.
"""
results = [
EvalMetricResult(
metric_name="metric1", threshold=0.5, eval_status=EvalStatus.PASSED
),
EvalMetricResult(
metric_name="metric2",
threshold=0.5,
eval_status=EvalStatus.NOT_EVALUATED,
),
]

assert (
eval_service._generate_final_eval_status(results)
== EvalStatus.NOT_EVALUATED
)


def test_generate_final_eval_status_failed_dominates_not_evaluated(
eval_service,
):
"""A genuine FAILED is real evidence and must still win over NOT_EVALUATED,
regardless of order -- only the PASSED-vs-NOT_EVALUATED case was buggy.
"""
failed_then_not_evaluated = [
EvalMetricResult(
metric_name="metric1", threshold=0.5, eval_status=EvalStatus.FAILED
),
EvalMetricResult(
metric_name="metric2",
threshold=0.5,
eval_status=EvalStatus.NOT_EVALUATED,
),
]
not_evaluated_then_failed = [
EvalMetricResult(
metric_name="metric1",
threshold=0.5,
eval_status=EvalStatus.NOT_EVALUATED,
),
EvalMetricResult(
metric_name="metric2", threshold=0.5, eval_status=EvalStatus.FAILED
),
]

assert (
eval_service._generate_final_eval_status(failed_then_not_evaluated)
== EvalStatus.FAILED
)
assert (
eval_service._generate_final_eval_status(not_evaluated_then_failed)
== EvalStatus.FAILED
)


@pytest.mark.asyncio
async def test_mcp_stdio_agent_no_runtime_error(mocker):
"""Test that LocalEvalService can handle MCP stdio agents without RuntimeError.
Expand Down