diff --git a/packages/client/src/launchdarkly_ai_server/evaluations/module.py b/packages/client/src/launchdarkly_ai_server/evaluations/module.py index 4cc684aa..4571f826 100644 --- a/packages/client/src/launchdarkly_ai_server/evaluations/module.py +++ b/packages/client/src/launchdarkly_ai_server/evaluations/module.py @@ -63,6 +63,25 @@ def _is_terminal_summary(summary: RunSummary) -> bool: ) +def _run_passed(summary: RunSummary, criteria: list[Criterion]) -> bool: + """Return whether a run meets its gate. + + When every criterion sets ``pass_rate_threshold``, the run passes if no + row is pending and the share of passed rows is at least the highest + threshold. Errored rows count as not passed. Otherwise the run passes + only if no row failed, errored, or is pending. + """ + if summary.pending_rows != 0: + return False + thresholds = [criterion.pass_rate_threshold for criterion in criteria] + if thresholds and all(threshold is not None for threshold in thresholds): + if summary.total_rows <= 0: + return False + required = max(t for t in thresholds if t is not None) + return summary.passed_rows / summary.total_rows >= required + return summary.error_rows == 0 and summary.failed_rows == 0 + + def _merge_generation( base: GenerationConfig, override: GenerationConfig | None ) -> GenerationConfig: @@ -309,16 +328,7 @@ async def run( f"{_segment(evaluation.id)}/runs/{_segment(evaluation_run.id)}" ) return EvalRunResult( - # failed_rows counts rows whose criteria were scored and did not - # meet their threshold, so a gate that ignores it exits 0 on a run - # where every row failed its judge. It was omissible while runs were - # generation-only -- a row either generated or errored, and nothing - # produced a fail -- and stops being so the moment criteria exist. - passed=( - summary.error_rows == 0 - and summary.failed_rows == 0 - and summary.pending_rows == 0 - ), + passed=_run_passed(summary, run_criteria), url=url, run_id=evaluation_run.id, summary=summary, diff --git a/packages/client/tests/test_evaluations_run.py b/packages/client/tests/test_evaluations_run.py index 4a4b8571..9681634d 100644 --- a/packages/client/tests/test_evaluations_run.py +++ b/packages/client/tests/test_evaluations_run.py @@ -2965,3 +2965,46 @@ def test_ai_config_variation_from_api_layers_the_model_config() -> None: unlinked = AIConfigVariation.from_api(latest) assert "provider" not in unlinked.generation assert unlinked.generation["parameters"] == {"temperature": 0.7} + + +def _scorer(**options: Any) -> Scorer: + return Scorer(name="accuracy", fn=lambda row, output: True, **options) + + +@pytest.mark.parametrize( + ("counts", "criteria", "expected"), + [ + # 28 of 30 rows passed: 93% meets a 0.9 pass rate. + ((30, 28, 1, 1, 0), [_scorer(pass_rate_threshold=0.9)], True), + # The strictest pass rate across criteria applies. + ( + (30, 28, 1, 1, 0), + [_scorer(pass_rate_threshold=0.9), _scorer(pass_rate_threshold=0.95)], + False, + ), + # Without a pass rate on every criterion, any failed row fails the run. + ((30, 28, 1, 1, 0), [_scorer(pass_rate_threshold=0.9), _scorer()], False), + ((30, 28, 1, 1, 0), [_scorer()], False), + ((30, 30, 0, 0, 0), [_scorer()], True), + # A pending row never passes. + ((30, 29, 0, 0, 1), [_scorer(pass_rate_threshold=0.5)], False), + # An empty run never passes a pass rate. + ((0, 0, 0, 0, 0), [_scorer(pass_rate_threshold=0.0)], False), + ], +) +def test_run_passed_honours_pass_rate_threshold( + counts: tuple[int, int, int, int, int], criteria: list[Scorer], expected: bool +) -> None: + from launchdarkly_ai_server.evaluations.module import _run_passed + from launchdarkly_ai_server.evaluations.types import RunSummary + + total, passed, failed, error, pending = counts + summary = RunSummary( + total_rows=total, + passed_rows=passed, + failed_rows=failed, + error_rows=error, + pending_rows=pending, + ) + + assert _run_passed(summary, criteria) is expected diff --git a/uv.lock b/uv.lock index d8fba767..865f4f69 100644 --- a/uv.lock +++ b/uv.lock @@ -844,7 +844,7 @@ wheels = [ [[package]] name = "launchdarkly-ai-claude-agents" -version = "0.2.3" +version = "0.2.4" source = { editable = "packages/claude-agents" } dependencies = [ { name = "anthropic" }, @@ -880,7 +880,7 @@ requires-dist = [ [[package]] name = "launchdarkly-ai-langchain-agents" -version = "0.2.3" +version = "0.2.4" source = { editable = "packages/langchain-agents" } dependencies = [ { name = "langchain-core" }, @@ -916,7 +916,7 @@ requires-dist = [ [[package]] name = "launchdarkly-ai-openai-agents" -version = "0.2.3" +version = "0.2.4" source = { editable = "packages/openai-agents" } dependencies = [ { name = "launchdarkly-ai-server" }, @@ -952,7 +952,7 @@ requires-dist = [ [[package]] name = "launchdarkly-ai-python" -version = "0.1.7" +version = "0.1.8" source = { editable = "packages/ai" } dependencies = [ { name = "launchdarkly-ai-server" }, @@ -972,7 +972,7 @@ provides-extras = ["otel"] [[package]] name = "launchdarkly-ai-server" -version = "0.2.3" +version = "0.2.4" source = { editable = "packages/client" } dependencies = [ { name = "opentelemetry-api" },