Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
30 changes: 20 additions & 10 deletions packages/client/src/launchdarkly_ai_server/evaluations/module.py
Original file line number Diff line number Diff line change
Expand Up @@ -63,6 +63,25 @@ def _is_terminal_summary(summary: RunSummary) -> bool:
)


def _run_passed(summary: RunSummary, criteria: list[Criterion]) -> bool:
"""Return whether a run meets its gate.

When every criterion sets ``pass_rate_threshold``, the run passes if no
row is pending and the share of passed rows is at least the highest
threshold. Errored rows count as not passed. Otherwise the run passes
only if no row failed, errored, or is pending.
"""
if summary.pending_rows != 0:
return False
thresholds = [criterion.pass_rate_threshold for criterion in criteria]
if thresholds and all(threshold is not None for threshold in thresholds):
if summary.total_rows <= 0:
return False
required = max(t for t in thresholds if t is not None)
return summary.passed_rows / summary.total_rows >= required
return summary.error_rows == 0 and summary.failed_rows == 0


def _merge_generation(
base: GenerationConfig, override: GenerationConfig | None
) -> GenerationConfig:
Expand Down Expand Up @@ -309,16 +328,7 @@ async def run(
f"{_segment(evaluation.id)}/runs/{_segment(evaluation_run.id)}"
)
return EvalRunResult(
# failed_rows counts rows whose criteria were scored and did not
# meet their threshold, so a gate that ignores it exits 0 on a run
# where every row failed its judge. It was omissible while runs were
# generation-only -- a row either generated or errored, and nothing
# produced a fail -- and stops being so the moment criteria exist.
passed=(
summary.error_rows == 0
and summary.failed_rows == 0
and summary.pending_rows == 0
),
passed=_run_passed(summary, run_criteria),
url=url,
run_id=evaluation_run.id,
summary=summary,
Expand Down
43 changes: 43 additions & 0 deletions packages/client/tests/test_evaluations_run.py
Original file line number Diff line number Diff line change
Expand Up @@ -2965,3 +2965,46 @@ def test_ai_config_variation_from_api_layers_the_model_config() -> None:
unlinked = AIConfigVariation.from_api(latest)
assert "provider" not in unlinked.generation
assert unlinked.generation["parameters"] == {"temperature": 0.7}


def _scorer(**options: Any) -> Scorer:
return Scorer(name="accuracy", fn=lambda row, output: True, **options)


@pytest.mark.parametrize(
("counts", "criteria", "expected"),
[
# 28 of 30 rows passed: 93% meets a 0.9 pass rate.
((30, 28, 1, 1, 0), [_scorer(pass_rate_threshold=0.9)], True),
# The strictest pass rate across criteria applies.
(
(30, 28, 1, 1, 0),
[_scorer(pass_rate_threshold=0.9), _scorer(pass_rate_threshold=0.95)],
False,
),
# Without a pass rate on every criterion, any failed row fails the run.
((30, 28, 1, 1, 0), [_scorer(pass_rate_threshold=0.9), _scorer()], False),
((30, 28, 1, 1, 0), [_scorer()], False),
((30, 30, 0, 0, 0), [_scorer()], True),
# A pending row never passes.
((30, 29, 0, 0, 1), [_scorer(pass_rate_threshold=0.5)], False),
# An empty run never passes a pass rate.
((0, 0, 0, 0, 0), [_scorer(pass_rate_threshold=0.0)], False),
],
)
def test_run_passed_honours_pass_rate_threshold(
counts: tuple[int, int, int, int, int], criteria: list[Scorer], expected: bool
) -> None:
from launchdarkly_ai_server.evaluations.module import _run_passed
from launchdarkly_ai_server.evaluations.types import RunSummary

total, passed, failed, error, pending = counts
summary = RunSummary(
total_rows=total,
passed_rows=passed,
failed_rows=failed,
error_rows=error,
pending_rows=pending,
)

assert _run_passed(summary, criteria) is expected
10 changes: 5 additions & 5 deletions uv.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

Loading