Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
19 changes: 19 additions & 0 deletions packages/client/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -82,6 +82,25 @@ sys.exit(asyncio.run(main()))

Generation and criterion events are the only path by which row results reach LaunchDarkly, so `init_evaluations()` raises rather than creating a run that can never complete unless it can resolve an event transport: either an SDK key (`sdk_key` or `LD_SDK_KEY`) or a client already initialized through `init_client(client=...)`. Bringing your own client lets a process emit evaluation events without an SDK key in scope. Every generated row is emitted and flushed unconditionally; no feature flag gates event publishing. The harness then polls the summary endpoint until row accounting shows processing is complete.

### Supply the dataset inline

Pass `rows` instead of `dataset` for datasets that live in code or are built at run time rather than stored in LaunchDarkly. The two are mutually exclusive, and `run()` raises unless exactly one is given. Each row is a `DatasetRow` or a mapping in the upload wire shape — any of `input`, `expectedOutput`, `variables` and `metadata`, plus an optional `rowIdx` — and its index is its position in the list.

```python
result = await evals.run(
project_key="my-project",
key="support-qa-2026-08-20",
rows=[
{"input": "How do I reset my password?", "expectedOutput": "Use the reset link."},
{"input": "Where is order {{order_id}}?", "variables": {"order_id": "A-17"}},
],
handler=create_openai_messages_handler(),
generation={"provider": "OpenAI", "model": "gpt-4o"},
)
```

The rows are uploaded to the run, in batches of up to 500, before any generation starts. They are uploaded unrendered, and `{{...}}` placeholders render exactly as they do for a stored dataset. A malformed row fails the run before any records are created. Events from an inline run carry no dataset id.

### Score rows with judges and scorers

Pass `criteria` to `run()` to score every generated row. A `Judge` references an AI Judge config that already exists in LaunchDarkly — the SDK creates no judges and ships none of its own — and a `Scorer` wraps a local function, so a run can mix model-graded and deterministic checks. Each criterion runs once per generated row, bounded by the same `concurrency` as generation, and emits one `$ld:ai:offline-evals:criterion` event per `(row, criterion)` carrying the criterion identity, the judge's variation key and version, the validated score, its reason, usage, and timings.
Expand Down
2 changes: 2 additions & 0 deletions packages/client/src/launchdarkly_ai_server/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -33,6 +33,7 @@
EvaluationsError,
EvaluationsModule,
GenerationConfig,
InlineDatasetRow,
Judge,
RunSummary,
Scorer,
Expand Down Expand Up @@ -194,6 +195,7 @@
"EvaluationsError",
"EvaluationsModule",
"GenerationConfig",
"InlineDatasetRow",

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

This is a new root name for a type alias that callers don't need, since rows= accepts a DatasetRow or a plain dict either way. The lifecycle draft moves evals to launchdarkly_ai_server.experimental.evaluations. I'd leave it in launchdarkly_ai_server.evaluations only and keep it out of the root __all__, so the 1.0 surface trim doesn't have to remove it.

"Judge",
"RunSummary",
"Scorer",
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -16,6 +16,7 @@
DatasetRow,
EvalRunResult,
GenerationConfig,
InlineDatasetRow,
RunSummary,
Usage,
)
Expand All @@ -30,6 +31,7 @@
"EvaluationsModule",
"GenerationConfig",
"HttpResponse",
"InlineDatasetRow",
"Judge",
"LDApiClient",
"LDApiError",
Expand Down
19 changes: 12 additions & 7 deletions packages/client/src/launchdarkly_ai_server/evaluations/api.py
Original file line number Diff line number Diff line change
Expand Up @@ -133,7 +133,15 @@ def request(
path: str,
body: Any = None,
params: dict[str, Any] | None = None,
*,
idempotent: bool = False,
) -> Any:
"""Send one request, retrying where a replay cannot duplicate work.

``idempotent`` marks a non-GET request the server applies at most once
per payload, so it may be replayed after a 5xx or transport failure.
"""
retry_safe = idempotent or method.upper() in RETRY_SAFE_METHODS
headers = {
"Authorization": self.api_token,
"Accept": "application/json",
Expand All @@ -152,10 +160,7 @@ def request(
method, self.url_for(path, params), headers, payload, self._timeout
)
except (TimeoutError, urllib.error.URLError) as error:
if (
method.upper() not in RETRY_SAFE_METHODS
or attempt >= self._max_retries
):
if not retry_safe or attempt >= self._max_retries:
raise EvaluationsError(
f"LaunchDarkly API {method} {path} failed after retries: {error}"
) from error
Expand All @@ -165,7 +170,7 @@ def request(
# A 429 is rejected before the server acts on it, so it is safe to
# replay for any method.
retryable = response.status == 429 or (
response.status >= 500 and method.upper() in RETRY_SAFE_METHODS
response.status >= 500 and retry_safe
)
if retryable and attempt < self._max_retries:
self._sleep(self._retry_delay(attempt, response))
Expand All @@ -190,5 +195,5 @@ def request(
def get(self, path: str, params: dict[str, Any] | None = None) -> Any:
return self.request("GET", path, params=params)

def post(self, path: str, body: Any = None) -> Any:
return self.request("POST", path, body=body)
def post(self, path: str, body: Any = None, *, idempotent: bool = False) -> Any:
return self.request("POST", path, body=body, idempotent=idempotent)
Original file line number Diff line number Diff line change
Expand Up @@ -37,14 +37,14 @@ class CriterionEventPayload:
evaluation_id: str
evaluation_run_id: str
run_id: str
dataset_id: str
dataset_id: str | None
row_index: int
criterion_type: str
kind: CriterionEventKind
event_id: str
emitted_at: str
evaluation_key: str
dataset_key: str
dataset_key: str | None
status: CriterionStatus
started_at: str
evaluated_at: str
Expand Down
177 changes: 164 additions & 13 deletions packages/client/src/launchdarkly_ai_server/evaluations/module.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,7 +6,7 @@
import math
import os
import time
from collections.abc import Mapping
from collections.abc import Mapping, Sequence
from typing import Any, cast

from ..lifecycle import get_client, init_client
Expand All @@ -24,14 +24,105 @@
ToolImplementation,
_provides_for,
_segment,
render_row,
)
from .types import (
AIConfig,
DatasetRef,
DatasetRow,
EvalRunResult,
GenerationConfig,
InlineDatasetRow,
RunSummary,
)
from .types import AIConfig, EvalRunResult, GenerationConfig, RunSummary

logger = logging.getLogger(__name__)

DEFAULT_UI_BASE_URI = "https://app.launchdarkly.com"
SUMMARY_POLL_INTERVAL_SECONDS = 2.0
SUMMARY_POLL_TIMEOUT_SECONDS = 180.0
INLINE_ROW_FIELDS = ("rowIdx", "input", "expectedOutput", "variables", "metadata")


def _render_inline_row(row: DatasetRow) -> DatasetRow:
return render_row(
row.row_index,
input_value=row.input,
expected_value=row.expected_output,
variables_value=row.variables,
metadata_value=row.metadata,
)


def _normalize_inline_rows(rows: Sequence[InlineDatasetRow]) -> list[DatasetRow]:
"""Validate caller-supplied rows, returning them raw and indexed by position.

Pure, so a malformed row fails before any records are created. The values
stay unrendered: they are uploaded as stored rows, which the server renders
the same way it renders a hosted dataset's.
"""
if not rows:
raise EvaluationsError("Inline dataset is empty")
normalized: list[DatasetRow] = []
for position, row in enumerate(rows):
if isinstance(row, DatasetRow):
if row.row_index != position:
raise EvaluationsError(
f"Inline dataset row {position} has row_index {row.row_index}; "
"an inline row's index is its position in the list"
)
values: Mapping[str, Any] = {
"input": row.input,
"expectedOutput": row.expected_output,
"variables": row.variables,
"metadata": row.metadata,
}
elif isinstance(row, Mapping):
unknown = sorted(str(key) for key in row if key not in INLINE_ROW_FIELDS)
if unknown:
raise EvaluationsError(
f"Inline dataset row {position} has unknown fields: "
+ ", ".join(repr(key) for key in unknown)
+ ". Expected any of: "
+ ", ".join(repr(key) for key in INLINE_ROW_FIELDS)
)
row_idx = row.get("rowIdx")
if row_idx is not None and (
isinstance(row_idx, bool) or row_idx != position
):
raise EvaluationsError(
f"Inline dataset row {position} has rowIdx {row_idx!r}; "
"an inline row's index is its position in the list"
)
values = row
else:
raise EvaluationsError(
f"Inline dataset row {position} must be a DatasetRow or a mapping"
)
for field_name in ("input", "expectedOutput"):
value = values.get(field_name)
if value is not None and not isinstance(value, str):
raise EvaluationsError(
f"Inline dataset row {position} {field_name} must be a string"
)
for field_name in ("variables", "metadata"):

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

This checks that variables / metadata are mappings, but not what's inside them. Reproduced at 196d0b1: variables={"when": datetime(2026, 1, 1)} raises a bare TypeError from json.dumps in api.py:154, after POST evaluations and POST runs. metadata={"score": math.nan} is sent as NaN, which isn't valid JSON. A strict json.dumps(..., allow_nan=False) per row here would keep the README's "fails before any records are created" promise true. §8.4 asks for the same check on inline tool schemas.

value = values.get(field_name)
if value is not None and not isinstance(value, Mapping):
raise EvaluationsError(
f"Inline dataset row {position} {field_name} must be a mapping"
)
variables = values.get("variables")
metadata = values.get("metadata")
normalized.append(
DatasetRow(
row_index=position,
input=values.get("input"),
expected_output=values.get("expectedOutput"),
variables=dict(variables) if variables else {},
metadata=dict(metadata) if metadata is not None else None,
)
)
return normalized


def _env(name: str) -> str | None:
Expand Down Expand Up @@ -125,7 +216,8 @@ async def run(
*,
project_key: str,
key: str,
dataset: str,
dataset: str | None = None,
rows: Sequence[InlineDatasetRow] | None = None,

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I'd like to have the rows of the dataset be passed in via the dataset field. If it makes sense we could also move the fetching of a dataset over to eval.dataset.get(...).

handler: EvalHandler,
generation: GenerationConfig | None = None,
ai_config: AIConfig | None = None,
Expand All @@ -145,6 +237,14 @@ async def run(
generated row, and one evaluation event is emitted per
``(row, criterion)`` result.

Pass exactly one of ``dataset``, the key of a dataset stored in
LaunchDarkly, or ``rows``, an inline dataset. Inline rows are
:class:`DatasetRow` values or mappings in the dataset-rows wire shape
(``input``, ``expectedOutput``, ``variables``, ``metadata``, optional
``rowIdx``); each row's index is its position in the list. They are
uploaded to the run before any generation starts, and templates in
them render exactly as a stored dataset's do.

A :class:`Judge` is an independent AI Config and may be served by a
different provider or mode than ``generation``. ``handler`` runs a judge
only when it provides for that judge's provider; pass handlers for any
Expand Down Expand Up @@ -173,12 +273,12 @@ async def run(
self._validate_run_args(
project_key=project_key,
key=key,
dataset=dataset,
handler=handler,
concurrency=concurrency,
poll_interval_seconds=poll_interval_seconds,
poll_timeout_seconds=poll_timeout_seconds,
)
inline_rows = self._validate_dataset_source(dataset=dataset, rows=rows)
self._validate_config_source(generation=generation, ai_config=ai_config)
pinned_tool_versions: dict[str, int] = {}
config_label = ""
Expand Down Expand Up @@ -238,12 +338,16 @@ async def run(
resolved_judges = await self._runner._resolve_judges(
project_key, ld_judges, handler, run_judge_handlers
)
dataset_ref = await asyncio.to_thread(
self._runner._fetch_dataset, project_key, dataset
)
rows = await asyncio.to_thread(
self._runner._get_dataset_rows, project_key, dataset
)
if dataset is not None:
dataset_ref = await asyncio.to_thread(
self._runner._fetch_dataset, project_key, dataset
)
dataset_rows = await asyncio.to_thread(
self._runner._get_dataset_rows, project_key, dataset
)
else:
dataset_ref = DatasetRef(id=None, key=None)
dataset_rows = [_render_inline_row(row) for row in inline_rows]
evaluation = await asyncio.to_thread(
self._runner._create_evaluation,
project_key,
Expand All @@ -258,9 +362,20 @@ async def run(
evaluation.id,
dataset_ref.id,
)
if dataset is None:
# Must finish before any event is tracked: the run starts with a
# placeholder row count of 1, so a result counted before the rows
# land would mark the run complete.
await asyncio.to_thread(
self._runner._upload_dataset_rows,

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

If any batch fails here, the run already exists and has some rows. Reproduced with 600 rows and a 400 on the second batch: 500 rows stored, EvaluationsError raised, and nothing ever completes the run. Should this mark the run failed before re-raising, or does the server expire runs that never get events? Either way the spec should say which.

project_key,
evaluation.id,
evaluation_run.id,
inline_rows,
)
config = self._runner._build_handler_config(generation, resolved_tools)
results = await self._runner._run_rows(
rows,
dataset_rows,
handler,
config,
run_tools,
Expand Down Expand Up @@ -431,7 +546,6 @@ def _validate_run_args(
*,
project_key: str,
key: str,
dataset: str,
handler: EvalHandler,
concurrency: int,
poll_interval_seconds: float,
Expand All @@ -440,7 +554,6 @@ def _validate_run_args(
for name, value in (
("project_key", project_key),
("key", key),
("dataset", dataset),
):
if not value.strip():
raise EvaluationsError(f"{name} must not be blank")
Expand All @@ -458,6 +571,44 @@ def _validate_run_args(
if seconds < 0:
raise EvaluationsError(f"{name} must not be negative")

@staticmethod
def _validate_dataset_source(
*,
dataset: str | None,
rows: Sequence[InlineDatasetRow] | None,
) -> list[DatasetRow]:
"""Require exactly one dataset source, returning any inline rows raw.

Types are checked at runtime as well: a ``str`` is itself a
``Sequence``, so a key passed as ``rows`` would otherwise be read as
one row per character. The list is empty for a hosted dataset; an
inline one is never empty.
"""
if dataset is not None and rows is not None:
raise EvaluationsError(
"dataset and rows are mutually exclusive: pass dataset for a "
"LaunchDarkly dataset key, or rows for an inline dataset"
)
if dataset is not None:
if not isinstance(dataset, str):
raise EvaluationsError(
"dataset must be a LaunchDarkly dataset key; pass inline "
"rows with rows="
)
if not dataset.strip():
raise EvaluationsError("dataset must not be blank")
return []
if rows is None:
raise EvaluationsError(
"Pass dataset, a LaunchDarkly dataset key, or rows, an inline dataset"
)
if isinstance(rows, str) or not isinstance(rows, Sequence):
raise EvaluationsError(
"rows must be a sequence of rows; pass a LaunchDarkly dataset "
"key with dataset="
)
return _normalize_inline_rows(rows)

@staticmethod
def _validate_config_source(
*,
Expand Down
Loading
Loading