Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 6 additions & 1 deletion engine/src/agent_control_engine/core.py
Original file line number Diff line number Diff line change
Expand Up @@ -320,7 +320,12 @@ async def _evaluate_leaf(
timeout = DEFAULT_EVALUATOR_TIMEOUT

result = await asyncio.wait_for(
evaluator.evaluate_with_context(data, request.step),
evaluator.evaluate_with_request_context(
data,
request.step,
target_type=request.target_type,
target_id=request.target_id,
),
timeout=timeout,
)
except TimeoutError:
Expand Down
19 changes: 14 additions & 5 deletions engine/tests/test_core.py
Original file line number Diff line number Diff line change
Expand Up @@ -39,7 +39,7 @@ class SimpleConfig(BaseModel):
# Shared state for coordination between test evaluators
_execution_log: list[str] = []
_blocker_event: asyncio.Event | None = None
_context_calls: list[tuple[Any, Step]] = []
_context_calls: list[tuple[Any, Step, str | None, str | None]] = []


def reset_test_state() -> None:
Expand Down Expand Up @@ -178,8 +178,15 @@ class ContextEvaluator(Evaluator[SimpleConfig]):
async def evaluate(self, data: Any) -> EvaluatorResult:
raise AssertionError("engine should call evaluate_with_context")

async def evaluate_with_context(self, data: Any, step: Step) -> EvaluatorResult:
_context_calls.append((data, step))
async def evaluate_with_request_context(
self,
data: Any,
step: Step,
*,
target_type: str | None = None,
target_id: str | None = None,
) -> EvaluatorResult:
_context_calls.append((data, step, target_type, target_id))
return EvaluatorResult(matched=False, confidence=1.0, message="context received")


Expand Down Expand Up @@ -296,11 +303,13 @@ async def test_context_evaluator_receives_selected_data_and_complete_step() -> N
agent_name="00000000-0000-0000-0000-000000000001",
step=step,
stage="pre",
target_type="log_stream",
target_id="run-1",
)
)

# Then: selector behavior is unchanged and the complete Step is separate
assert _context_calls == [("answer", step)]
assert _context_calls == [("answer", step, "log_stream", "run-1")]


@pytest.mark.asyncio
Expand Down Expand Up @@ -331,7 +340,7 @@ async def test_cached_context_evaluator_handles_concurrent_steps_without_retaini
)

# Then: each selected value remains paired with its own full Step
assert {(data, step.ground_truth) for data, step in _context_calls} == {
assert {(data, step.ground_truth) for data, step, _, _ in _context_calls} == {
("first", "one"),
("second", "two"),
}
Expand Down
25 changes: 25 additions & 0 deletions evaluators/builtin/src/agent_control_evaluators/_base.py
Original file line number Diff line number Diff line change
Expand Up @@ -178,6 +178,31 @@ async def evaluate_with_context(self, data: Any, step: Step) -> EvaluatorResult:
"""
return await self.evaluate(data)

async def evaluate_with_request_context(
self,
data: Any,
step: Step,
*,
target_type: str | None = None,
target_id: str | None = None,
) -> EvaluatorResult:
"""Evaluate selected data with step and opaque request target metadata.

The default implementation delegates to :meth:`evaluate_with_context`
so existing evaluators that override that hook keep working unchanged.

Args:
data: Data extracted by the configured selector.
step: Complete runtime step for the current request.
target_type: Optional target kind attached to the evaluation request.
target_id: Optional opaque target ID attached to the evaluation request.

Returns:
EvaluatorResult produced by this evaluator.
"""
del target_type, target_id
return await self.evaluate_with_context(data, step)

def get_timeout_seconds(self) -> float:
"""Get timeout in seconds from config or metadata default."""
timeout_ms: int = getattr(self.config, "timeout_ms", self.metadata.timeout_ms)
Expand Down
36 changes: 33 additions & 3 deletions evaluators/builtin/tests/test_base.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,7 +6,6 @@
from typing import Any

import pytest

from agent_control_evaluators import Evaluator, EvaluatorConfig, EvaluatorMetadata
from agent_control_models import EvaluatorResult, Step

Expand Down Expand Up @@ -40,6 +39,15 @@ async def evaluate(self, data: Any) -> EvaluatorResult:
)


class LegacyContextOverrideEvaluator(MockEvaluator):
"""Models an installed evaluator overriding the pre-existing context hook."""

async def evaluate_with_context(self, data: Any, step: Step) -> EvaluatorResult:
result = await self.evaluate(data)
result.metadata["step_name"] = step.name
return result


class TestEvaluatorMetadata:
"""Tests for EvaluatorMetadata dataclass."""

Expand Down Expand Up @@ -114,13 +122,35 @@ async def test_contextual_evaluation_delegates_to_existing_evaluate(self):
evaluator = MockEvaluator.from_dict({"should_match": True})
step = Step(type="llm", name="answer", input="full input")

# When: the engine-facing contextual hook is called
result = await evaluator.evaluate_with_context("selected data", step)
# When: the engine-facing hook includes opaque request target metadata
result = await evaluator.evaluate_with_request_context(
"selected data",
step,
target_type="log_stream",
target_id="run-1",
)

# Then: the legacy evaluate implementation handles the selected data
assert result.matched is True
assert result.metadata == {"data": "selected data"}

@pytest.mark.asyncio
async def test_request_context_preserves_existing_context_overrides(self):
"""The new request hook delegates to older evaluate_with_context overrides."""
evaluator = LegacyContextOverrideEvaluator.from_dict({"should_match": True})
step = Step(type="llm", name="answer", input="full input")

result = await evaluator.evaluate_with_request_context(
"selected data",
step,
target_type="log_stream",
target_id="run-1",
)

assert result.matched is True
assert result.metadata["data"] == "selected data"
assert result.metadata["step_name"] == "answer"

def test_evaluator_config_stored(self):
"""Test that evaluator stores config."""
evaluator = MockEvaluator.from_dict({"should_match": True})
Expand Down
15 changes: 11 additions & 4 deletions evaluators/contrib/galileo/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -9,10 +9,17 @@ The `galileo.luna2` evaluator ID has been removed. Existing controls that use
configuration to use the direct Luna scorer fields. `scorer_id` is required;
`scorer_label` and `scorer_version_id` are optional. `scorer_version_id` is a
deprecated optional compatibility identifier; Orbit currently invokes the
scorer's current default version. The evaluator calls the
URL configured by `GALILEO_LUNA_INVOKE_URL`; the target must support the Luna
scorer invoke request/response contract and internal Galileo secret auth. Also
set `threshold` and `operator` as needed. If you still need the legacy Luna2
scorer's current default version. When `GALILEO_API_KEY` and `GALILEO_API_URL`
are both configured, an invoked evaluator exchanges the application key for
one short-lived Orbit scorer grant scoped to the `target_id` on the Agent
Control evaluation request. Configure `agent_control.init()` with
`target_type="log_stream"` and that run's `target_id`; Orbit receives the ID as
`run_id`. The grant is reused across scorers for that run and sent to the Luna
invoke URL configured by `GALILEO_LUNA_INVOKE_URL`. The application key is
sent only to Orbit. If only
`GALILEO_API_SECRET_KEY` or `GALILEO_API_SECRET` is configured, the legacy
internal JWT flow remains active and does not request a scorer grant. Also set
`threshold` and `operator` as needed. If you still need the legacy Luna2
evaluator, pin
`agent-control-evaluator-galileo <8`.

Expand Down
Loading
Loading