diff --git a/integrations/google-adk/README.md b/integrations/google-adk/README.md index 692c0bb..a7c4c66 100644 --- a/integrations/google-adk/README.md +++ b/integrations/google-adk/README.md @@ -60,6 +60,7 @@ record = plugin.build_record( invocation_id, subject="spiffe://example.org/agent/research-bot", policy_bundle=open("policy.cedar", "rb").read(), + enforcement_mode="declared", workload_digest="sha256:...", data_class="internal", model_provider="google", @@ -68,9 +69,11 @@ signed = sign_record(record, generate_key()) plugin.discard(invocation_id) ``` -`enforcement_mode` defaults to `declared`: the policy is bound into the signed -record, but Google ADK itself did not evaluate it. Override that value only when -a separate enforcement layer actually evaluated the policy. +`enforcement_mode` is required. A bare ADK run is `declared`: the policy is bound +into the signed record, but Google ADK itself did not evaluate it. TRACE spec +section 4.3 says `declared` MUST NOT be a default, so the caller states it. Pass +another value only when a separate enforcement layer actually evaluated the +policy. One plugin can observe concurrent invocations. It retains state by ADK invocation id until `discard()` is called, so long-running processes should @@ -89,6 +92,6 @@ python -m pytest test_google_adk_interop.py -q The first suite exercises evidence construction without installing ADK. The second uses the released runner and checks success, tool failure, cancellation, concurrent invocations, payload exclusion, signed TRACE validation, and Level 0 -conformance for both the default `declared` policy mode and the optional +conformance for both the `declared` policy mode and the optional externally enforced `advisory` path. `declared` needs `agentrust-trace-tests` 0.5.1 or later; 0.5.0 rejects it with `TR-POL-002`. diff --git a/integrations/google-adk/google_adk_to_trace.py b/integrations/google-adk/google_adk_to_trace.py index 5f838fa..6897d3e 100644 --- a/integrations/google-adk/google_adk_to_trace.py +++ b/integrations/google-adk/google_adk_to_trace.py @@ -7,9 +7,10 @@ and responses, tool arguments and results, and exception messages are deliberately never read into evidence. -Google ADK does not evaluate a TRACE policy. Records therefore default to -``policy.enforcement_mode: declared`` and ``appraisal.status: none``. They have -no ``origin`` block because this is first-party, in-process observation. +Google ADK does not evaluate a TRACE policy, so ``enforcement_mode`` has no +default: a bare ADK run passes ``declared`` (TRACE spec section 4.3 says +``declared`` MUST NOT be a default). Records carry ``appraisal.status: none`` +and no ``origin`` block because this is first-party, in-process observation. """ from __future__ import annotations @@ -352,7 +353,7 @@ def build_record( *, subject: str, policy_bundle: bytes, - enforcement_mode: str = "declared", + enforcement_mode: str, workload_digest: str, data_class: str, model_provider: str | None = None, @@ -409,7 +410,7 @@ def build_record( *, subject: str, policy_bundle: bytes, - enforcement_mode: str = "declared", + enforcement_mode: str, workload_digest: str, data_class: str, model_provider: str | None, diff --git a/integrations/google-adk/test_google_adk_interop.py b/integrations/google-adk/test_google_adk_interop.py index a29b62f..0a1a6eb 100644 --- a/integrations/google-adk/test_google_adk_interop.py +++ b/integrations/google-adk/test_google_adk_interop.py @@ -136,6 +136,7 @@ def build_signed(plugin: GoogleAdkTracePlugin, invocation_id: str) -> dict: invocation_id, subject="spiffe://example.org/agent/google-adk", policy_bundle=b'{"rules":["no-payload-egress"]}', + enforcement_mode="declared", workload_digest=DIGEST, data_class="confidential", model_provider="test-provider", diff --git a/integrations/google-adk/test_google_adk_to_trace.py b/integrations/google-adk/test_google_adk_to_trace.py index f7d410b..ca22980 100644 --- a/integrations/google-adk/test_google_adk_to_trace.py +++ b/integrations/google-adk/test_google_adk_to_trace.py @@ -39,6 +39,7 @@ def kwargs(**overrides): values = { "subject": SUBJECT, "policy_bundle": b'{"rules":["no-payload-egress"]}', + "enforcement_mode": "declared", "workload_digest": DIGEST, "data_class": "confidential", "model_provider": "test-provider", @@ -396,6 +397,23 @@ def test_invalid_workload_digest_is_refused() -> None: ) +def test_enforcement_mode_has_no_default() -> None: + """TRACE spec section 4.3: declared MUST NOT be a default. The caller states it.""" + values = kwargs(model_id="test-model") + del values["enforcement_mode"] + with pytest.raises(TypeError, match="enforcement_mode"): + build_record(**values, transcript=b"{}", tool_count=0) + + +def test_declared_is_accepted_when_the_caller_states_it() -> None: + record = build_record( + **kwargs(model_id="test-model", enforcement_mode="declared"), + transcript=b"{}", + tool_count=0, + ) + assert record["policy"]["enforcement_mode"] == "declared" + + def test_invalid_enforcement_mode_is_refused() -> None: with pytest.raises(MissingEvidence, match="enforcement_mode"): build_record( diff --git a/integrations/langchain/README.md b/integrations/langchain/README.md index b4b0bad..adc1e9f 100644 --- a/integrations/langchain/README.md +++ b/integrations/langchain/README.md @@ -50,22 +50,22 @@ agent.invoke({"input": "..."}, config={"callbacks": [handler]}) record = handler.build_record( subject="spiffe://example.org/agent/research-bot", policy_bundle=open("policy.cedar", "rb").read(), - # enforcement_mode defaults to "declared"; see below + enforcement_mode="declared", # required; see below workload_digest="sha256:...", data_class="internal", ) signed = sign_record(record, generate_key()) ``` -## `enforcement_mode` defaults to `declared` +## `enforcement_mode` is required -**LangChain enforces no policy.** It has no policy engine, and this handler is an observer that cannot block anything. So the default is `declared`: the policy is named and bound into the signed record, and nothing evaluated it. +**LangChain enforces no policy.** It has no policy engine, and this handler is an observer that cannot block anything. So a bare LangChain run is `declared`: the policy is named and bound into the signed record, and nothing evaluated it. `declared` needs `agentrust-trace>=0.9`. -That value did not exist when this adapter was written. `enforce`, `advisory` and `silent` all presuppose that *something evaluated the policy*, so the adapter refused to default the field and made the caller pick a value that overstated their run. TRACE 0.9.0 added `declared` for exactly this case, and it needs `agentrust-trace>=0.9`. +The adapter does not pick the mode for you. `enforce`, `advisory` and `silent` all presuppose that *something evaluated the policy*, so defaulting to any of them claims an evaluation nobody observed, and TRACE spec section 4.3 says `declared` MUST NOT be a default. Leaving the argument out raises `TypeError`. | Value | When it is true here | |---|---| -| `declared` | The default. The policy is named and bound; nothing evaluated it. | +| `declared` | A bare LangChain run. The policy is named and bound; nothing evaluated it. | | `enforce` | Only when a real enforcement layer (cMCP, a policy proxy) sat in front of the tools. | | `advisory` | Only when something evaluated the policy and chose not to act on it. | | `silent` | Only when something enforced it with the operational logs suppressed. | diff --git a/integrations/langchain/langchain_to_trace.py b/integrations/langchain/langchain_to_trace.py index 79ac7a2..fec073b 100644 --- a/integrations/langchain/langchain_to_trace.py +++ b/integrations/langchain/langchain_to_trace.py @@ -15,15 +15,13 @@ **The honest limit, stated where you cannot miss it.** LangChain enforces no policy. It has no policy engine, and this handler is an observer with no ability -to block anything. ``enforcement_mode`` therefore defaults to ``declared``: the -policy is named and bound into the signed record, and nothing evaluated it. - -That value did not exist when this adapter was written. The three original modes -all assert that something evaluated the policy, so the adapter refused to default -the field and made the caller choose a value that overstated their run. TRACE -0.9.0 added ``declared`` for exactly this case. Override the default only when a -real enforcement layer sat in front of the tools; a record claiming ``enforce`` -from a bare LangChain run describes enforcement that did not happen. +to block anything. So ``enforcement_mode`` has no default and the caller states +it. For a bare LangChain run that is ``declared``: the policy is named and bound +into the signed record, and nothing evaluated it. TRACE spec section 4.3 says +``declared`` MUST NOT be a default, and any other default would claim an +evaluation nobody observed. Pass ``enforce`` only when a real enforcement layer +sat in front of the tools; a record claiming ``enforce`` from a bare LangChain +run describes enforcement that did not happen. Callback signatures are taken from ``langchain_core.callbacks.base`` and are the public, documented API. Payloads never enter the record: ``on_tool_start`` @@ -215,7 +213,7 @@ def build_record( *, subject: str, policy_bundle: bytes, - enforcement_mode: str = "declared", + enforcement_mode: str, workload_digest: str, data_class: str, model_provider: str | None = None, @@ -225,9 +223,9 @@ def build_record( ) -> dict[str, Any]: """Assemble the unsigned Trust Record for this run. - ``enforcement_mode`` defaults to ``declared`` because LangChain itself - evaluates no policy. Override it only when a separate enforcement layer - actually evaluated the declared bundle. See the README. + ``enforcement_mode`` is required. LangChain itself evaluates no policy, + so a bare run passes ``declared``; pass another mode only when a + separate enforcement layer actually evaluated the bundle. See the README. ``attestation`` is ``{"platform": ..., "measurement": ...}`` when the deployment runs in a TEE, which lifts the record to Level 1. Absent, the @@ -257,7 +255,7 @@ def build_record( *, subject: str, policy_bundle: bytes, - enforcement_mode: str = "declared", + enforcement_mode: str, workload_digest: str, data_class: str, model_provider: str | None, @@ -284,9 +282,9 @@ def build_record( ) if enforcement_mode not in ENFORCEMENT_MODES: raise MissingEvidence( - f"enforcement_mode must be one of {', '.join(ENFORCEMENT_MODES)}. The default " - "is 'declared', which is what a LangChain run actually is: the policy is named " - "and bound, and nothing evaluated it." + f"enforcement_mode must be one of {', '.join(ENFORCEMENT_MODES)}. A bare " + "LangChain run is 'declared': the policy is named and bound, and nothing " + "evaluated it." ) if not model_provider or not model_id: raise MissingEvidence( diff --git a/integrations/langchain/test_langchain_to_trace.py b/integrations/langchain/test_langchain_to_trace.py index 8516934..0c8b5dc 100644 --- a/integrations/langchain/test_langchain_to_trace.py +++ b/integrations/langchain/test_langchain_to_trace.py @@ -119,16 +119,17 @@ def test_transcript_is_order_sensitive() -> None: # --- refusals -------------------------------------------------------------- -def test_enforcement_mode_defaults_to_declared() -> None: - """TRACE 0.9.0 added the value that is actually true of a framework run. - - Before it, the three modes all asserted that something evaluated the policy, - so this adapter refused to default the field and made the caller pick a value - that overstated their run. - """ +def test_enforcement_mode_has_no_default() -> None: + """TRACE spec section 4.3: declared MUST NOT be a default, and any other + default claims an evaluation nobody observed. The caller states the mode.""" kwargs = _kwargs() del kwargs["enforcement_mode"] - record = _handler_with_two_tools().build_record(**kwargs) + with pytest.raises(TypeError, match="enforcement_mode"): + _handler_with_two_tools().build_record(**kwargs) + + +def test_declared_is_accepted_when_the_caller_states_it() -> None: + record = _handler_with_two_tools().build_record(**_kwargs(enforcement_mode="declared")) assert record["policy"]["enforcement_mode"] == "declared" diff --git a/integrations/langchain/test_langgraph_interop.py b/integrations/langchain/test_langgraph_interop.py index c60e288..ea791bf 100644 --- a/integrations/langchain/test_langgraph_interop.py +++ b/integrations/langchain/test_langgraph_interop.py @@ -59,6 +59,7 @@ def test_langgraph_tool_run_emits_a_valid_trace_record() -> None: handler.build_record( subject="spiffe://example.org/agent/langgraph", policy_bundle=b'{"rules":["no-payload-egress"]}', + enforcement_mode="declared", workload_digest=DIGEST, data_class="confidential", model_provider="test-provider", diff --git a/integrations/llamaindex/README.md b/integrations/llamaindex/README.md index b904075..36f3dbb 100644 --- a/integrations/llamaindex/README.md +++ b/integrations/llamaindex/README.md @@ -50,6 +50,7 @@ async def run_with_record( unsigned = tracker.build_record( subject=subject, policy_bundle=policy_bundle, # bytes of the declared policy + enforcement_mode="declared", # required; nothing evaluated the policy workload_digest=workload_digest, # digest of your artifact model_provider=model_provider, model_id=model_id, @@ -133,9 +134,11 @@ retain `appraisal.status: none`. Signing binds the record to its signing key; it does not attest the observer, authenticate model-supplied tool identity, prove safe behavior, or establish hardware provenance or runtime integrity. -`policy.enforcement_mode` defaults to `declared`: the caller's policy is named -and hashed, but LlamaIndex has not evaluated or enforced it. Supplying another -mode requires an actual external policy layer. Supplied attestation fields are +`policy.enforcement_mode` is required. A bare LlamaIndex run is `declared`: the +caller's policy is named and hashed, but LlamaIndex has not evaluated or +enforced it. TRACE spec section 4.3 says `declared` MUST NOT be a default, so the +caller states it. Supplying another mode requires an actual external policy +layer. Supplied attestation fields are passed through by the existing record builder; this adapter does not verify them or independently establish Level 1 assurance. diff --git a/integrations/llamaindex/llamaindex_to_trace.py b/integrations/llamaindex/llamaindex_to_trace.py index e710a33..e2ce339 100644 --- a/integrations/llamaindex/llamaindex_to_trace.py +++ b/integrations/llamaindex/llamaindex_to_trace.py @@ -209,7 +209,7 @@ def build_record( *, subject: str, policy_bundle: bytes, - enforcement_mode: str = "declared", + enforcement_mode: str, workload_digest: str, data_class: str, model_provider: str | None = None, @@ -244,7 +244,7 @@ def build_record( *, subject: str, policy_bundle: bytes, - enforcement_mode: str = "declared", + enforcement_mode: str, workload_digest: str, data_class: str, model_provider: str | None, @@ -256,10 +256,11 @@ def build_record( ) -> dict[str, Any]: """Assemble the unsigned record. Raises rather than inventing a field. - ``enforcement_mode`` defaults to ``declared``: the policy is named and bound - into the signed record and nothing evaluated it, which is what a LlamaIndex - run is. TRACE 0.9.0 added that value; before it, every available value - overstated a bare run and this adapter refused to default the field. + ``enforcement_mode`` is required. A bare LlamaIndex run is ``declared``: the + policy is named and bound into the signed record and nothing evaluated it. + TRACE spec section 4.3 says ``declared`` MUST NOT be a default, and any + other default would claim an evaluation nobody observed, so the caller + states the mode. """ if not _SUBJECT_RE.match(subject or ""): raise MissingEvidence( @@ -274,9 +275,9 @@ def build_record( ) if enforcement_mode not in ENFORCEMENT_MODES: raise MissingEvidence( - f"enforcement_mode must be one of {', '.join(ENFORCEMENT_MODES)}. The default " - "is 'declared', which is what a LlamaIndex run actually is: the policy is named " - "and bound, and nothing evaluated it." + f"enforcement_mode must be one of {', '.join(ENFORCEMENT_MODES)}. A bare " + "LlamaIndex run is 'declared': the policy is named and bound, and nothing " + "evaluated it." ) if not model_provider or not model_id: raise MissingEvidence( diff --git a/integrations/llamaindex/test_llamaindex_interop.py b/integrations/llamaindex/test_llamaindex_interop.py index 4a01f92..142d56e 100644 --- a/integrations/llamaindex/test_llamaindex_interop.py +++ b/integrations/llamaindex/test_llamaindex_interop.py @@ -130,6 +130,7 @@ def signed_record(tracker, *, subject=SUBJECT): tracker.build_record( subject=subject, policy_bundle=b'{"declared-policy":true}', + enforcement_mode="declared", workload_digest="sha256:" + "a" * 64, data_class="internal", model_provider="local-test-double", diff --git a/integrations/llamaindex/test_llamaindex_to_trace.py b/integrations/llamaindex/test_llamaindex_to_trace.py index d2a22ea..613fea8 100644 --- a/integrations/llamaindex/test_llamaindex_to_trace.py +++ b/integrations/llamaindex/test_llamaindex_to_trace.py @@ -167,11 +167,17 @@ def test_transcript_is_order_sensitive() -> None: # --- refusals -------------------------------------------------------------- -def test_enforcement_mode_defaults_to_declared() -> None: - """TRACE 0.9.0 added the value that is actually true of a framework run.""" +def test_enforcement_mode_has_no_default() -> None: + """TRACE spec section 4.3: declared MUST NOT be a default. The caller states it.""" kwargs = _kwargs() del kwargs["enforcement_mode"] - assert _handler().build_record(**kwargs)["policy"]["enforcement_mode"] == "declared" + with pytest.raises(TypeError, match="enforcement_mode"): + _handler().build_record(**kwargs) + + +def test_declared_is_accepted_when_the_caller_states_it() -> None: + record = _handler().build_record(**_kwargs(enforcement_mode="declared")) + assert record["policy"]["enforcement_mode"] == "declared" def test_unknown_enforcement_mode_is_still_refused() -> None: diff --git a/integrations/openai-agents/README.md b/integrations/openai-agents/README.md index e4d7aa6..1098b54 100644 --- a/integrations/openai-agents/README.md +++ b/integrations/openai-agents/README.md @@ -19,14 +19,15 @@ would produce a worse description, not a safer one. Where the deployment runs inside a TEE, passing an attestation lifts the same record from Level 0 to Level 1 and nothing else about the call changes. -## `enforcement_mode` defaults to `declared` +## `enforcement_mode` is required The Agents SDK enforces no policy. Guardrails exist and can stop a run, but they -are the operator's own code, not a policy engine evaluating a bundle. So the -default is `declared`: the policy is named and bound into the signed record, and -nothing evaluated it. +are the operator's own code, not a policy engine evaluating a bundle. So a bare +run is `declared`: the policy is named and bound into the signed record, and +nothing evaluated it. TRACE spec section 4.3 says `declared` MUST NOT be a +default, so the caller states it; leaving it out raises `TypeError`. -Override it only when a real enforcement layer sat in front of the tools. A +Pass another value only when a real enforcement layer sat in front of the tools. A record claiming `enforce` from a bare Agents SDK run describes enforcement that did not happen. @@ -79,6 +80,7 @@ Runner.run_sync(agent, "...") record = processor.build_record( subject="spiffe://example.org/agent/support-bot", policy_bundle=open("policy.cedar", "rb").read(), + enforcement_mode="declared", workload_digest="sha256:...", data_class="internal", model_provider="openai", # the span that reports it is not read; see above diff --git a/integrations/openai-agents/openai_agents_to_trace.py b/integrations/openai-agents/openai_agents_to_trace.py index 3bff5cb..cf92ba8 100644 --- a/integrations/openai-agents/openai_agents_to_trace.py +++ b/integrations/openai-agents/openai_agents_to_trace.py @@ -17,10 +17,12 @@ **The honest limit, stated where you cannot miss it.** The Agents SDK enforces no policy of its own. Guardrails exist and can stop a run, but they are the operator's code, not a policy engine evaluating a bundle. So -``enforcement_mode`` defaults to ``declared``: the policy is named and bound -into the signed record, and nothing evaluated it. Override that only when a -real enforcement layer sat in front of the tools. A record claiming ``enforce`` -from a bare Agents SDK run describes enforcement that did not happen. +``enforcement_mode`` has no default and the caller states it. A bare Agents SDK +run is ``declared``: the policy is named and bound into the signed record, and +nothing evaluated it. TRACE spec section 4.3 says ``declared`` MUST NOT be a +default. Pass ``enforce`` only when a real enforcement layer sat in front of +the tools; a record claiming ``enforce`` from a bare Agents SDK run describes +enforcement that did not happen. **Payloads never enter the record.** ``FunctionSpanData`` carries ``input`` and ``output``, and ``GenerationSpanData`` carries the full message list. None of it @@ -46,6 +48,7 @@ record = processor.build_record( subject="spiffe://example.org/agent/support-bot", policy_bundle=open("policy.cedar", "rb").read(), + enforcement_mode="declared", workload_digest="sha256:...", data_class="internal", model_provider="openai", @@ -271,7 +274,7 @@ def build_record( tools: tuple[ToolCall, ...] = (), handoffs: tuple[tuple[str, str], ...] = (), agents: tuple[str, ...] = (), - enforcement_mode: str = "declared", + enforcement_mode: str, attestation: dict[str, str] | None = None, iat: int | None = None, ) -> dict[str, Any]: @@ -289,9 +292,9 @@ def build_record( ) if enforcement_mode not in ENFORCEMENT_MODES: raise MissingEvidence( - f"enforcement_mode must be one of {', '.join(ENFORCEMENT_MODES)}. The " - "default is 'declared', which is what a bare Agents SDK run is: the policy " - "is named and bound, and nothing evaluated it." + f"enforcement_mode must be one of {', '.join(ENFORCEMENT_MODES)}. A bare " + "Agents SDK run is 'declared': the policy is named and bound, and nothing " + "evaluated it." ) if not model_provider or not model_id: raise MissingEvidence( diff --git a/integrations/openai-agents/test_openai_agents_interop.py b/integrations/openai-agents/test_openai_agents_interop.py index 1885c8b..64b9c4e 100644 --- a/integrations/openai-agents/test_openai_agents_interop.py +++ b/integrations/openai-agents/test_openai_agents_interop.py @@ -105,6 +105,7 @@ def test_a_real_run_produces_a_valid_record(processor) -> None: record = processor.build_record( subject=SUBJECT, policy_bundle=POLICY, + enforcement_mode="declared", workload_digest=WORKLOAD, data_class="internal", model_provider="openai", @@ -132,6 +133,7 @@ def test_a_real_tool_call_reaches_the_transcript_without_its_arguments(processor record = processor.build_record( subject=SUBJECT, policy_bundle=POLICY, + enforcement_mode="declared", workload_digest=WORKLOAD, data_class="internal", model_provider="openai", diff --git a/integrations/openai-agents/test_openai_agents_to_trace.py b/integrations/openai-agents/test_openai_agents_to_trace.py index e14814e..fe29db0 100644 --- a/integrations/openai-agents/test_openai_agents_to_trace.py +++ b/integrations/openai-agents/test_openai_agents_to_trace.py @@ -83,6 +83,7 @@ def record(**kwargs) -> dict: base = dict( subject=SUBJECT, policy_bundle=POLICY, + enforcement_mode="declared", workload_digest=WORKLOAD, data_class="internal", model_provider="openai", @@ -95,10 +96,23 @@ def record(**kwargs) -> dict: # --- honesty rules -------------------------------------------------------- -def test_enforcement_mode_defaults_to_declared() -> None: - """The SDK enforces no policy. Anything else would describe enforcement that - did not happen.""" - assert record()["policy"]["enforcement_mode"] == "declared" +def test_enforcement_mode_has_no_default() -> None: + """TRACE spec section 4.3: declared MUST NOT be a default, and any other + default describes enforcement that did not happen. The caller states it.""" + base = dict( + subject=SUBJECT, + policy_bundle=POLICY, + workload_digest=WORKLOAD, + data_class="internal", + model_provider="openai", + model_id="gpt-5", + ) + with pytest.raises(TypeError, match="enforcement_mode"): + build_record(**base) + + +def test_declared_is_accepted_when_the_caller_states_it() -> None: + assert record(enforcement_mode="declared")["policy"]["enforcement_mode"] == "declared" def test_a_bare_run_carries_no_origin_block() -> None: @@ -205,6 +219,7 @@ def test_no_payload_reaches_the_record() -> None: processor.build_record( subject=SUBJECT, policy_bundle=POLICY, + enforcement_mode="declared", workload_digest=WORKLOAD, data_class="internal", model_provider="openai", @@ -264,6 +279,7 @@ def test_concurrent_runs_do_not_share_a_transcript() -> None: args = dict( subject=SUBJECT, policy_bundle=POLICY, + enforcement_mode="declared", workload_digest=WORKLOAD, data_class="internal", model_provider="openai", @@ -282,6 +298,7 @@ def test_building_before_any_run_is_refused() -> None: TraceRecordProcessor().build_record( subject=SUBJECT, policy_bundle=POLICY, + enforcement_mode="declared", workload_digest=WORKLOAD, data_class="internal", model_provider="openai",