From 028d782c2fc91f9683f60e373be6247f56e130a3 Mon Sep 17 00:00:00 2001 From: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Fri, 25 Sep 2026 06:47:15 +0000 Subject: [PATCH 01/10] feat(exec): compute ToolExecution-annotated calc defs and usages by tool MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A calc def or calc usage carrying AnalysisTooling::ToolExecution is computed by the named tool on every calc surface (sysml -calc, %calc, EvaluateCalc, derived attributes, document formulas): its ToolVariable-named in/inout parameters are the request's inputs, its ToolVariable-named out/inout parameters and its result parameter — keyed by its ToolVariable name, else its declared name, else result — bind from the reply, and the body is never evaluated or compiled. A tool unregistered, refusing or failing fails the calculation with the same typed errors a performance gets; equal inputs answered differently are noted as a ToolDivergence. Co-Authored-By: jason.han --- .../unreleased/tool-execution-calc.added.md | 1 + cmd/sysml/tool_calc_test.go | 139 +++++++ docs/guide/10-troubleshooting.md | 2 +- docs/project/spec-compliance.md | 2 +- docs/reference/cli.md | 5 +- docs/reference/environment.md | 15 +- internal/doc/queryexec/derived.go | 8 +- .../exec/analysis/testdata/toolcalc/main.go | 111 ++++++ internal/exec/analysis/tool_calc_test.go | 237 ++++++++++++ internal/exec/runtime/calc_usage.go | 35 +- internal/exec/runtime/declared.go | 11 + internal/exec/runtime/invoke_calc.go | 20 +- .../exec/runtime/robustness_toolcalc_test.go | 112 ++++++ internal/exec/runtime/tool.go | 68 ++-- internal/exec/runtime/tool_calc.go | 125 +++++++ internal/exec/runtime/tool_calc_test.go | 344 ++++++++++++++++++ internal/frontend/grpc/service.go | 7 + internal/frontend/repl/tool_calc_test.go | 104 ++++++ internal/frontend/repl/view.go | 1 + internal/frontend/usage/environment.go | 4 +- packaging/man/man1/sysml-grpc.1 | 6 +- packaging/man/man1/sysml.1 | 6 +- tests/grpc/conformance_test.go | 118 +++++- tests/grpc/testdata/conformance/README.md | 10 +- .../evaluate_calc_tool.expected.json | 13 + .../conformance/evaluate_calc_tool.sysml | 13 + ...luate_calc_tool_unregistered.expected.json | 9 + .../evaluate_calc_tool_unregistered.sysml | 13 + 28 files changed, 1479 insertions(+), 60 deletions(-) create mode 100644 changes/unreleased/tool-execution-calc.added.md create mode 100644 cmd/sysml/tool_calc_test.go create mode 100644 internal/exec/analysis/testdata/toolcalc/main.go create mode 100644 internal/exec/analysis/tool_calc_test.go create mode 100644 internal/exec/runtime/robustness_toolcalc_test.go create mode 100644 internal/exec/runtime/tool_calc.go create mode 100644 internal/exec/runtime/tool_calc_test.go create mode 100644 internal/frontend/repl/tool_calc_test.go create mode 100644 tests/grpc/testdata/conformance/evaluate_calc_tool.expected.json create mode 100644 tests/grpc/testdata/conformance/evaluate_calc_tool.sysml create mode 100644 tests/grpc/testdata/conformance/evaluate_calc_tool_unregistered.expected.json create mode 100644 tests/grpc/testdata/conformance/evaluate_calc_tool_unregistered.sysml diff --git a/changes/unreleased/tool-execution-calc.added.md b/changes/unreleased/tool-execution-calc.added.md new file mode 100644 index 0000000000..09c9b2a606 --- /dev/null +++ b/changes/unreleased/tool-execution-calc.added.md @@ -0,0 +1 @@ +- **`ToolExecution` on a `calc def` or calc usage.** A `calc def` or calc usage annotated `metadata ToolExecution { toolName = "…"; uri = "…"; }` is computed by the named tool wherever a calc is invoked — `sysml -calc`, `%calc`, `EvaluateCalc`, derived attributes and document formulas — its `in` parameters sent by their `ToolVariable` names, its `out` parameters and result parameter (under its `ToolVariable` name, else its declared name, else `result`) bound from the reply. The body never evaluates: a tool unregistered, refusing or failing fails the calculation with the same typed errors a performance gets, and equal inputs answered differently are noted as a divergence. diff --git a/cmd/sysml/tool_calc_test.go b/cmd/sysml/tool_calc_test.go new file mode 100644 index 0000000000..e11723275c --- /dev/null +++ b/cmd/sysml/tool_calc_test.go @@ -0,0 +1,139 @@ +package main + +import ( + "fmt" + "os" + "os/exec" + "path/filepath" + "strconv" + "sync" + "testing" + + "github.com/Open-MBEE/OpenSysML/internal/exec/analysis" + "github.com/Open-MBEE/OpenSysML/tests/testutil/gobuild" +) + +var ( + toolCalcOnce sync.Once + toolCalcPath string + toolCalcErr error +) + +// toolCalcStandin builds the tool stand-in of the analysis package's tests once per +// test binary. +func toolCalcStandin(t *testing.T) string { + t.Helper() + toolCalcOnce.Do(func() { + dir, err := os.MkdirTemp("", "toolcalc") + if err != nil { + toolCalcErr = err + return + } + toolCalcPath = filepath.Join(dir, "toolcalc") + build := exec.Command("go", gobuild.Args(toolCalcPath)...) + build.Dir = filepath.Join("..", "..", "internal", "exec", "analysis", "testdata", "toolcalc") + if out, err := build.CombinedOutput(); err != nil { + toolCalcErr = fmt.Errorf("go build: %v\n%s", err, out) + } + }) + if toolCalcErr != nil { + t.Fatalf("building the tool stand-in: %v", toolCalcErr) + } + return toolCalcPath +} + +// toolCalcManifest writes a manifest naming Thermo and points OPENSYSML_TOOLS at it. +func toolCalcManifest(t *testing.T) { + t.Helper() + dir := t.TempDir() + entry := `{"kind":"tool","toolName":"Thermo","version":"1.0.0",` + + `"executable":` + strconv.Quote(toolCalcStandin(t)) + `,` + + `"variables":["mass","power","warn","Tmax"]}` + if err := os.WriteFile(filepath.Join(dir, "thermo.json"), []byte(entry), 0o600); err != nil { + t.Fatal(err) + } + t.Setenv(analysis.ToolsEnv, dir) +} + +// toolCalcModel is a calc the tool computes; its bodyless result is answered from the reply. +const toolCalcModel = `package thermo { + private import ScalarValues::*; + private import AnalysisTooling::*; + private import ISQ::*; + + calc def Thermal { + metadata ToolExecution { toolName = "Thermo"; uri = "thermo://local"; } + in m : MassValue { @ToolVariable { name = "mass"; } } + in p : PowerValue { @ToolVariable { name = "power"; } } + out warn : Boolean { @ToolVariable { name = "warn"; } } + return : TemperatureValue { @ToolVariable { name = "Tmax"; } } + } +} +` + +// toolCalcDocument is a document whose table column reads a derived attribute that +// calls the tool calc — the formula the render run evaluates. +const toolCalcDocument = toolCalcModel + `package boards { + private import DocumentQueries::*; + private import KerML::Root::Element; + private import ScalarValues::*; + private import ISQ::*; + private import thermo::Thermal; + + part def Board { + attribute mass : MassValue = 2 [SI::kg]; + attribute power : PowerValue = 10 [SI::W]; + attribute Tmax : TemperatureValue = Thermal(mass, power); + } + part rack { + part b1 : Board; + } + + calc def Boards :> Query { + in root : Element = rack; + Project( + source = WhereType(source = Descendants(source = root, maxDepth = 1), type = "PartUsage"), + properties = ("name"), + columns = (Column(name = "Tmax", expression = Board::Tmax)) + ) + } + + part def BoardReport :> Document { + attribute redefines title = "Board Thermals"; + + part temps : Table { + attribute redefines caption = "Peak per board"; + calc rows : Boards; + } + } +} +` + +// sysml -calc answers what the tool answered when the manifest registers it. +func TestCLIToolCalcAnswersFromTheTool(t *testing.T) { + binary := buildCLI(t) + toolCalcManifest(t) + + wantReport(t, check(t, binary, toolCalcModel, + "-calc", "thermo::Thermal(2 [SI::kg], 10 [SI::W])"), 0, "= 20 [SI::K]") +} + +// Without a manifest entry the same invocation fails as not registered. +func TestCLIToolCalcRefusesAnUnregisteredTool(t *testing.T) { + binary := buildCLI(t) + t.Setenv(analysis.ToolsEnv, t.TempDir()) + + wantReport(t, check(t, binary, toolCalcModel, + "-calc", "thermo::Thermal(2 [SI::kg], 10 [SI::W])"), 2, + "tool 'Thermo' is not registered; set OPENSYSML_TOOLS") +} + +// A document whose table column reads a derived attribute calling the tool calc +// renders the value the tool answered. +func TestCLIToolCalcInADocumentFormula(t *testing.T) { + binary := buildCLI(t) + toolCalcManifest(t) + + wantReport(t, check(t, binary, toolCalcDocument, + "-render-document", "boards::BoardReport"), 0, "Tmax", "20") +} diff --git a/docs/guide/10-troubleshooting.md b/docs/guide/10-troubleshooting.md index 4c86c2731c..a827991ae9 100644 --- a/docs/guide/10-troubleshooting.md +++ b/docs/guide/10-troubleshooting.md @@ -30,7 +30,7 @@ in [reference/environment.md](../reference/environment.md). - Otherwise the solver could not decide the arithmetic, and the reason it reports says so **A run fails with "tool 'ModelCenter' is not registered; set OPENSYSML_TOOLS":** -- The action performed carries `AnalysisTooling::ToolExecution`, and an annotated action is only ever performed by the tool it names — never by evaluating its body. Point `OPENSYSML_TOOLS` at a directory holding one JSON file per tool (`toolName`, `version`, `executable`, `variables`), as [reference/environment.md](../reference/environment.md#external-tools) describes; `sysml -engines` then lists the tool as `tool:ModelCenter` with whether its executable was found +- The action performed — or the `calc def` or calc usage invoked — carries `AnalysisTooling::ToolExecution`, and an annotated action or calc is only ever run by the tool it names — never by evaluating its body. Point `OPENSYSML_TOOLS` at a directory holding one JSON file per tool (`toolName`, `version`, `executable`, `variables`), as [reference/environment.md](../reference/environment.md#external-tools) describes; `sysml -engines` then lists the tool as `tool:ModelCenter` with whether its executable was found - A tool that exits non-zero, answers something other than one JSON object of `outputs`, omits an output, names one no parameter receives, writes more than `OPENSYSML_TOOL_MAX_OUTPUT` (default 64 MiB), or takes longer than `OPENSYSML_TOOL_TIMEOUT` (default `10s`) fails the performance with that reason; no value is invented in its place **`sysml -engines` does not list the engine in `OPENSYSML_ENGINES`, or lists it `unavailable`:** diff --git a/docs/project/spec-compliance.md b/docs/project/spec-compliance.md index af7cd68ab4..5a0647e750 100644 --- a/docs/project/spec-compliance.md +++ b/docs/project/spec-compliance.md @@ -3132,7 +3132,7 @@ library types shows their signatures but no prose documentation, since the libra | Composition under `all`: a witnessed violation stands over any universal claim; a universal claim an execution refutes is a *disagreement*, resolved in the interpreter's favor, the refuted result demoted to *not covered* with the reason and both named in the plan; agreeing universal claims stand at the strongest earned, never promoted; differing observed values are a witnessed sensitivity; an engine the plan's deadline cancels is kept in the plan, marked cancelled with the bound it reached, and the composed result is at the strength the finished engines earned | `analysis/compose.go` `Compose`, `Disagreement`; `analysis/dispatch.go` `Step.Cancelled`, `Step.Bounds`, `Plan.Disagreements`, `Plan.Results` | `analysis/compose_test.go:TestComposeResolvesAContradictionInTheInterpretersFavor` (a proof-claiming engine in the shape the SMT engine will fill against `explore`'s witness), `:TestAllDemotesTheContradictedProofInThePlan`, `:TestComposeTakesTheStrongestEarnedWithoutPromotion`, `:TestComposeIsIndependentOfFinishOrder`, `:TestComposeSatisfiableWitnessStandsOverUnsat`, `:TestComposeDifferingValuesAreASensitivity`; `analysis/selection_test.go:TestAllKeepsFinishedResultsAndMarksTheCancelled`, `:TestAllWithNothingFinishedFailsWithTheDeadline`, `:TestAllWithACallerGoneKeepsWhatFinished`, `:TestAllStopsOnARunError`, `:TestAllRefusedByEveryEngineNamesEachRefusal` | ✅ As designed | | The engines of the build are listed — name, kind (`built-in` or the manifest entry kind), protocol (`-`, `object`, `stdio/1`), authority, the questions each answers, the bounds it declares and whether its process was found (`solve` names the solver it discovered), and for a manifest entry its file, command and version — in name order, on every surface; listing starts no process | `analysis/listing.go` `Registry.Listings`, `Listing`, `Lines`; `cmd/sysml/main.go` `-engines`; `repl/meta.go` `%engines`; `grpc/engines.go` `Service.ListEngines` | `analysis/registry_test.go:TestDefaultHoldsTheFrameworksEngines`, `:TestAbsentProcessListsAndRefuses`; `engines/engines_test.go:TestDefaultHoldsTheFrameworksEnginesAndSMT`, `:TestSMTStatusIsTheSolvers`, `:TestPresentProcessIsListed`; `cmd/sysml/engines_test.go:TestEnginesListsTheBuild`, `:TestEnginesExitsWithoutAModel`; `repl/engines_test.go:TestEnginesListsEveryRegisteredEngine`; `grpc/engines_test.go:TestListEnginesNamesEveryEngine`; `client/python/tests/test_engines.py` | ✅ | | Machine-readable standing: `-json` checks carry `plan` (selection, composed standing, each step's engine and status, the disagreements) and `results[]` (per answering engine: `engine`, `claim`, `strength`, `bounds` with `reached`, `witness`, `standing`) beside the keys they always carried; the wire carries `engine`, `strength` and `bounds` on `Verdict`, `RunAnalysisResponse`, `RunSweepResponse` and the verification responses, `engine` on their requests (unset is `auto`; a set field requires the `engines` capability), and `ListEngines`; the Python client takes `engine=` and reads `Verdict.engine`, `.strength`, `.bounds`; no existing key, field or flag changed | `cmd/sysml/report.go` `checkPlan`, `checkResultOf`, `checkBound`, `checkWitness`; `api/proto/sysml.proto` `Bound`, `EngineInfo`, `ListEngines`; `grpc/engines.go` `CapabilityEngines`; `grpc/analysis.go`, `grpc/sweep.go`, `grpc/verify.go`; `client/python/opensysml/engines.py` `Bound`, `Standing`, `EngineInfo`, `connection.py` `Connection.list_engines` | `cmd/sysml/engines_test.go:TestJSONReportsThePlan`, `:TestJSONReportsTheExploredPlan`; `grpc/engines_test.go:TestEngineFieldNeedsTheEnginesCapability`, `:TestVerdictsCarryEngineStrengthAndBounds`; `client/python/tests/test_engines.py`; `make proto-breaking`, `make man-check` | ✅ (whether the `-json` keys are a patch or a minor change is an item of the release checklist in `CONTRIBUTING.md`) | -| External tools: a performance of an action carrying `AnalysisTooling::ToolExecution` is put to the `tool:` engine of its `toolName`, never to the action's body — one process per performance, one JSON object each way over its standard input and output, `toolName` and `uri` passed through uninterpreted, `inputs`/`outputs` keyed by `ToolVariable.name` with value and unit, outputs converted to the coherent unit of the parameter's declared quantity kind; each `*.json` entry of the directory `OPENSYSML_TOOLS` names (`toolName`, `version`, `executable`, `variables`) registers one engine answering `compute` at authority *observed*, listed with its process status like `solve`; a tool with no entry is refused with `tool 'ModelCenter' is not registered; set OPENSYSML_TOOLS`; a non-zero exit, malformed or missing output, an output no parameter receives, a unit the model does not declare and the timeout `OPENSYSML_TOOL_TIMEOUT` (default 10s) are typed errors that fail the performance — no default value is invented; equal inputs answered differently are a `tool-divergence` note of the run | `analysis/manifest.go` `LoadManifest`, `ExternalsFromEnv`, `ToolEntry`, `ManifestError`, `ToolAbsentError`; `analysis/registry.go` `DefaultFromEnv`; `analysis/tool.go` `NewTool`, `toolEngine.Covers`, `toolEngine.Run`, `ToolRequestOf`, `ToolReplyOf`, `WrongToolError`, `ToolVariableError`; `analysis/tool_runner.go` `toolRunner.RunTool`; `analysis/dispatch.go` `Registry.AnswerWith` (the plan-scoped runner); `runtime/tool.go` `ActionExecutor.performByTool`, `ToolCall`, `ToolRunner`, `ToolNotRegisteredError`, `ToolError`, `ToolDivergence`; `semantics/coherent_unit.go` `Model.CoherentUnitFor`; `cmd/sysml/main.go`, `grpc/service.go` (`DefaultFromEnv` at startup) | `analysis/tool_fixture_test.go:TestPilotFixtureRunsAgainstTheStandIn` (the pilot `AnalysisAnnotation` against the Go stand-in `testdata/toolstandin`), `:TestPilotFixtureFailsWithTheToolsFault`, `:TestPilotFixtureTimesOut`, `:TestPilotFixtureRefusesAnUnregisteredTool`, `:TestPilotFixtureRefusesAnAbsentExecutable`, `:TestPilotFixtureReportsANonDeterministicTool`, `:TestToolEngineDispatch`; `analysis/tool_test.go:TestManifestReadsOneEntryPerJSONFile`, `:TestManifestFaultsAreTyped`, `:TestManifestRefusesTwoEntriesForOneTool`, `:TestExternalsFromEnvUnsetIsNoTool`, `:TestExternalsFromEnvRegisterEachEntry`, `:TestToolRegistrationIsIsolatedAndUnique`, `:TestToolEngineCoversItsOwnComputationsOnly`, `:TestToolTimeoutFromEnv`, `:TestToolRequestCarriesTheCallUninterpreted`, `:TestToolReplyIsOneObjectOfOutputsOrAnError`; `runtime/tool_test.go:TestToolExecutionPerformsThroughTheRunner`, `:TestToolExecutionWithoutRunnerIsNotRegistered`, `:TestToolExecutionRefusesBadAnswers`, `:TestToolExecutionNotesDivergence` | ✅ As designed (under `smt`, a tool output is a free input — the contract that engine will meet; it is not on this branch) | +| External tools: a performance of an action carrying `AnalysisTooling::ToolExecution` is put to the `tool:` engine of its `toolName`, never to the action's body — one process per performance, one JSON object each way over its standard input and output, `toolName` and `uri` passed through uninterpreted, `inputs`/`outputs` keyed by `ToolVariable.name` with value and unit, outputs converted to the coherent unit of the parameter's declared quantity kind; each `*.json` entry of the directory `OPENSYSML_TOOLS` names (`toolName`, `version`, `executable`, `variables`) registers one engine answering `compute` at authority *observed*, listed with its process status like `solve`; a tool with no entry is refused with `tool 'ModelCenter' is not registered; set OPENSYSML_TOOLS`; a non-zero exit, malformed or missing output, an output no parameter receives, a unit the model does not declare and the timeout `OPENSYSML_TOOL_TIMEOUT` (default 10s) are typed errors that fail the performance — no default value is invented; equal inputs answered differently are a `tool-divergence` note of the run. `ToolExecution` on a `calc def` or calc usage is the same contract on every calc surface (`sysml -calc`, `%calc`, `EvaluateCalc`, derived attributes, document formulas): the `in`/`inout` parameters carrying `ToolVariable` are the inputs, the `out`/`inout` parameters carrying it plus the result parameter — under its `ToolVariable` name, else its declared name, else `result` — bind from the reply, and the body never evaluates | `analysis/manifest.go` `LoadManifest`, `ExternalsFromEnv`, `ToolEntry`, `ManifestError`, `ToolAbsentError`; `analysis/registry.go` `DefaultFromEnv`; `analysis/tool.go` `NewTool`, `toolEngine.Covers`, `toolEngine.Run`, `ToolRequestOf`, `ToolReplyOf`, `WrongToolError`, `ToolVariableError`; `analysis/tool_runner.go` `toolRunner.RunTool`; `analysis/dispatch.go` `Registry.AnswerWith` (the plan-scoped runner); `runtime/tool.go` `ActionExecutor.performByTool`, `ToolCall`, `ToolRunner`, `ToolNotRegisteredError`, `ToolError`, `ToolDivergence`; `runtime/tool_calc.go` `calcToolCall`, `computeCalcByTool` with the hooks in `runtime/invoke_calc.go` and `runtime/calc_usage.go`; `runtime/declared.go` `NewDeclaredReaderIn` (derived features read through the held runner); `semantics/coherent_unit.go` `Model.CoherentUnitFor`; `cmd/sysml/main.go`, `grpc/service.go` (`DefaultFromEnv` at startup) | `analysis/tool_fixture_test.go:TestPilotFixtureRunsAgainstTheStandIn` (the pilot `AnalysisAnnotation` against the Go stand-in `testdata/toolstandin`), `:TestPilotFixtureFailsWithTheToolsFault`, `:TestPilotFixtureTimesOut`, `:TestPilotFixtureRefusesAnUnregisteredTool`, `:TestPilotFixtureRefusesAnAbsentExecutable`, `:TestPilotFixtureReportsANonDeterministicTool`, `:TestToolEngineDispatch`; `analysis/tool_test.go:TestManifestReadsOneEntryPerJSONFile`, `:TestManifestFaultsAreTyped`, `:TestManifestRefusesTwoEntriesForOneTool`, `:TestExternalsFromEnvUnsetIsNoTool`, `:TestExternalsFromEnvRegisterEachEntry`, `:TestToolRegistrationIsIsolatedAndUnique`, `:TestToolEngineCoversItsOwnComputationsOnly`, `:TestToolTimeoutFromEnv`, `:TestToolRequestCarriesTheCallUninterpreted`, `:TestToolReplyIsOneObjectOfOutputsOrAnError`; `runtime/tool_test.go:TestToolExecutionPerformsThroughTheRunner`, `:TestToolExecutionWithoutRunnerIsNotRegistered`, `:TestToolExecutionRefusesBadAnswers`, `:TestToolExecutionNotesDivergence`; `runtime/tool_calc_test.go` (`TestToolCalc*` — binding, conversion, derived attributes, a body never run, every typed failure), `runtime/robustness_toolcalc_test.go:TestRuntimeRobustnessToolCalc`; `analysis/tool_calc_test.go` (`TestToolCalc*` against the `testdata/toolcalc` stand-in — answer, refusal, exit, timeout, unregistered); `repl/tool_calc_test.go`, `cmd/sysml/tool_calc_test.go` (`%calc`, `-calc`, `-render-document`); `tests/grpc/testdata/conformance/evaluate_calc_tool*` | ✅ As designed (under `smt`, a tool output is a free input — the contract that engine will meet; it is not on this branch) | | External engines: each `kind: engine` entry of the directory `OPENSYSML_ENGINES` names (read beside `OPENSYSML_TOOLS` under one rule set — outside every workspace, owner-writable only, `command` confined to the manifest directory and never looked up on `PATH`, duplicate names refused across both directories, `policy` and `sampler` entries parsed and listed as not served) registers an engine under its own name, spoken to over its standard input by JSON-RPC-shaped lines — `describe` checked against the entry field by field, `covers`, `run`, `cancel`, `progress` coalesced to a quarter second per open run, errors `unsupported`/`budget`/`internal` — one process per plan, several open requests on it when `concurrent`, one line and the captured standard error each bounded by `OPENSYSML_TOOL_MAX_OUTPUT` (default 64 MiB); the model is handed as `sources` or as `graphs:1`, the versioned byte-stable export of the lowered `ActionGraph`/`StateGraph`, `rdf` refused as a later stage; nothing the engine claims stands unchecked: `violated` is *witnessed* when its schedule replays and the property is false at the move named, `sensitive` when two schedules replay and end the feature differently, `satisfiable` when the assignment is confirmed, a universal claim *observed* over `executions` that replay and otherwise *not covered* with the claim kept, `admit` refused naming the referee-record stage; `auto` reaches an external engine after every built-in engine refused, `all` composes it with them and a stood claim beside a built-in witness is a disagreement; `-engines`, `%engines` and `ListEngines` list it (`-engines -probe` and `%engines probe` start each engine once), the service lists but does not serve it until `-serve-external-engines` | `analysis/manifest.go` `ManifestsFromEnv`, `ExternalsFromEnv`, `EntryKind`; `analysis/engine_entry.go` `EngineEntry`, `confinedPath`, `NotServedError`; `analysis/process.go` `outputLimitFromEnv`, `boundedBuffer`; `analysis/session.go` `session`, `enginePool`, `Reporter`, `ProgressInterval`; `analysis/enginewire/wire.go`; `analysis/external.go` `externalEngine`, `SubjectError`; `analysis/external_question.go`; `analysis/external_standing.go`; `analysis/withheld.go`; `analysis/listing.go` `Listing.Origin`, `Registry.Probed`; `analysis/modelform/graphs.go` `GraphsOf`, `GraphsVersion`; `analysis/modelform/sources.go` `SourcesOf`; `cmd/sysml/main.go` `-engines -probe`; `repl/engines.go` `%engines probe`; `grpc/engines.go` `ListEngines`; `cmd/sysml-grpc/main.go` `-serve-external-engines`; `docs/reference/engine-protocol.schema.json` | `analysis/engine_entry_test.go:TestManifestReadsEveryKind`, `:TestEngineEntryFaultsAreTyped`, `:TestManifestRefusesTwoEntriesForOneEngineName`, `:TestEngineCommandIsConfinedToTheManifestDirectory`, `:TestToolExecutableIsConfinedToTheManifestDirectory`, `:TestEngineEntryPresentChecksTheProgramWithoutRunning`, `:TestManifestUnderAWorkspaceIsNotRead`, `:TestManifestRefusesWritableByOthers`; `analysis/process_test.go:TestOutputLimitFromEnv`, `:TestBoundedBufferKeepsThePrefixAndStops`, `:TestToolOverflowNamesTheBound`; `analysis/session_test.go:TestSessionHandshakeCoversAndRun`, `:TestSessionHandshakeMismatchIsTypedPerField`, `:TestSessionStartupFailuresAreTyped`, `:TestSessionErrorCodesAreTyped`, `:TestSessionProtocolBreaksEndTheSession`, `:TestSessionLineOverTheBoundEndsTheSession`, `:TestSessionExitDuringRunIsTyped`, `:TestSessionCancelIsAnswered`, `:TestSessionIgnoredCancelEndsTheProcess`, `:TestSessionProgressIsCoalesced`, `:TestSessionMatchesConcurrentAnswersByID`, `:TestEnginePoolHonorsConcurrent`, `:TestEnginePoolEndsTheEngineForThePlanAtTheFirstFault`; `analysis/external_standing_test.go:TestExternalViolationStandsAfterReplay`, `:TestExternalViolationRefusedWhenThePropertyHolds`, `:TestExternalViolationIsJudgedAtTheMoveNamed`, `:TestExternalWitnessThatDoesNotReplay`, `:TestExternalWitnessShapeIsChecked`, `:TestExternalWitnessInputsAreRefused`, `:TestExternalEntryWithoutWitnessesEarnsNoExistential`, `:TestExternalSensitivityNeedsTwoDivergingSchedules`, `:TestExternalExecutionsAreObservedAfterReplay`, `:TestExternalExecutionsFailingTheClaimEarnNothing`, `:TestExternalUniversalClaimsWithoutExecutionsAreNotCovered`, `:TestExternalNoneIsTheEnginesOwnRefusal`, `:TestExternalAssignmentIsConfirmed`, `:TestExternalEngineInAutoAndAll`, `:TestExternalStoodClaimDisagreesWithBuiltIn`, `:TestExternalPlanIsTheSameUnderEitherConcurrency`, `:TestExternalSubjectsAreDeclarationKinds`; `analysis/schema_test.go:TestSchemaValidatesEveryStandinMessage`, `:TestSchemaRefusesWhatTheSessionRefuses`, `:TestSchemaMatchesTheWireTypes`; `analysis/modelform/graphs_test.go:TestGraphsActionCarriesTheLoweredGraph`, `:TestGraphsStateCarriesTheLoweredGraph`, `:TestGraphsAreByteStable`, `:TestSourcesOfListsEveryDocumentInOrder`, `:TestRefuseRDFFormIsTyped`, `:TestGraphsMatchTheGoldens`; `cmd/sysml/external_engines_test.go:TestEnginesSpawnsNothingAndProbeSpawnsEachEntryOnce`, `:TestEnginesUnderTheWorkspaceIsNotRead`, `:TestProgressGoesToStandardError`; `repl/engines_test.go:TestEnginesListsManifestEnginesAndProbesOnRequest`, `:TestProgressIsPrintedWhereTheSessionSays`; `grpc/engines_test.go:TestManifestEnginesAreListedButNotServedByDefault`, `:TestServeExternalEnginesRunsTheNamedManifestEngines` | ✅ As designed (stage 1 of the [bring-your-own-engine design](../internals/design/bring-your-own-engines.md); referee records and `admit`, `policy`/`sampler` strategies, the `rdf` form, in-process Go, WebAssembly and the `grpc` transport are later stages) | | Parallel runs: `Budget.Jobs` (`-jobs `, `%jobs `, `OPENSYSML_JOBS`; default one per CPU; a count below one or no integer is the typed `JobsError` before anything runs; the flag overrides the environment; the gRPC service takes the serving binary's count, no request field) is how many runs of one plan go at once, each on a worker of its own — a resolver and semantic model per job over the shared frozen index, built lazily per plan, `Model.NewContext` being job 0 — so no run sees another's memo; the result of a plan does not depend on `Jobs`: `explore` runs its prefixes on a work queue ordered as the sequential exploration takes them (`Runs` a cut in that order, committed and speculative runs, at most `Jobs` speculative runs discarded in a plan's lifetime, never more than `Runs + Jobs` executions, the table merged by outcome identity with the least witness; a violating run is an outcome of the table as under one job, `explore` answering the universal `outcomes` alone, so the witness cut awaits an existential question on the queue), `all` composes in name order whatever the order of finishing, and `sweep` runs its rows on the `Jobs` workers, each row in a fresh run-owned context over the plan's worker (`SweepRun` takes the row's context; `SweepRow.Context` keeps it for the row's readers, so nothing a row produced is read through another context) and assembles its table in plan order whatever order the rows finish in, `Runs` remaining the row limit and one job the former loop; the REPL, CLI and gRPC paths instantiate a row's subject and `self` in the row's context and carry the arguments' values into it (`Context.Carry`: evaluated once at the prompt, so a held feature a run wrote reads as written; every object a value names — alone, in a collection, as a function's self — the one the row makes for it, the row's subject when they coincide; a deferred expression or a function closing over a run's bindings the typed `NotPortableError` under `SweptArgumentError`); the gRPC response renumbers each row's objects so the request-wide `instances` table names every row's own; `%sweep` on a held object runs each row on an object of its own that is the held object as the sweep found it: the held object's declaration materialized afresh when the held closure is pristine — reached from a declaration (one nested in another's feature on the like of its root, walked along the same path), the root not destroyed, no feature written since materialization, no signal posted to it awaiting dispatch, every behavior its type exhibits or performs still as its start left it (the executors record whether they have moved at their own step points — a step that fails after moving a token, writing a performance or leaving a state included — `ActionExecutor.moved`/`StateExecutor.moved`, carried through `Snapshot`/`Restore`), none waiting on a clock that has left zero, no signal open to any taker in flight while it runs a behavior — and otherwise a copy from one image of the held graph (`Context.Image`, `HeldImage.Materialize`: the roots' closure by value — identities, lives, owners, feature values, the messages bound for it and, where it runs a behavior, those open to any taker, occurrences and the executors' captured state — taken once while the session's state is held and made in every row's context under the same identities, with fresh executors on the row's clock; `#id`-named objects admitted when the image holds the identity, feature paths when the copy resolves them, which it does whenever the held graph does), refusing with the typed `SweptObjectError` naming the reason what the image cannot carry (`ErrSnapshotMidRun`, `ErrSnapshotPausedBody`, `NotPortableError`, `HeldImageError`), never sharing the session's context, whose state lock is released while the rows run; a deadline met mid-sweep starts no further row, discards the rows in flight and reports the caller's error and no table; `-json` `plan` carries `workers` and `warming` (ms), the human-readable report neither | `analysis/jobs.go` `ParseJobs`, `JobsFromEnv`, `DefaultJobs`, `JobsError`; `analysis/result.go` `Budget.Jobs`, `BudgetOf`; `analysis/worker.go` `Model.WorkerAt`, `Model.NewContextOn`, `warmed`; `analysis/dispatch.go` `answerAll`, `allCoordinator`, `Plan.Workers`, `Plan.Warming`; `runtime/explore.go` `ExploreWith` (`Explore` is its one-job form); `runtime/explore_queue.go` `exploreQueue` (`work`, `next`, `startable`, `finish`, `fold`); `runtime/sweep.go` `SweepRun`, `SweepRow.Context`, `runSweepRow`; `runtime/sweep_queue.go` `RunSweepWith`, `sweepQueue` (`work`, `take`); `runtime/pristine.go` `Context.Pristine`, `HeldStateError`; `runtime/action_executor.go` `ActionExecutor.moved`, `newActionExecutorOn`; `runtime/state_executor.go` `StateExecutor.moved`, `newStateExecutorOn`; `runtime/snapshot.go` `actionCapture.moved`, `stateCapture.moved`; `runtime/held_image.go` `Context.Image`, `HeldImage.Materialize`, `HeldImage.Holds`, `HeldImageError`, `ErrImageIdentityTaken`, `ErrImageBindingTaken`, `ErrImageClock`, `ErrImageBound`, `ErrImageRoot` (a destroyed root refused as `ErrOccurrenceDestroyed`; a failed materialization leaving the destination as it found it); `runtime/held_image_behavior.go` (executor state by value and its translation); `runtime/classifier_behavior.go` `bindClassifierBehavior`; `runtime/carry.go` `Context.Carry`, `Bring`, `NotPortableError`; `analysis/sweep.go` `sweepEngine.Run` (`Model.NewContextOn` per job); `repl/sweep.go` `Session.runSweep`, `sweptArgs`, `sweptValue`, `sweptHeld`, `sweptImage`, `rowObjects`, `sweptObject`, `sweptOwner`, `sweptRef`, `SweptObjectError`, `SweptArgumentError`, `sweepTableLines`; `repl/lookup.go` `Session.heldRoot`, `heldLabel`; `grpc/sweep.go` `Service.RunSweep`, `sweepResponse`, `rowIDs`; `grpc/verify.go` `verifyContext.on`; `grpc/convert.go` `ValueToProtoIn`, `InstanceToProto`; `cmd/sysml/main.go` `-jobs`; `cmd/sysml/report.go` `checkPlan.Workers`/`Warming`; `repl/jobs.go` `Session.Jobs`, `Session.SetJobs`; `repl/meta.go` `%jobs`; `grpc/service.go` `Service.jobs` (from `JobsFromEnv` at construction); `usage/environment.go` `OPENSYSML_JOBS` | `runtime/explore_queue_test.go:TestExploreWithIsExploreOverTheConformanceCorpus` (one job against eight over every case with an admissible set), `:TestExploreWithKeepsTheLeastWitnessOfALaterFasterViolation`, `:TestExploreWithCutsRunsJustAboveTheWitness` (`testdata/later_prefix_violates_faster.sysml`), `:TestExploreWithNeverRunsMoreThanRunsPlusJobs` (conformance `action_explore_slow_first_writer.sysml`, a bounded recursion for the slow prefix), `:TestExploreWithBuildsEachRunOnItsJob`, `:TestExploreWithStopsWhenTheCallerGoesAway`, `:TestExploreWithFailsWhenFreshFails`; `analysis/jobs_test.go:TestJobsFromEnv`, `:TestParseJobsNamesItsSource`; `analysis/isolation_test.go:TestPlansOnOneModelHaveWorkersOfTheirOwn` (`Jobs` workers per plan, two plans on two goroutines under `-race`); `analysis/all_test.go:TestAllPlanCountsTheWorkersOfEveryEngine`; `cmd/sysml/jobs_test.go:TestJobsFlagAndEnvironment`, `:TestJSONIsTheSameOnOneJobAsOnEight` (the three determinism fixtures and a sweep through `-json`), `:TestSweepReportsAreTheSameOnOneJobAsOnEight` (every CLI sweep golden and the trade-study sweep, text and `-json`), `:TestJSONReportsThePlanWorkers`; `repl/jobs_test.go:TestJobsShowsAndSetsTheCount`, `:TestSetJobsReachesTheBudgetAndKeepsTheSession`; `runtime/sweep_queue_test.go:TestRunSweepWithTablesRowsInPlanOrderWhateverTheirArrival` (a context per row among its checks), `:TestRunSweepWithKeepsEachRowsWritesToItself`, `:TestRunSweepWithStopsAtTheDeadline`, `:TestRunSweepWithFailsWhenAJobHasNoContext`; `runtime/carry_test.go:TestCarryTakesAValueIntoAnotherContext`, `:TestCarryRefusesWhatNoOtherContextHolds`; `runtime/value_kinds_test.go:TestEveryValueKindIsDispatched` (carrying among the surfaces walked); `analysis/isolation_test.go:TestSweepsOnOneModelHaveWorkersOfTheirOwn` (two sweeps on one model on two goroutines under `-race`); `repl/sweep_rows_test.go:TestSweepRowsReadAlikeOnOneJobAndOnEight` (every REPL sweep golden, the trade-study sweeps and a descending bounded recursion whose rows arrive out of plan order), `:TestSweepRowsWritingTheSubjectKeepTheWritesToThemselves`, `:TestSweepOverAnObjectNotAsItsDeclarationMadeItRunsOnItsImage`, `:TestSweepOverAMovedStateMachineRunsOnItsImage`, `:TestSweepOverAnObjectNoImageCarriesIsRefused`, `:TestSweepOverADestroyedObjectIsRefused`; `repl/sweep_held_test.go:TestSweepOverAFreshObjectRunningABehaviourReadsAsTheSequentialFormDid` (the sequential form's table pinned on one job and on eight), `:TestSweepOverAMovedPerformedActionRunsOnItsImage`; `runtime/pristine_test.go:TestPristineFollowsAStateMachinesMoves`, `:TestPristineFollowsAPerformedActionsMoves` (the record through the moves and through `Snapshot`/`Restore`; a signal posted refused), `:TestPristineRefusesATimedBehaviorOnceTheClockHasMoved`, `:TestPristineRefusesABehavingObjectWhileAnOpenMessageIsInFlight`, `:TestActionStepFailingAfterAMoveRecordsIt`, `:TestActionBodyFailingAfterAWriteRecordsTheMove`, `:TestStateChangeTransitionFailingRecordsTheMove`; `runtime/held_image_test.go:TestHeldImageCarriesAMovedStateMachine`, `:TestHeldImageOfAFreshObjectIsPristine`, `:TestHeldImageCarriesAParkedAction`, `:TestHeldImageCarriesTheRunsSchedulePolicy`, `:TestHeldImageRefusesWhatItCannotCarry` (a destroyed root among them), `:TestHeldImageRefusesAMessageNamingAnObjectNotHeld`, `:TestHeldImageOfABehaviorlessClosureLeavesTheBusBe`, `:TestHeldImageRefusesADestinationWithAWaitDueBeforeItsInstant` (one already due at the destination's instant among them), `:TestHeldImageRefusesAPausedBody`, `:TestHeldImageMaterializeFailsWhole` (a destination on a shared sequence among them), `:TestHeldImageKeepsTheIdentitiesSetAsideForConnectors`, `:TestHeldImageRefusesAnIdentitySetAsideByTheDestination`, `:TestHeldImageRefusesAUsageDenotingAnotherObjectOfTheDestination`, `:TestHeldImageRoundTrip` (every conformance case the image admits, captured in one context, made in another and run beside the source), `:TestHeldImageServesConcurrentSweeps` (one image, two sweeps of eight rows on two goroutines under `-race`); `cmd/sysml/sweep_test.go:TestSweepOverAnInstantiatedObjectRunningABehaviourThroughCLI`; `grpc/sweep_test.go:TestRunSweepOverASubjectRunningABehaviour`, `:TestRunSweepOverSubjectsRunningABehaviourConcurrently`, `:TestSweepOverAnObjectReachedThroughAnotherRunsOnItsLike`, `:TestSweepArgumentsReadAsThePromptReadsThem`, `:TestSweepArgumentsNamingAnObjectBindTheRowsOwn`, `:TestSweepLeavesADebuggerStepping`; `grpc/sweep_test.go:TestRunSweepAnswersAlikeOnOneJobAndOnEight`, `:TestRunSweepRowsNameTheirOwnObjects`; `make man-check` | ✅ Faithful (what the image does not carry is refused, typed: a body paused mid-statement, a session inside a step, a value closed over its run) | | The `smt` engine (`internal/exec/smt`, the [SMT model checking design](../internals/design/smt-model-checking.md), stages 1 to 3): a `holds` question about an action with the schedule free, the inputs free or both is put to an SMT solver over the lowered `ActionGraph` — straight-line bodies, fork, join, merge, decisions (two holding guards are the executor's *decision branch* choice, ranged over, not asserted impossible), body loops unrolled `Budget.Unroll` times (`analysis.DefaultUnroll`, 4, when zero; the `-check-unroll` flag and `%check-bounds unroll=` set it; reported as the `unroll` bound), pins and object flows; clock, messages, nested flows, object-valued assignments, calc invocation and nonlinear arithmetic refuse the whole behavior before any query, naming the node and construct. In `s_0` a feature the model binds — a default, a value the performing object holds, a value the caller fixed through `Start` — is asserted equal to it and a feature it leaves unbound is a free variable whose domain is the declared type's: `Boolean`, `Integer` within the runtime's signed 64-bit range, `Natural` as an integer `>= 0`, `Real`, `Rational` and a quantity type over one as the solver's reals, an enumeration or variation point as its constructors; a `HoldsAsk.Inputs` name (`-check-input`, `%check-input`) drops a bound feature's binding so it ranges over that domain, and a name that is no feature the action reads (nothing, an output, a result, a node) is the typed `InputError` before any query; a declared type the encoding cannot narrow to a sort (`String`, a collection, an object-valued feature, a type with no translation) is the typed `DomainError` naming the feature and type, *not covered* before any query, never a silent unconstrained variable; a pinned value is checked against its domain the same way. A `HoldsAsk.Assume` constraint or requirement (`-check-assume`, `%check-assume`) is translated as any condition is and asserted over `s_0`, one the translator refuses being refused naming the construct before any query; the assumptions are asked for consistency first and a set no initial state satisfies is *not covered: assumptions admit no initial state*, never *proved*. The requirement or constraint property and the deadlock property are asked as `∃ i ≤ k. ¬R(s_i)` and `∃ i. stutter ∧ ¬complete` over `k = Budget.Depth` moves (`smt.DefaultMoves`, 40, when the engine is asked alone without a depth) under `Budget.Solver`; `unsat` is *proved* only when no schedule reaches the move, unroll or slot bound at `k` and *bounded* otherwise, the proof meaning *for every value of the free inputs in their domains* and its `Result.Inputs` and `Result.Assumptions` listing each input with its type, domain, value or freedom and each assumption; every `sat` model is decoded to the free inputs' values (`InputTaken`, a rational spelled exactly, a constructor by its qualified name) and `ChoiceTaken` lines, replayed through the interpreter under `replay:` with the inputs fixed before the defaults and the moves followed after, and is *violated (witnessed)* only when the replay reaches the state the solver described — a replay that diverges or cannot set an input, `unknown`, a timeout and a refusal are *not covered* with the reason. A `sensitive` question — asked when `-check-diverge`/`%check-diverge` names a feature, whichever engine answers, so `check` and `smt` read one flag and `-engine all` puts one question to both — is decided per feature by a two-copy query: copy B of the whole encoded relation shares `s_0` and the free inputs with copy A and renames every other variable, both copies `complete` within `k` moves and the feature's final values differ; a `sat` decodes to two schedules, both replayed through `runtime.Replay` before the verdict is *sensitive (witnessed)* with the two final values, the first move the schedules part at and the move each took, `Result.Witness` and `Result.Contrast` the pair and `-check-witness` writing `-A` and `-B` files either of which `replay:`/`%replay` follows, a replay that does not reproduce its value *not covered* with the disagreement; on `unsat` the deadlock and typed-error properties are asked next, a `sat` there that *violated* finding, then `∃ schedule. cut_k ∨ loopflag`, whose `sat` is `no sensitivity found within k moves` as (`holds`, *bounded*) — the pair `check`'s clean exhaustive search of the same question answers — and whose `unsat` alone is (`holds`, *proved*); the three are separate queries, so no `CapIncremental` is needed; a feature of the performing object (`this.level`) and a body stage 4 encodes (the clock, a paused nested flow) are per-feature and whole-behavior typed refusals, never a silently narrowed answer, and absent a named feature the default set is `check`'s — the action's attributes, answered, and the performer's, refused each by name. A fixed schedule, `outcomes` and every other question kind are refused with a typed reason; `explore` and `check` keep refusing `FreeInputs` with theirs, so `-engine all` with `-check-input` shows `check` refused in the plan beside `smt`'s answer. The engine is `analysis.External` over `OPENSYSML_SMT`, `z3`, `cvc5`, registered at authority *proved* in the build's registry `engines.Default()` (`internal/exec/engines`, which composes `analysis.Default()` — the framework's own engines, which `smt` imports and so cannot be constructed from — with `smt`; the CLI, REPL and gRPC service build it), so `-engine smt` and `%engine smt` reach it and `-engines`, `%engines` and `ListEngines` list it with its solver as its status; no surface asks `holds` under `auto`, so its registration moved no verdict or plan line, and a tool-computed output is not yet a free input because a body performing a `tool:` engine is refused as stage 1 refuses it | `smt/support.go` `Analyze`, `Flow`, `BodyLoop`, `DefaultUnroll`; `smt/state.go` `Sorts`, `State`, `Move`; `smt/encode.go` `Encode`, `Encoding`, `Encoding.initial`, `Encoding.domain`; `smt/input.go` `Encoding.frees`, `Encoding.checkReleases`, `Encoding.domainText`, `noDomain`; `smt/property.go` `Encoding.Conditions`, `Encoding.Assume`, `Encoding.Deadlock`, `Encoding.Violation`, `Encoding.Failure`, `Encoding.Uncertainty`, `Encoding.Cuts`, `Encoding.Completion`, `Encoding.Outputs`; `smt/witness.go` `Encoding.Decode`, `Encoding.decodeRun`, `Encoding.decodeInputs`, `Encoding.DecodeOutputs`, `Encoding.Fix`; `smt/twocopy.go` `CopyPrefix`, `CopyNames`, `Pair`, `Encoding.Pair`, `Encoding.Sensitivity`, `Encoding.DecodePair`, `Diverging`; `smt/sensitive.go` `run.decideSensitive`, `run.diverging`, `run.sensitive`, `run.replayRun`, `NoSensitivityWithin`; `analysis/ask.go` `CheckKind`; `analysis/check.go` `SensitivityFile`, `divergenceWitness`; `analysis/compose.go` (a stated `ClaimSensitive` stands over `ClaimHolds` about the same feature); `runtime/check.go` `ResolveCheckFeatures`, `FeatureOwner`, `ActionExecutor.PerformerAttributes`; `smt/engine.go` `New`, `Engine.Unrolling`, `Engine.Covers`, `Engine.Run`, `Engine.Process`, `DefaultMoves`, `NoInitialState`, `run.decide`, `run.inputs`, `run.replay`, `run.write`, `ScheduleError`; `smt/errors.go` `ErrNotEncoded`, `UnsupportedError`, `FlowError`, `ErrSlotOverflow`; `analysis/engine.go` `InputError`, `ErrInput`, `DomainError`, `ErrDomain`, `FreedomError`; `analysis/question.go` `HoldsAsk` (`Conditions`, `Inputs`, `Assume`, `Diverge`), `Question.Holds`, `FreeInputs`; `analysis/result.go` `Result.Contrast`; `analysis/result.go` `Input`, `Result.Inputs`, `Result.Assumptions`, `Budget.Unroll`, `DefaultUnroll`, `Witness.Inputs`; `analysis/standing.go` `Result.inputsEvidence`; `engines/engines.go` `Default`, `DefaultFromEnv`; `runtime/replay.go` `InputTaken`, `ParseInput`, `ParseWitness`, `Witness.Inputs`, `WitnessInputError`; `runtime/action_executor.go` `ActionExecutor.fixWitnessInputs`; `cmd/sysml/usage.go` (`-check-input`, `-check-assume`, `-check-unroll`); `cmd/sysml/report.go` `checkInput`; `repl/meta.go` (`%check-input`, `%check-assume`, `%check-bounds unroll=`) | `smt/referee_test.go:TestRefereeCorpus` (checks 1–3 over every corpus action case with `outcomes` whose body encodes: outcome sets equal `explore`'s, every witness replays, the verdict agrees with exhaustive exploration; check 5: every feature the case's action writes is asked as a `sensitive` question, a *sensitive* verdict has both witnesses replayed to two values among the case's `outcomes` and a *not sensitive* feature has one value across them; the refused cases are counted and named), `:TestRefereeInputs` (check 4: every *violated* witness with inputs, replayed under `explore` with those inputs pinned, reproduces the violation), `smt/input_test.go:TestUnboundInputsAreFreeAndBoundOnesPinned`, `:TestFreeInputRangesOverItsDomain`, `:TestReleasingABoundInputDropsItsBinding`, `:TestFreeInputWithoutADomainIsRefused`; `smt/input_engine_test.go:TestEngineRangesOverUnboundInputs`, `:TestEngineNarrowsTheDomainToTheDeclaredType` (*proved* under `Natural`, *violated* under `Integer`), `:TestEngineWitnessNamesAnEnumerationConstructor`, `:TestEngineAssumesOverTheInitialState` (an assumption turns *violated* into *proved*; a contradictory set is *not covered*, never *proved*), `:TestEngineReleasesABoundInput`, `:TestEngineRefusesInputsItCannotFree`, `:TestEngineRefusesAnAssumptionItCannotTranslate`, `:TestEngineWitnessWithInputsReplaysThroughTheStart`, `:TestEngineWitnessWithoutInputsReplaysAsBefore`; `smt/sensitivity_test.go:TestSensitivityOfForkBranchesWritingOneFeature` (`x` sensitive with both witnesses replayed to the oracle's two values, `leftRan` and `rightRan` *proved* not sensitive), `:TestSensitivityWritesBothWitnesses`, `:TestSensitivityShortOfCompletionIsBounded` (*bounded*, never *proved*, below the completing depth), `:TestSensitivityOfJoinWaitingForSlowestBranch` (`arrived` not sensitive), `:TestSensitivityUnderDeadlockIsTheDeadlock`, `:TestSensitivityRefusesTheClockAndPausedFlows` (the stage-4 row, refused naming the construct), `:TestSensitivityRefusesAPairTheInterpreterRefutes`, `:TestPairIsTheRelationTwiceOver` (copy B keeps every sort, declaration, assertion, objective, pin and flag of the query, renamed); `analysis/check_test.go:TestCheckKindAsksSensitiveForANamedFeature`, `:TestCheckAnswersASensitiveQuestionWithAPair`; `analysis/compose_test.go:TestComposeSensitivityWitnessStandsOverNotSensitive`; `cmd/sysml/smt_engine_test.go:TestEngineSMTDecidesSensitivity`, `:TestEngineAllComposesSensitivity`; `repl/smt_test.go:TestEngineSMTDecidesSensitivity`; `smt/discipline_test.go:TestMergeLoopBeyondTheBoundIsBounded`, `:TestBodyWhileBeyondUnrollingIsBounded`, `:TestDivisionByZeroIsReportedNotBounded`, `:TestPinnedMergeOutcomes`, `:TestEngineWithoutASolverIsAbsent`, `:TestEncodingEmitsOnlyPortableFeatures` (the two-copy query of every output among the queries checked); `smt/engine_test.go:TestEngineDescribesItself`, `:TestEngineRefusesWhatItDoesNotAnswer`, `:TestEngineDecidesConditions`, `:TestEngineDecidesDeadlock`, `:TestEngineRefusesWhatItDoesNotEncode`, `:TestEngineReportsUnknownAsNotCovered`, `:TestEngineWitnessesIntegerOverflow`; `smt/property_test.go:TestConditionPropertiesFollowTheRun`, `:TestDeadlockProperty`, `:TestConditionReadingNoValueIsUndefined`, `:TestConditionOverAnObjectIsRefused`; `smt/encode_test.go:TestEncodeForkJoinCompletes`, `:TestEncodePinsAndObjectFlows`, `:TestSortsNameEveryNodeEdgeAndSlot`; `smt/support_test.go:TestAnalyzeNumbersForkJoinFlow`, `:TestAnalyzeRecordsBodyLoops`, `:TestAnalyzeRefusesMessages`, `:TestAnalyzeRefusesNoInitial`; `solve/portability_test.go:TestPortability` (a datatype the query declares for the nodes of a flow; an incremental dialogue over named variables; a free input over an enumeration's datatype; a real-sorted input read back exactly); `runtime/replay_input_test.go:TestParseInputReadsEverySpelling`, `:TestParseWitnessReadsInputsBeforeChoices`, `:TestReplayFixesWitnessInputs` (a witness naming a feature the model lacks is refused naming it); `engines/engines_test.go:TestDefaultHoldsTheFrameworksEnginesAndSMT`, `:TestSMTStatusIsTheSolvers`, `:TestHoldsRanksSMTOverCheck`; `analysis/registry_test.go:TestDefaultHoldsTheFrameworksEngines`, `:TestDefaultPutsHoldsToCheckAlone`; `repl/smt_test.go:TestEngineSMTRangesOverAReleasedInput`, `:TestEngineSMTAssumesOverTheInitialState`, `:TestEngineAllShowsCheckRefusingFreeInputs`, `:TestEngineSMTRefusesAnUnknownInput`; `cmd/sysml/report_test.go:TestCheckResultsReportInputsAndWitnessValues`; `cmd/sysml/engines_test.go:TestEnginesListsTheBuild`; `grpc/engines_test.go:TestListEnginesNamesEveryEngine` | ✅ As designed for stages 1 to 3 (every solver test skips with a named reason without a solver and fails under `OPENSYSML_REQUIRE_SMT=1`; a *not sensitive* verdict is a fact about the model asked, not about the runtime's scheduling, so no ordering row moves on its account); clock, messages, nested flows, k-induction, heap and calc inlining, state machines and the engine option on the wire are later stages and refuse with the reason | diff --git a/docs/reference/cli.md b/docs/reference/cli.md index d3fefdb2cf..27289e3003 100644 --- a/docs/reference/cli.md +++ b/docs/reference/cli.md @@ -1510,8 +1510,9 @@ sweep built-in - observed sweep ready Every tool the manifest directory `OPENSYSML_TOOLS` names adds a `tool:` engine, listed the same way with its executable's status (`tool:ModelCenter tool object observed compute ready (ModelCenter 14.1 at /opt/modelcenter/bin/mc-batch)`); it answers the `compute` a -performance of an action annotated `ToolExecution` asks, and nothing else does, so a tool that -is unregistered or fails stops that performance rather than falling back to the action's body +performance of an action — or an invocation of a `calc def` or calc usage — annotated +`ToolExecution` asks, and nothing else does, so a tool that +is unregistered or fails stops that performance or calculation rather than falling back to the body ([External tools](environment.md#external-tools)). Every engine the manifest directory `OPENSYSML_ENGINES` names is listed under its own name, with the manifest's authority and answers and the status its file can tell; a `policy`, `sampler` or `module` entry is listed as diff --git a/docs/reference/environment.md b/docs/reference/environment.md index 8864420c8a..a1e43ac4bf 100644 --- a/docs/reference/environment.md +++ b/docs/reference/environment.md @@ -66,7 +66,11 @@ Nothing else in the toolchain reads these variables, and the concrete evaluator An action of an analysis case carrying the `AnalysisTooling::ToolExecution` metadata (its `toolName` and `uri`), with `ToolVariable` on the parameters the tool knows by other names, is -performed by that tool rather than by its body. The tools a `sysml` or `sysml-grpc` process may +performed by that tool rather than by its body. A `calc def` or calc usage carrying the same +metadata is computed the same way — its `in` parameters go to the tool, its result parameter +and `out` parameters are bound from the reply, and its body is never evaluated — wherever a +calc is invoked: `sysml -calc`, `%calc`, `EvaluateCalc`, a derived attribute (`attribute x = +toolCalc(a, b)`) and the formulas a rendered document evaluates. The tools a `sysml` or `sysml-grpc` process may run are the entries of the directory `OPENSYSML_TOOLS` names, read once at startup; each becomes an engine `tool:` that `-engines`, `%engines` and `ListEngines` list with its status, and a manifest that cannot be read is reported at startup, as a bad run bound is. @@ -255,6 +259,15 @@ The body is never run when the metadata is present: with `OPENSYSML_TOOLS` unset absent from it, the performance fails with `tool 'ModelCenter' is not registered; set OPENSYSML_TOOLS`. +A calc computed by a tool follows the same exchange: `inputs` are the `in` and `inout` +parameters carrying `ToolVariable`, `outputs` the `out` and `inout` parameters carrying it, +together with the result parameter — keyed by its `ToolVariable` name when it carries one, else +its declared name, else `result`. The result a `calc def` invocation returns is the value bound +under that key; a calc usage invoked without arguments reports every output it names. The same +typed errors, the same refusal to run the body and the same divergence reporting apply — the +calculation fails as the tool failed, and no value is invented for an output the tool did not +answer. + A tool's answer stands as the value of that performance at strength *observed*: nothing in OpenSysML knows what the tool should have computed. Two invocations with equal inputs answering different outputs are reported as a divergence in the run's notes (`%trace` summarizes them), so diff --git a/internal/doc/queryexec/derived.go b/internal/doc/queryexec/derived.go index 042aaa767e..0f9c50dbcb 100644 --- a/internal/doc/queryexec/derived.go +++ b/internal/doc/queryexec/derived.go @@ -25,7 +25,13 @@ type verificationKey struct { func (d *derivedValues) get(context Context) *runtime.DeclaredReader { if d.reader == nil { - d.reader = runtime.NewDeclaredReader(context.Model, context.Resolver) + // Over a held runtime, derived features calling a tool-computed calc + // answer through the same runner the query's context drives. + if context.Runtime != nil { + d.reader = runtime.NewDeclaredReaderIn(context.Runtime) + } else { + d.reader = runtime.NewDeclaredReader(context.Model, context.Resolver) + } } return d.reader } diff --git a/internal/exec/analysis/testdata/toolcalc/main.go b/internal/exec/analysis/testdata/toolcalc/main.go new file mode 100644 index 0000000000..26a6070b7f --- /dev/null +++ b/internal/exec/analysis/testdata/toolcalc/main.go @@ -0,0 +1,111 @@ +// Command toolcalc stands in for a tool-computed calc's external tool in the tests of +// the tool protocol: it reads the one JSON request on standard input and answers +// Tmax = mass*power in the unit TOOL_CALC_UNIT names (K by default), plus warn = +// mass*power > 100. TOOL_CALC_MODE picks a failure instead: error, exit or hang. +// It is built by the tests that run it and lives under testdata, out of every build. +package main + +import ( + "encoding/json" + "fmt" + "io" + "os" + "time" +) + +// ModeEnv selects how the stand-in answers; the empty mode answers the request. +const ModeEnv = "TOOL_CALC_MODE" + +// UnitEnv names the unit the Tmax answer is measured in. +const UnitEnv = "TOOL_CALC_UNIT" + +// value is one protocol value: a JSON value and, for a quantity, its unit. +type value struct { + Value json.RawMessage `json:"value"` + Unit string `json:"unit,omitempty"` +} + +// request is the protocol's request object. +type request struct { + ToolName string `json:"toolName"` + URI string `json:"uri"` + Inputs map[string]value `json:"inputs"` +} + +func main() { + if err := run(); err != nil { + fmt.Fprintln(os.Stderr, "toolcalc:", err) + os.Exit(3) + } +} + +func run() error { + data, err := io.ReadAll(os.Stdin) + if err != nil { + return err + } + var req request + if err := json.Unmarshal(data, &req); err != nil { + return fmt.Errorf("request is not the protocol's object: %w", err) + } + switch mode := os.Getenv(ModeEnv); mode { + case "", "answer": + return reply(answer(req)) + case "error": + return json.NewEncoder(os.Stdout).Encode(map[string]string{"error": "thermal model did not converge"}) + case "exit": + fmt.Fprintln(os.Stderr, "license server unreachable") + os.Exit(2) + case "hang": + time.Sleep(time.Minute) + return reply(answer(req)) + default: + return fmt.Errorf("%s=%q is not a mode", ModeEnv, mode) + } + return nil +} + +// number reads one input's magnitude as the protocol carries it. +func number(in value) (float64, error) { + var n float64 + if err := json.Unmarshal(in.Value, &n); err != nil { + return 0, fmt.Errorf("input %s is not a number: %w", in.Value, err) + } + return n, nil +} + +// answer is the stand-in's computation: the product of the mass and power sent. +func answer(req request) map[string]value { + mass, err := number(req.Inputs["mass"]) + if err != nil { + mass = 0 + } + power, err := number(req.Inputs["power"]) + if err != nil { + power = 0 + } + product := mass * power + unit := os.Getenv(UnitEnv) + if unit == "" { + unit = "K" + } + return map[string]value{ + "Tmax": {Value: raw(product), Unit: unit}, + "warn": {Value: raw(product > 100)}, + } +} + +func raw(x any) json.RawMessage { + data, err := json.Marshal(x) + if err != nil { + return json.RawMessage("null") + } + return data +} + +// reply writes the protocol's reply object. +func reply(outputs map[string]value) error { + return json.NewEncoder(os.Stdout).Encode(struct { + Outputs map[string]value `json:"outputs"` + }{Outputs: outputs}) +} diff --git a/internal/exec/analysis/tool_calc_test.go b/internal/exec/analysis/tool_calc_test.go new file mode 100644 index 0000000000..37a87a7d7b --- /dev/null +++ b/internal/exec/analysis/tool_calc_test.go @@ -0,0 +1,237 @@ +package analysis + +import ( + "context" + "errors" + "fmt" + "os" + "os/exec" + "path/filepath" + "strings" + "sync" + "testing" + + "github.com/Open-MBEE/OpenSysML/internal/check/passes" + "github.com/Open-MBEE/OpenSysML/internal/exec/runtime" + "github.com/Open-MBEE/OpenSysML/internal/semantic/resolve" + "github.com/Open-MBEE/OpenSysML/internal/semantic/symbols" + "github.com/Open-MBEE/OpenSysML/internal/syntax/parser" + "github.com/Open-MBEE/OpenSysML/internal/syntax/source" + "github.com/Open-MBEE/OpenSysML/internal/workspace/libs" + "github.com/Open-MBEE/OpenSysML/tests/testutil/gobuild" +) + +// toolCalcEnv are the stand-in's own variables, from testdata/toolcalc. +const ( + toolCalcMode = "TOOL_CALC_MODE" + toolCalcUnit = "TOOL_CALC_UNIT" +) + +// toolCalcDriver is a calc def computed by the stand-in and a usage of it. +const toolCalcDriver = `package Calc { + private import AnalysisTooling::*; + private import ScalarValues::*; + private import ISQ::*; + + calc def Thermal { + metadata ToolExecution { toolName = "Thermo"; uri = "thermo://x"; } + in m : MassValue { @ToolVariable { name = "mass"; } } + in p : PowerValue { @ToolVariable { name = "power"; } } + out warn : Boolean { @ToolVariable { name = "warn"; } } + return : TemperatureValue { @ToolVariable { name = "Tmax"; } } + } + + calc t : Thermal { + in m = 2 [SI::kg]; + in p = 10 [SI::W]; + } +}` + +var ( + toolCalcOnce sync.Once + toolCalcBinPath string + toolCalcBinErr error +) + +// toolcalc builds the calc stand-in once per test binary and returns its path. +func toolcalc(t *testing.T) string { + t.Helper() + toolCalcOnce.Do(func() { + dir, err := os.MkdirTemp("", "toolcalc") + if err != nil { + toolCalcBinErr = err + return + } + toolCalcBinPath = filepath.Join(dir, "toolcalc") + build := exec.Command("go", gobuild.Args(toolCalcBinPath)...) + build.Dir = filepath.Join("testdata", "toolcalc") + if out, err := build.CombinedOutput(); err != nil { + toolCalcBinErr = fmt.Errorf("go build: %v\n%s", err, out) + } + }) + if toolCalcBinErr != nil { + t.Fatalf("building the calc stand-in: %v", toolCalcBinErr) + } + return toolCalcBinPath +} + +// thermoEntry is the manifest entry the driver's ToolExecution resolves to. +func thermoEntry(executable string) ToolEntry { + return ToolEntry{ToolName: "Thermo", Version: "stand-in", Executable: executable, + Variables: []string{"mass", "power", "warn", "Tmax"}} +} + +// cprobe is the calc driver indexed over the standard libraries. +type cprobe struct { + idx *symbols.Index + pkg *symbols.Scope +} + +func parseCProbe(t *testing.T) *cprobe { + t.Helper() + idx := libs.NewModelIndex() + p := parser.New(source.New("cprobe.sysml", []byte(toolCalcDriver))) + file := p.ParseFile() + if len(p.Diagnostics) > 0 { + t.Fatalf("parse: %v", p.Diagnostics) + } + idx.AddDocument("cprobe.sysml", file) + idx.ExpandWildcardImports() + pkg, ok := idx.DocumentRoot("cprobe.sysml").LookupLocal("Calc") + if !ok || pkg.Scope == nil { + t.Fatal("Calc package not indexed") + } + return &cprobe{idx: idx, pkg: pkg.Scope} +} + +func (p *cprobe) context() *runtime.Context { + resolver := resolve.New(p.idx) + model := runtime.NewModel(passes.NewTypedModel(resolver), resolver) + model.SetExpressionParser(parser.ParseOneExpression) + return runtime.NewContext(model, fixtureSteps) +} + +func (p *cprobe) symbol(t *testing.T, name string) *symbols.Symbol { + t.Helper() + sym, ok := p.pkg.LookupLocal(name) + if !ok { + t.Fatalf("%s not indexed", name) + } + return sym +} + +// evalText evaluates one expression in the driver's scope, for invocation arguments. +func (p *cprobe) evalText(t *testing.T, ctx *runtime.Context, text string) runtime.Value { + t.Helper() + expr, ok := parser.ParseOneExpression("cprobe.sysml", text) + if !ok { + t.Fatalf("parsing %q", text) + } + value, err := runtime.NewEvalContextIn(ctx, p.pkg, nil).Eval(expr) + if err != nil { + t.Fatalf("evaluating %q: %v", text, err) + } + return value +} + +// tooled attaches the registry's runner to the context, as a surface does. +func (p *cprobe) tooled(r *Registry, ctx *runtime.Context) { + ctx.SetToolRunner(r.ToolRunner(context.Background(), ctx, Budget{}, Auto())) +} + +// invoke puts one calculation of the driver's Thermal to the stand-in. +func (p *cprobe) invoke(t *testing.T, r *Registry) (runtime.Value, error) { + t.Helper() + ctx := p.context() + p.tooled(r, ctx) + args := []runtime.Value{p.evalText(t, ctx, "2 [SI::kg]"), p.evalText(t, ctx, "10 [SI::W]")} + return ctx.InvokeCalc(p.symbol(t, "Thermal"), args, p.pkg) +} + +// The registry's runner attached to a held context computes the annotated calc: the +// result is the stand-in's answer converted to the result parameter's coherent unit. +func TestToolCalcThroughTheRegistryRunner(t *testing.T) { + p := parseCProbe(t) + t.Setenv(ToolEnvPassthroughEnv, toolCalcUnit) + t.Setenv(toolCalcUnit, "K") + r := toolRegistry(t, manifestDir(t, thermoEntry(toolcalc(t)))) + + result, err := p.invoke(t, r) + if err != nil { + t.Fatalf("InvokeCalc: %v", err) + } + if got := runtime.FormatValue(result); got != "20 [SI::K]" { + t.Fatalf("result = %s, want 20 [SI::K]", got) + } +} + +// Every way the stand-in fails is the typed error of its kind: the refusal it +// states, its process's exit, its silence past the timeout. +func TestToolCalcFailsWithTheToolsFault(t *testing.T) { + cases := []struct { + mode string + kind runtime.ToolErrorKind + text string + }{ + {"error", runtime.ToolRefused, "thermal model did not converge"}, + {"exit", runtime.ToolProcessFailed, "license server unreachable"}, + {"hang", runtime.ToolTimeout, ToolTimeoutEnv}, + } + p := parseCProbe(t) + for _, tc := range cases { + t.Run(tc.mode, func(t *testing.T) { + t.Setenv(ToolEnvPassthroughEnv, toolCalcMode) + t.Setenv(toolCalcMode, tc.mode) + if tc.mode == "hang" { + t.Setenv(ToolTimeoutEnv, "200ms") + } + r := toolRegistry(t, manifestDir(t, thermoEntry(toolcalc(t)))) + _, err := p.invoke(t, r) + var fault *runtime.ToolError + if !errors.As(err, &fault) || fault.Kind != tc.kind || fault.Tool != "Thermo" { + t.Fatalf("InvokeCalc = %v, want a ToolError of kind %s", err, tc.kind) + } + if !strings.Contains(err.Error(), tc.text) { + t.Errorf("error %q does not carry %q", err, tc.text) + } + }) + } +} + +// A manifest naming another tool leaves Thermo unregistered: the calculation is +// refused and no answer is invented. +func TestToolCalcRefusesAnUnregisteredTool(t *testing.T) { + p := parseCProbe(t) + other := ToolEntry{ToolName: "Other", Executable: toolcalc(t), Variables: []string{"mass"}} + r := toolRegistry(t, manifestDir(t, other)) + _, err := p.invoke(t, r) + var refusal *runtime.ToolNotRegisteredError + if !errors.As(err, &refusal) || refusal.Tool != "Thermo" { + t.Fatalf("InvokeCalc = %v, want ToolNotRegisteredError", err) + } + if want := "tool 'Thermo' is not registered; set OPENSYSML_TOOLS"; !strings.Contains(err.Error(), want) { + t.Errorf("error %q does not carry %q", err, want) + } +} + +// A usage of the annotated calc, read as a feature, computes through the tool as +// the invocation does: its inputs bind from its own member values. +func TestToolCalcUsageThroughTheRegistryRunner(t *testing.T) { + p := parseCProbe(t) + t.Setenv(ToolEnvPassthroughEnv, toolCalcUnit) + r := toolRegistry(t, manifestDir(t, thermoEntry(toolcalc(t)))) + ctx := p.context() + p.tooled(r, ctx) + + outputs, err := ctx.CalcUsageOutputs(p.symbol(t, "t"), p.pkg, nil) + if err != nil { + t.Fatalf("CalcUsageOutputs: %v", err) + } + got := map[string]string{} + for _, out := range outputs { + got[out.Name] = runtime.FormatValue(out.Value) + } + if got["warn"] != "false" { + t.Fatalf("outputs %v, want warn = false", got) + } +} diff --git a/internal/exec/runtime/calc_usage.go b/internal/exec/runtime/calc_usage.go index 676f139a12..57dc3b79e7 100644 --- a/internal/exec/runtime/calc_usage.go +++ b/internal/exec/runtime/calc_usage.go @@ -888,6 +888,31 @@ func (ctx *Context) runCalcUsage(start *calcUsageStart) (*calcRun, error) { ctx.clock.attach(flow) defer ctx.clock.detach(flow) } + run := newCalcRun(shape, reader.scope, reader.self, env) + run.outer = nested + run.activation, run.perf, run.boundInputs = engine.activation, host.performance(), start.inputs + + // A tool-computed calc runs no body: the tool's answers are its outputs, the + // result parameter's answer its result, and reading an unanswered one is the + // reply's failure rather than a binding to evaluate. + if shape.Tool != nil { + result, outputs, err := ctx.computeCalcByTool(shape, ec.scope, env.lookup) + if err != nil { + if ec.trace != nil { + ec.trace.RecordCalculationExitError(shape.Kind, shape.Name, err) + } + return nil, calcFrame(shape.Kind, shape.Name, err) + } + run.result, run.returned = result, true + for name, value := range outputs { + run.outputs[name] = value + } + if ec.trace != nil { + ec.trace.RecordCalculationExit(shape.Kind, shape.Name, result) + } + return run, nil + } + steps := shape.Steps if start.deferResults { steps, _ = shape.observationSteps() @@ -910,9 +935,7 @@ func (ctx *Context) runCalcUsage(start *calcUsageStart) (*calcRun, error) { } } - run := newCalcRun(shape, reader.scope, reader.self, env) - run.outer, run.result, run.returned = nested, result, returned - run.activation, run.perf, run.boundInputs = engine.activation, host.performance(), start.inputs + run.result, run.returned = result, returned // The returned value is the result parameter's, read under its name or as // `result`; every other output states its own value, never the returned one. if returned { @@ -956,6 +979,12 @@ func (run *calcRun) value(ctx *Context, out calcOutput) (Value, error) { if value, ok := run.outputs[out.Name]; ok && out.Name != "" { return value, nil } + if run.shape.Tool != nil { + // The tool's reply is the output's whole computation: one it left + // unanswered has no binding to fall back on. + return Value{}, &ToolError{Tool: run.shape.Tool.tool, Kind: ToolMissingOutput, + Detail: fmt.Sprintf("%s (%s of %s) was not answered", out.Name, run.outputDescription(out), run.shape.Label)} + } if out.Value == nil || out.IsInitial { // An output the body assigned, and an `inout` the invocation bound, are values // the activation left behind rather than bindings to evaluate. diff --git a/internal/exec/runtime/declared.go b/internal/exec/runtime/declared.go index 48bc3d6c36..8bbf114d16 100644 --- a/internal/exec/runtime/declared.go +++ b/internal/exec/runtime/declared.go @@ -25,6 +25,17 @@ func NewDeclaredReader(model *semantics.Model, resolver *resolve.Resolver) *Decl return &DeclaredReader{ctx: ctx, objects: make(map[*symbols.Symbol]*Instance)} } +// NewDeclaredReaderIn creates a reader on a fresh declarative context seeded +// from held: its tool runner, so tool-computed calcs a derived feature calls +// answer through the same runner, and the parser reading the units they +// answer in. +func NewDeclaredReaderIn(held *Context) *DeclaredReader { + r := NewDeclaredReader(held.Semantics(), held.Resolver()) + r.ctx.model.parse = held.model.parse + r.ctx.SetToolRunner(held.ToolRunner()) + return r +} + // Read evaluates the named feature of the element sym denotes, resolving every // leaf through that element's redefinitions. An unbound leaf is a *NoValueError. func (r *DeclaredReader) Read(sym *symbols.Symbol, name string) (Value, error) { diff --git a/internal/exec/runtime/invoke_calc.go b/internal/exec/runtime/invoke_calc.go index c324e3a0c1..8a17604900 100644 --- a/internal/exec/runtime/invoke_calc.go +++ b/internal/exec/runtime/invoke_calc.go @@ -178,6 +178,9 @@ type calcShape struct { // Uncomputed says why the calc computes nothing (no body, no bound output); // the calc is still a function value, and invoking it reports this. Uncomputed error + // Tool is the ToolExecution the calc carries, when it is computed by an + // external tool rather than by a body it need not state. + Tool *toolExecution // compiled is the body in the compiled tier once compileState says it is // eligible; a shape found ineligible keeps the evaluator for good. compiled *compiledCalc @@ -253,6 +256,11 @@ func (ctx *Context) calcInterfaceOf(sym *symbols.Symbol) (*calcShape, error) { shape.BodyOutputs = assignedOutputs(shape.Steps, shape.Outputs, shape.Aliases) shape.Bindings = calcBindings(chain) shape.ResultExpr = resultBindingExpr(shape.Bindings) + if tool, err := ctx.toolExecutionOf(sym); err != nil { + return nil, err + } else { + shape.Tool = tool + } // A calc computes nothing unless it returns or binds an output; a case also // computes through its steps, or answers with its verdicts alone, and a // library function the runtime implements natively computes through that. @@ -260,7 +268,7 @@ func (ctx *Context) calcInterfaceOf(sym *symbols.Symbol) (*calcShape, error) { _, native := ctx.libraryFunctionFor(sym) computes := lower.Returns(shape.Body) || len(shape.BodyOutputs) > 0 || shape.ResultExpr != nil || shape.hasInitialOutput() || performs || native switch { - case computes: + case computes || shape.Tool != nil: case len(shape.Outputs) > 0 && shape.resultOutput() == nil: shape.Uncomputed = fmt.Errorf("%w: %s binds none of its outputs (%s)", ErrNoResultExpression, label, shape.outputNames()) @@ -646,7 +654,7 @@ func (ctx *Context) invokeCalcShapeIn(shape *calcShape, args calcArgs, callerSco // sub-expression, an argument is not a scalar, a bound object may answer // a library constant the body reads before the library does, or the body // reads the bindings enclosing it. - if ctx.compileCalcs && ctx.trace == nil && len(enclosing) == 0 { + if shape.Tool == nil && ctx.compileCalcs && ctx.trace == nil && len(enclosing) == 0 { if compiled := ctx.compiledCalcOf(shape); compiled != nil && (self == nil || !compiled.readsLibrary) { if result, ran, err := compiled.invokeBoxed(ctx, args); ran { return result, err @@ -691,7 +699,13 @@ func (ctx *Context) invokeCalcShapeIn(shape *calcShape, args calcArgs, callerSco return Value{}, err } - result, err := ctx.runCalcBody(shape, frame, callerScope, self, activation, enclosing) + var result Value + var err error + if shape.Tool != nil { + result, _, err = ctx.computeCalcByTool(shape, ec.scope, locals.lookup) + } else { + result, err = ctx.runCalcBody(shape, frame, callerScope, self, activation, enclosing) + } if ec.trace != nil { if err != nil { ec.trace.RecordCalculationExitError(shape.Kind, shape.Name, err) diff --git a/internal/exec/runtime/robustness_toolcalc_test.go b/internal/exec/runtime/robustness_toolcalc_test.go new file mode 100644 index 0000000000..a7907d0c58 --- /dev/null +++ b/internal/exec/runtime/robustness_toolcalc_test.go @@ -0,0 +1,112 @@ +package runtime + +import ( + "errors" + "testing" +) + +// TestRuntimeRobustnessToolCalc exercises the failure modes of a calc annotated +// ToolExecution: every one is the typed error of its kind, never a panic, a hang +// or a body the calc was not to run. +func TestRuntimeRobustnessToolCalc(t *testing.T) { + t.Run("no_runner", func(t *testing.T) { + ctx, scope := analysisFixture(t, toolCalcModel) + _, err := ctx.InvokeCalc(calcNamed(t, scope, "Thermal"), thermalArgs(t, ctx, scope), scope) + if !errors.Is(err, ErrToolNotRegistered) { + t.Fatalf("InvokeCalc = %v, want ErrToolNotRegistered", err) + } + }) + t.Run("no_runner_usage", func(t *testing.T) { + ctx, scope := analysisFixture(t, toolCalcModel) + _, err := ctx.CalcUsageOutputs(calcNamed(t, scope, "t"), scope, nil) + if !errors.Is(err, ErrToolNotRegistered) { + t.Fatalf("CalcUsageOutputs = %v, want ErrToolNotRegistered", err) + } + }) + t.Run("runner_fault_usage", func(t *testing.T) { + ctx, scope := analysisFixture(t, toolCalcModel) + fault := &ToolError{Tool: "Thermo", Kind: ToolProcessFailed, Detail: "exit 2"} + ctx.SetToolRunner(&recordingRunner{err: fault}) + _, err := ctx.CalcUsageOutputs(calcNamed(t, scope, "t"), scope, nil) + if !errors.Is(err, fault) { + t.Fatalf("CalcUsageOutputs = %v, want %v", err, fault) + } + }) + t.Run("unbound_input", func(t *testing.T) { + // A required input no argument and no usage binding supplies is + // ErrUnboundParameter, as a body-run calc reports it, and no tool runs. + ctx, scope := analysisFixture(t, toolCalcRobustnessModel) + runner := &recordingRunner{answer: map[string]ToolValue{}} + ctx.SetToolRunner(runner) + _, err := ctx.CalcUsageOutputs(calcNamed(t, scope, "tbare"), scope, nil) + if !errors.Is(err, ErrUnboundParameter) { + t.Fatalf("CalcUsageOutputs = %v, want ErrUnboundParameter", err) + } + if len(runner.calls) != 0 { + t.Fatalf("tool ran %d times for an unbound input", len(runner.calls)) + } + }) + t.Run("ambiguous_variable", func(t *testing.T) { + ctx, scope := analysisFixture(t, toolCalcRobustnessModel) + runner := &recordingRunner{answer: map[string]ToolValue{"y": {Value: toolReal(1)}}} + ctx.SetToolRunner(runner) + _, err := ctx.InvokeCalc(calcNamed(t, scope, "Colliding"), []Value{realOf(1), realOf(2)}, scope) + var failure *ToolError + if !errors.As(err, &failure) || failure.Kind != ToolAmbiguousVariable { + t.Fatalf("InvokeCalc = %v, want ToolAmbiguousVariable", err) + } + if len(runner.calls) != 0 { + t.Fatalf("tool ran %d times under an ambiguous variable", len(runner.calls)) + } + }) + t.Run("unanswered_output", func(t *testing.T) { + // A reply the tool bound before the run ended reaches no binding: the + // output read asks the run, which answers the reply's failure as a typed + // error rather than evaluating the calc's bindings. + ctx, scope := analysisFixture(t, toolCalcRobustnessModel) + ctx.SetToolRunner(dodgyCalcRunner{}) + _, err := ctx.CalcUsageOutput(calcNamed(t, scope, "tw"), "warn", scope, nil) + var failure *ToolError + if !errors.As(err, &failure) || failure.Kind != ToolMissingOutput { + t.Fatalf("CalcUsageOutput = %v, want ToolMissingOutput", err) + } + }) +} + +// dodgyCalcRunner answers the call's outputs except one, bypassing Bind's check, +// as a runner is free to do. +type dodgyCalcRunner struct{} + +func (dodgyCalcRunner) RunTool(call *ToolCall) (ToolAnswer, error) { + return ToolAnswer{Outputs: map[string]Value{"result": {Kind: ValConst, Const: toolReal(1)}}}, nil +} + +// toolCalcRobustnessModel is the failure-mode calcs: one whose usage binds no +// input, one whose inputs share a ToolVariable name, and one a runner can leave +// an output of unanswered. +const toolCalcRobustnessModel = `package test { + private import ScalarValues::*; + private import AnalysisTooling::*; + private import ISQ::*; + + calc def Needing { + metadata ToolExecution { toolName = "Thermo"; uri = "u"; } + in m : MassValue { @ToolVariable { name = "mass"; } } + return : TemperatureValue { @ToolVariable { name = "Tmax"; } } + } + calc tbare : Needing; + + calc def Colliding { + metadata ToolExecution { toolName = "Thermo"; uri = "u"; } + in a : Real { @ToolVariable { name = "x"; } } + in b : Real { @ToolVariable { name = "x"; } } + out y : Real { @ToolVariable { name = "y"; } } + } + + calc def Tw { + metadata ToolExecution { toolName = "Thermo"; uri = "u"; } + out warn : Boolean { @ToolVariable { name = "warn"; } } + return : Real { @ToolVariable { name = "r"; } } + } + calc tw : Tw; +}` diff --git a/internal/exec/runtime/tool.go b/internal/exec/runtime/tool.go index 05d3d412ff..65e5f38044 100644 --- a/internal/exec/runtime/tool.go +++ b/internal/exec/runtime/tool.go @@ -23,18 +23,20 @@ const ( toolVariableName = "name" ) -// ToolCall is one performance of an action annotated ToolExecution, as the external tool -// sees it: the tool and URI the metadata names, the inputs read from the run keyed by -// their ToolVariable names, and the outputs the tool is to answer. +// ToolCall is one performance or calculation of an action or calc annotated +// ToolExecution, as the external tool sees it: the tool and URI the metadata names, +// the inputs read from the run keyed by their ToolVariable names, and the outputs +// the tool is to answer. type ToolCall struct { - // Action is the annotated action definition or usage performed. + // Action is the annotated action or calc the tool computes. Action *symbols.Symbol ToolName string URI string Inputs []ToolInput Outputs []ToolOutput - exec *ActionExecutor + ctx *Context + scope *symbols.Scope } // ToolInput is one `in` or `inout` parameter's value, under its ToolVariable name. @@ -67,9 +69,9 @@ type ToolAnswer struct { Diverged bool } -// ToolRunner runs the tool a ToolExecution names for one performance of the action, +// ToolRunner runs the tool a ToolExecution names for one performance or calculation, // binding the tool's outputs through ToolCall.Bind. A context with no runner attached -// refuses every tool-computed action as not registered. +// refuses every tool-computed action or calc as not registered. type ToolRunner interface { RunTool(call *ToolCall) (ToolAnswer, error) } @@ -157,8 +159,8 @@ func (e *ToolError) Is(target error) bool { return target == ErrTool } // ToolDivergenceCode is the diagnostic code of a ToolDivergence note. const ToolDivergenceCode = "tool-divergence" -// ToolDivergence is a tool answering two invocations with equal inputs differently: the -// outcome table over it is not reproducible, and the run says so without changing. +// ToolDivergence is a tool answering two performances or calculations with equal inputs +// differently: the outcome table over it is not reproducible, and the run says so without changing. type ToolDivergence struct { Tool string Action string @@ -176,7 +178,7 @@ func (d ToolDivergence) String() string { return "tool divergence: " + d.Describe() } -// Location is the annotated action's declaration. +// Location is the annotated action's or calc's declaration. func (d ToolDivergence) Location() (string, source.Span) { return d.File, d.Span } @@ -193,7 +195,8 @@ func (d ToolDivergence) Diagnostic() diag.Diagnostic { } } -// SetToolRunner attaches the runner tool-computed actions of this context's runs invoke. +// SetToolRunner attaches the runner tool-computed actions and calcs of this +// context's runs invoke. func (ctx *Context) SetToolRunner(runner ToolRunner) { ctx.tools = runner } @@ -203,15 +206,16 @@ func (ctx *Context) ToolRunner() ToolRunner { return ctx.tools } -// toolExecution is the ToolExecution an action carries, as its performances read it: -// the tool and URI its bindings state, and the action or supertype annotated. +// toolExecution is the ToolExecution an action or calc carries, as its performances +// read it: the tool and URI its bindings state, and the element or supertype annotated. type toolExecution struct { tool, uri string on *symbols.Symbol } -// toolExecutionOf reads the ToolExecution annotating an action, or the definition it is -// typed by. Nil for an action carrying none; an annotation is one whatever its toolName. +// toolExecutionOf reads the ToolExecution annotating an action or calc, or the +// definition it is typed by. Nil for one carrying none; an annotation is one +// whatever its toolName. func (ctx *Context) toolExecutionOf(action *symbols.Symbol) (*toolExecution, error) { if ctx.model == nil || action == nil { return nil, nil @@ -370,7 +374,7 @@ func (e *ActionExecutor) performByTool(execution *toolExecution) error { // binds (e.action) names and types each parameter, and the performance holds it under that name. func (e *ActionExecutor) toolCall(execution *toolExecution) (*ToolCall, error) { tool := execution.tool - call := &ToolCall{Action: execution.on, ToolName: tool, URI: execution.uri, exec: e} + call := &ToolCall{Action: execution.on, ToolName: tool, URI: execution.uri, ctx: e.ctx, scope: e.root.scope} namedBy := make(map[string]string) for _, param := range e.ctx.model.semantics.BehaviorParametersOf(e.action) { if param.Symbol == nil || param.Symbol.Name == "" { @@ -523,7 +527,7 @@ func (c *ToolCall) Bind(outputs map[string]ToolValue) (map[string]Value, error) return nil, &ToolError{Tool: c.ToolName, Kind: ToolMissingOutput, Detail: fmt.Sprintf("%s (%s of %s) was not answered", out.Variable, out.Parameter, symbolText(c.Action))} } - value, err := c.exec.toolOutput(c.ToolName, out, answered) + value, err := c.ctx.toolOutput(c.scope, c.ToolName, out, answered) if err != nil { return nil, err } @@ -532,22 +536,23 @@ func (c *ToolCall) Bind(outputs map[string]ToolValue) (map[string]Value, error) return bound, nil } -// toolOutput reads one answered value as the parameter's: a string or bare number as is, +// toolOutput reads one answered value as the parameter's, the unit spellings read in +// scope: a string or bare number as is, // a quantity converted to the coherent unit of the parameter's declared quantity kind, // spelt as the declared type prefers. A unit is refused unless the parameter is a quantity, // and a value the parameter's declaration cannot hold is malformed. -func (e *ActionExecutor) toolOutput(tool string, out ToolOutput, answered ToolValue) (Value, error) { +func (ctx *Context) toolOutput(scope *symbols.Scope, tool string, out ToolOutput, answered ToolValue) (Value, error) { malformed := func(format string, args ...any) error { return &ToolError{Tool: tool, Kind: ToolMalformed, Detail: out.Variable + ": " + fmt.Sprintf(format, args...)} } - value, err := e.toolOutputValue(malformed, out, answered) + value, err := ctx.toolOutputValue(scope, malformed, out, answered) if err != nil { return Value{}, err } - mult, _ := e.ctx.extractMultiplicity(out.Declared) - target := &writeTarget{name: out.Parameter, typ: e.ctx.extractType(out.Declared), mult: mult} - if err := e.ctx.checkWrite(e.ctx.protocolScope(e.root.scope), out.Parameter, target, &value); err != nil { + mult, _ := ctx.extractMultiplicity(out.Declared) + target := &writeTarget{name: out.Parameter, typ: ctx.extractType(out.Declared), mult: mult} + if err := ctx.checkWrite(ctx.protocolScope(scope), out.Parameter, target, &value); err != nil { return Value{}, malformed("%v", err) } return value, nil @@ -564,7 +569,7 @@ func (ctx *Context) protocolScope(fallback *symbols.Scope) *symbols.Scope { // toolOutputValue converts one answered value to the run's, by its unit and the // parameter's declared quantity kind; malformed builds the refusal of one that cannot be. -func (e *ActionExecutor) toolOutputValue(malformed func(string, ...any) error, out ToolOutput, answered ToolValue) (Value, error) { +func (ctx *Context) toolOutputValue(scope *symbols.Scope, malformed func(string, ...any) error, out ToolOutput, answered ToolValue) (Value, error) { if answered.Value.Kind == semantics.ValInvalid { if answered.Unit != "" { return Value{}, malformed("text %q is measured in %s", answered.Text, answered.Unit) @@ -577,7 +582,7 @@ func (e *ActionExecutor) toolOutputValue(malformed func(string, ...any) error, o if !answered.Value.IsNumeric() { return Value{}, malformed("a truth is measured in %s", answered.Unit) } - unit, err := e.toolUnit(answered.Unit) + unit, err := ctx.UnitOf(scope, answered.Unit) switch { case errors.Is(err, ErrNoExpressionParser): return Value{}, err @@ -585,14 +590,14 @@ func (e *ActionExecutor) toolOutputValue(malformed func(string, ...any) error, o return Value{}, malformed("%v", err) } q := Quantity{Num: answered.Value, Unit: unit} - dim, ok := e.ctx.model.semantics.DimensionOfFeature(out.Declared) + dim, ok := ctx.model.semantics.DimensionOfFeature(out.Declared) if !ok { - if !e.ctx.quantityTyped(out.Declared) { + if !ctx.quantityTyped(out.Declared) { return Value{}, malformed("%s is not a quantity to be measured in %s", out.Parameter, answered.Unit) } return quantityResult(q, nil) } - coherent, ok := e.ctx.model.semantics.CoherentUnitFor(dim, out.Declared) + coherent, ok := ctx.model.semantics.CoherentUnitFor(dim, out.Declared) if !ok { return NewQuantityValue(&q), nil } @@ -624,13 +629,6 @@ type toolUnitKey struct { text string } -// toolUnit reads a unit the protocol spells, in the action's scope, else in the library's -// SI package so a tool's `m/s**2` reads whatever the model imports; the reading is -// memoized per scope since resolution memoizes per parsed name. -func (e *ActionExecutor) toolUnit(text string) (semantics.Unit, error) { - return e.ctx.UnitOf(e.root.scope, text) -} - // UnitOf reads a unit spelled as expression text in scope, else in the library's SI // package, which a nil scope reads alone; the reading is memoized per scope. func (ctx *Context) UnitOf(scope *symbols.Scope, text string) (semantics.Unit, error) { diff --git a/internal/exec/runtime/tool_calc.go b/internal/exec/runtime/tool_calc.go new file mode 100644 index 0000000000..fa96a0b11f --- /dev/null +++ b/internal/exec/runtime/tool_calc.go @@ -0,0 +1,125 @@ +package runtime + +import ( + "fmt" + "sort" + + "github.com/Open-MBEE/OpenSysML/internal/semantic/symbols" + "github.com/Open-MBEE/OpenSysML/internal/syntax/ast" +) + +// A calc annotated ToolExecution is computed by the tool the metadata names, exactly as an +// annotated action is performed by it: the calc's `in`/`inout` parameters carrying a +// ToolVariable are sent, its `out`/`inout` ones and its result parameter are bound from +// the reply, and the body — when the calc states one — is never evaluated. + +// calcToolCall is the calculation as the tool sees it: of the calc the annotation holds +// for (the definition or usage annotated), every `in`/`inout` parameter carrying a +// ToolVariable is an input whose value held answers, every `out`/`inout` one an output, +// and the result parameter is always an output — keyed by its ToolVariable name, else its +// declared name, else `result`. The same refusals as a performance's call apply: an +// unbound non-optional input is ErrUnboundParameter, a ToolVariable name two parameters +// carry is a ToolError. The name the result is bound under is returned with the call. +func (ctx *Context) calcToolCall(shape *calcShape, scope *symbols.Scope, held func(param string) (Value, bool)) (*ToolCall, string, error) { + execution := shape.Tool + tool := execution.tool + call := &ToolCall{Action: execution.on, ToolName: tool, URI: execution.uri, ctx: ctx, scope: scope} + namedBy := make(map[string]string) + var result *symbols.Symbol + for _, param := range ctx.model.semantics.BehaviorParametersOf(shape.Sym) { + if param.Symbol == nil { + continue + } + if param.IsResult { + result = param.Symbol + continue + } + if param.Symbol.Name == "" { + continue + } + variable, named, err := ctx.toolVariableOf(param.Symbol) + if err != nil { + return nil, "", err + } + if !named { + continue + } + if other, taken := namedBy[variable]; taken { + return nil, "", &ToolError{Tool: tool, Kind: ToolAmbiguousVariable, + Detail: fmt.Sprintf("%s names both %s and %s of %s", variable, other, param.Symbol.Name, shape.Label)} + } + namedBy[variable] = param.Symbol.Name + name := param.Symbol.Name + reads := param.Direction == ast.DirIn || param.Direction == ast.DirInOut + writes := param.Direction == ast.DirOut || param.Direction == ast.DirInOut + if reads { + value, bound := held(name) + if !bound && !ctx.model.semantics.OptionalParameter(param.Symbol) { + return nil, "", fmt.Errorf("%w: %s: input parameter %s is bound by no argument", + ErrUnboundParameter, shape.Label, name) + } + if bound { + sent, err := toolInput(tool, param.Symbol, value) + if err != nil { + return nil, "", err + } + call.Inputs = append(call.Inputs, ToolInput{Variable: variable, Parameter: name, Value: sent}) + } + } + if writes { + call.Outputs = append(call.Outputs, ToolOutput{Variable: variable, Parameter: name, Declared: param.Symbol}) + } + } + resultKey := resultOutputName + if result != nil { + variable := resultOutputName + if result.Name != "" { + resultKey, variable = result.Name, result.Name + } + if named, has, err := ctx.toolVariableOf(result); err != nil { + return nil, "", err + } else if has { + variable = named + } + if other, taken := namedBy[variable]; taken { + return nil, "", &ToolError{Tool: tool, Kind: ToolAmbiguousVariable, + Detail: fmt.Sprintf("%s names both %s and %s of %s", variable, other, resultKey, shape.Label)} + } + call.Outputs = append(call.Outputs, ToolOutput{Variable: variable, Parameter: resultKey, Declared: result}) + } + sort.Slice(call.Inputs, func(i, j int) bool { return call.Inputs[i].Variable < call.Inputs[j].Variable }) + sort.Slice(call.Outputs, func(i, j int) bool { return call.Outputs[i].Variable < call.Outputs[j].Variable }) + return call, resultKey, nil +} + +// computeCalcByTool computes a tool-annotated calc: the tool shape.Tool names is invoked +// once with the inputs held answers, its outputs stand as the calc's outputs, and the +// result parameter's value is the calculation's result. The failures of a performance's +// tool apply unchanged: an annotation naming no tool, or a context with no runner, is +// not registered and the body never stands in. +func (ctx *Context) computeCalcByTool(shape *calcShape, scope *symbols.Scope, held func(param string) (Value, bool)) (Value, map[string]Value, error) { + tool := shape.Tool.tool + if tool == "" { + return Value{}, nil, &ToolNotRegisteredError{Tool: tool} + } + call, resultKey, err := ctx.calcToolCall(shape, scope, held) + if err != nil { + return Value{}, nil, err + } + if ctx.tools == nil { + return Value{}, nil, &ToolNotRegisteredError{Tool: tool} + } + answer, err := ctx.tools.RunTool(call) + if err != nil { + return Value{}, nil, err + } + if answer.Diverged { + ctx.note(ToolDivergence{ + Tool: tool, + Action: ctx.qualifiedSymbolName(shape.Tool.on), + File: shape.Tool.on.DocName, + Span: shape.Tool.on.DeclSpan, + }) + } + return answer.Outputs[resultKey], answer.Outputs, nil +} diff --git a/internal/exec/runtime/tool_calc_test.go b/internal/exec/runtime/tool_calc_test.go new file mode 100644 index 0000000000..8fdb62894a --- /dev/null +++ b/internal/exec/runtime/tool_calc_test.go @@ -0,0 +1,344 @@ +package runtime + +import ( + "errors" + "strings" + "testing" + + "github.com/Open-MBEE/OpenSysML/internal/semantic/semantics" + "github.com/Open-MBEE/OpenSysML/internal/semantic/symbols" +) + +// toolCalcModel is a tool-computed calc definition and a calc usage of it, a calc +// whose body must never run, one naming no tool, a part reading the tool's result +// as a derived attribute, and a driver invoking it positionally. +const toolCalcModel = `package test { + private import ScalarValues::*; + private import AnalysisTooling::*; + private import ISQ::*; + + calc def Thermal { + metadata ToolExecution { toolName = "Thermo"; uri = "thermo://x"; } + in m : MassValue { @ToolVariable { name = "mass"; } } + in p : PowerValue { @ToolVariable { name = "power"; } } + out warn : Boolean { @ToolVariable { name = "warn"; } } + out rating : PowerValue { @ToolVariable { name = "rating"; } } + return : TemperatureValue { @ToolVariable { name = "Tmax"; } } + } + + calc def Poisoned { + metadata ToolExecution { toolName = "Thermo"; uri = "u"; } + in a : Real { @ToolVariable { name = "a"; } } + return : Real = 1 / 0; + } + + calc def Nameless { + metadata ToolExecution { toolName = ""; uri = "u"; } + in a : Real { @ToolVariable { name = "a"; } } + return : Real; + } + + calc t : Thermal { + in m = 2 [SI::kg]; + in p = 10 [SI::W]; + } + + part def Board { + attribute mass : MassValue = 2 [SI::kg]; + attribute power : PowerValue = 10 [SI::W]; + attribute Tmax : TemperatureValue = Thermal(mass, power); + } + + calc def Driven { + return : TemperatureValue = Thermal(2 [SI::kg], 10 [SI::W]); + } +}` + +// evalText evaluates one expression in scope, for building invocation arguments. +func evalText(t *testing.T, ctx *Context, scope *symbols.Scope, text string) Value { + t.Helper() + expr, ok, err := ctx.model.parseOneExpression("", text) + if err != nil || !ok { + t.Fatalf("parsing %q: %v", text, err) + } + value, err := NewEvalContextIn(ctx, scope, nil).Eval(expr) + if err != nil { + t.Fatalf("evaluating %q: %v", text, err) + } + return value +} + +// thermalArgs is the invocation's two quantities, as a caller evaluates them. +func thermalArgs(t *testing.T, ctx *Context, scope *symbols.Scope) []Value { + t.Helper() + return []Value{evalText(t, ctx, scope, "2 [SI::kg]"), evalText(t, ctx, scope, "10 [SI::W]")} +} + +// A calc annotated ToolExecution is computed by the tool: the inputs arrive under +// their ToolVariable names, and the result parameter is bound from the answer — the +// calculation's return — while the calc states no body at all. +func TestToolCalcBindsTheResultFromTheTool(t *testing.T) { + ctx, scope := analysisFixture(t, toolCalcModel) + runner := &recordingRunner{answer: map[string]ToolValue{ + "Tmax": {Value: toolReal(340), Unit: "K"}, + "warn": {Value: semanticsBool(true)}, + "rating": {Value: toolReal(30), Unit: "W"}, + }} + ctx.SetToolRunner(runner) + + result, err := ctx.InvokeCalc(calcNamed(t, scope, "Thermal"), thermalArgs(t, ctx, scope), scope) + if err != nil { + t.Fatalf("InvokeCalc: %v", err) + } + if got := FormatValue(result); got != "340.0 [SI::K]" { + t.Fatalf("result = %s, want 340.0 [SI::K]", got) + } + if len(runner.calls) != 1 { + t.Fatalf("tool invoked %d times, want once", len(runner.calls)) + } + call := runner.calls[0] + if call.ToolName != "Thermo" || call.URI != "thermo://x" { + t.Fatalf("call names %q at %q", call.ToolName, call.URI) + } + if call.Action == nil || call.Action.Name != "Thermal" { + t.Fatalf("call.Action = %v, want the Thermal definition", call.Action) + } + inputs := map[string]ToolValue{} + for _, in := range call.Inputs { + inputs[in.Variable] = in.Value + } + if got, ok := inputs["mass"]; !ok || got.Unit != "kg" || !nearly(got.Value, toolReal(2)) { + t.Fatalf("inputs %v, want mass = 2 kg", inputs) + } + if got, ok := inputs["power"]; !ok || got.Unit != "W" || !nearly(got.Value, toolReal(10)) { + t.Fatalf("inputs %v, want power = 10 W", inputs) + } + var outputs []string + for _, out := range call.Outputs { + outputs = append(outputs, out.Variable) + } + if strings.Join(outputs, ",") != "Tmax,rating,warn" { + t.Fatalf("outputs %v, want Tmax, rating and warn", outputs) + } +} + +func semanticsBool(b bool) semantics.Value { + return semantics.Value{Kind: semantics.ValBool, Bool: b} +} + +// A calc usage of the annotated definition reads its outputs from the tool's +// answer: `warn` under its own name, the anonymous result under `result`, and the +// usage's own input bindings are what the tool was sent. +func TestToolCalcUsageReadsItsOutputsFromTheTool(t *testing.T) { + ctx, scope := analysisFixture(t, toolCalcModel) + runner := &recordingRunner{answer: map[string]ToolValue{ + "Tmax": {Value: toolReal(340), Unit: "K"}, + "warn": {Value: semanticsBool(false)}, + "rating": {Value: toolReal(30), Unit: "W"}, + }} + ctx.SetToolRunner(runner) + + outputs, err := ctx.CalcUsageOutputs(calcNamed(t, scope, "t"), scope, nil) + if err != nil { + t.Fatalf("CalcUsageOutputs: %v", err) + } + if len(runner.calls) != 1 { + t.Fatalf("tool invoked %d times, want once", len(runner.calls)) + } + if len(runner.calls[0].Inputs) != 2 { + t.Fatalf("inputs %+v, want mass and power", runner.calls[0].Inputs) + } + got := map[string]string{} + for _, out := range outputs { + got[out.Name] = FormatValue(out.Value) + } + if got["warn"] != "false" || got["rating"] != "30.0 [SI::W]" { + t.Fatalf("outputs %v, want warn = false and rating = 30 W", got) + } + result, err := ctx.CalcUsageOutput(calcNamed(t, scope, "t"), "result", scope, nil) + if err != nil { + t.Fatalf("CalcUsageOutput(result): %v", err) + } + if FormatValue(result) != "340.0 [SI::K]" { + t.Fatalf("result = %s, want 340.0 [SI::K]", FormatValue(result)) + } +} + +// The answer's unit is converted to the result parameter's coherent unit; one of +// another dimension is the reply's failure, a malformed ToolError. +func TestToolCalcConvertsTheAnsweredUnit(t *testing.T) { + t.Run("converted", func(t *testing.T) { + ctx, scope := analysisFixture(t, toolCalcModel) + ctx.SetToolRunner(&recordingRunner{answer: map[string]ToolValue{ + "Tmax": {Value: toolReal(340), Unit: "K"}, + "warn": {Value: semanticsBool(true)}, + "rating": {Value: toolReal(3), Unit: "kW"}, + }}) + outputs, err := ctx.CalcUsageOutputs(calcNamed(t, scope, "t"), scope, nil) + if err != nil { + t.Fatalf("CalcUsageOutputs: %v", err) + } + got := map[string]string{} + for _, out := range outputs { + got[out.Name] = FormatValue(out.Value) + } + if got["rating"] != "3000.0 [SI::W]" { + t.Fatalf("rating = %s, want 3000.0 [SI::W]", got["rating"]) + } + }) + t.Run("wrong dimension", func(t *testing.T) { + ctx, scope := analysisFixture(t, toolCalcModel) + ctx.SetToolRunner(&recordingRunner{answer: map[string]ToolValue{ + "Tmax": {Value: toolReal(340), Unit: "kg"}, + "warn": {Value: semanticsBool(true)}, + "rating": {Value: toolReal(30), Unit: "W"}, + }}) + _, err := ctx.InvokeCalc(calcNamed(t, scope, "Thermal"), thermalArgs(t, ctx, scope), scope) + var failure *ToolError + if !errors.As(err, &failure) || failure.Kind != ToolMalformed { + t.Fatalf("InvokeCalc = %v, want a malformed ToolError", err) + } + }) +} + +// A derived attribute bound to the tool-computed calc reads the tool's answer: +// the attribute's evaluation invokes the calc, which goes to the tool. +func TestToolCalcThroughADerivedAttribute(t *testing.T) { + ctx, scope := analysisFixture(t, toolCalcModel) + runner := &recordingRunner{answer: map[string]ToolValue{ + "Tmax": {Value: toolReal(340), Unit: "K"}, + "warn": {Value: semanticsBool(true)}, + "rating": {Value: toolReal(30), Unit: "W"}, + }} + ctx.SetToolRunner(runner) + + inst, err := ctx.Instantiate(calcNamed(t, scope, "Board")) + if err != nil { + t.Fatalf("Instantiate: %v", err) + } + fv, err := inst.GetFeatureValue(ctx, "Tmax") + if err != nil { + t.Fatalf("GetFeatureValue(Tmax): %v", err) + } + if got := FormatValue(fv.HeldValue()); got != "340.0 [SI::K]" { + t.Fatalf("Tmax = %s, want 340.0 [SI::K]", got) + } + if len(runner.calls) != 1 { + t.Fatalf("tool invoked %d times, want once", len(runner.calls)) + } +} + +// The body is never evaluated: the calc stating a result expression that cannot +// evaluate is computed by the tool all the same. +func TestToolCalcNeverRunsTheBody(t *testing.T) { + ctx, scope := analysisFixture(t, toolCalcModel) + runner := &recordingRunner{answer: map[string]ToolValue{"result": {Value: toolReal(4)}}} + ctx.SetToolRunner(runner) + + result, err := ctx.InvokeCalc(calcNamed(t, scope, "Poisoned"), []Value{realOf(1)}, scope) + if err != nil { + t.Fatalf("InvokeCalc: %v", err) + } + if got := FormatValue(result); got != "4.0" { + t.Fatalf("result = %s, want the tool's 4.0, not the body's 1 / 0", got) + } + if len(runner.calls) != 1 || len(runner.calls[0].Inputs) != 1 || runner.calls[0].Inputs[0].Variable != "a" { + t.Fatalf("calls %+v, want one sending a", runner.calls) + } +} + +// Every refusal is the performance's unchanged: no runner or an empty toolName is +// not registered, the runner's own failure propagates as is, and a reply missing +// or adding an output is a ToolError of its kind. +func TestToolCalcRefusesLikeAPerformance(t *testing.T) { + args := func(t *testing.T, ctx *Context, scope *symbols.Scope) []Value { + t.Helper() + return thermalArgs(t, ctx, scope) + } + t.Run("no runner", func(t *testing.T) { + ctx, scope := analysisFixture(t, toolCalcModel) + _, err := ctx.InvokeCalc(calcNamed(t, scope, "Thermal"), args(t, ctx, scope), scope) + var refusal *ToolNotRegisteredError + if !errors.As(err, &refusal) || !errors.Is(err, ErrToolNotRegistered) || refusal.Tool != "Thermo" { + t.Fatalf("InvokeCalc = %v, want ToolNotRegisteredError", err) + } + }) + t.Run("empty tool name", func(t *testing.T) { + ctx, scope := analysisFixture(t, toolCalcModel) + runner := &recordingRunner{answer: map[string]ToolValue{"result": {Value: toolReal(1)}}} + ctx.SetToolRunner(runner) + _, err := ctx.InvokeCalc(calcNamed(t, scope, "Nameless"), []Value{realOf(1)}, scope) + var refusal *ToolNotRegisteredError + if !errors.As(err, &refusal) || refusal.Tool != "" { + t.Fatalf("InvokeCalc = %v, want tool '' not registered", err) + } + if len(runner.calls) != 0 { + t.Fatalf("tool invoked %d times for an empty toolName", len(runner.calls)) + } + }) + t.Run("runner failure", func(t *testing.T) { + ctx, scope := analysisFixture(t, toolCalcModel) + fault := &ToolError{Tool: "Thermo", Kind: ToolTimeout, Detail: "10s"} + ctx.SetToolRunner(&recordingRunner{err: fault}) + _, err := ctx.InvokeCalc(calcNamed(t, scope, "Thermal"), args(t, ctx, scope), scope) + if !errors.Is(err, fault) { + t.Fatalf("InvokeCalc = %v, want %v", err, fault) + } + }) + t.Run("missing output", func(t *testing.T) { + ctx, scope := analysisFixture(t, toolCalcModel) + ctx.SetToolRunner(&recordingRunner{answer: map[string]ToolValue{ + "warn": {Value: semanticsBool(true)}, + "rating": {Value: toolReal(30), Unit: "W"}, + }}) + _, err := ctx.InvokeCalc(calcNamed(t, scope, "Thermal"), args(t, ctx, scope), scope) + var failure *ToolError + if !errors.As(err, &failure) || failure.Kind != ToolMissingOutput { + t.Fatalf("InvokeCalc = %v, want ToolMissingOutput", err) + } + }) + t.Run("unknown output", func(t *testing.T) { + ctx, scope := analysisFixture(t, toolCalcModel) + ctx.SetToolRunner(&recordingRunner{answer: map[string]ToolValue{ + "Tmax": {Value: toolReal(340), Unit: "K"}, + "warn": {Value: semanticsBool(true)}, + "rating": {Value: toolReal(30), Unit: "W"}, + "z": {Value: toolReal(1)}, + }}) + _, err := ctx.InvokeCalc(calcNamed(t, scope, "Thermal"), args(t, ctx, scope), scope) + var failure *ToolError + if !errors.As(err, &failure) || failure.Kind != ToolUnknownOutput { + t.Fatalf("InvokeCalc = %v, want ToolUnknownOutput", err) + } + }) +} + +// A runner reporting unequal answers for equal inputs leaves the run a +// ToolDivergence, as a performance's does. +func TestToolCalcNotesDivergence(t *testing.T) { + ctx, scope := analysisFixture(t, toolCalcModel) + ctx.SetToolRunner(divergingCalcRunner{}) + if _, err := ctx.InvokeCalc(calcNamed(t, scope, "Thermal"), thermalArgs(t, ctx, scope), scope); err != nil { + t.Fatalf("InvokeCalc: %v", err) + } + notes := ctx.Notes() + if len(notes) != 1 { + t.Fatalf("notes = %v, want one divergence", notes) + } + d, ok := notes[0].(ToolDivergence) + if !ok || d.Tool != "Thermo" || !strings.Contains(d.Describe(), "answered differently for equal inputs") { + t.Fatalf("note = %#v", notes[0]) + } +} + +// divergingCalcRunner answers the calc's call and reports the answer as changed. +type divergingCalcRunner struct{} + +func (divergingCalcRunner) RunTool(call *ToolCall) (ToolAnswer, error) { + outputs, err := call.Bind(map[string]ToolValue{ + "Tmax": {Value: toolReal(340), Unit: "K"}, + "warn": {Value: semanticsBool(true)}, + "rating": {Value: toolReal(30), Unit: "W"}, + }) + return ToolAnswer{Outputs: outputs, Diverged: true}, err +} diff --git a/internal/frontend/grpc/service.go b/internal/frontend/grpc/service.go index a236e851e2..6210b9f19b 100644 --- a/internal/frontend/grpc/service.go +++ b/internal/frontend/grpc/service.go @@ -456,6 +456,13 @@ func (s *Service) newRuntimeContext(model *runtime.Model) *runtime.Context { // Unreachable: NewService validated these budgets. panic(fmt.Sprintf("grpc: invalid service budgets: %v", err)) } + // The runner puts tool-computed actions and calcs of contexts held outside a plan — + // feature values, documents, calc usages — to the engines as Compute questions. + schedule := ctx.Schedule() + if schedule == (runtime.SchedulePolicy{}) { + schedule = runtime.DefaultSchedulePolicy + } + ctx.SetToolRunner(s.engines.ToolRunner(context.Background(), ctx, analysis.BudgetOf(s.budgets, schedule, analysis.Compute, s.jobs), analysis.Auto())) return ctx } diff --git a/internal/frontend/repl/tool_calc_test.go b/internal/frontend/repl/tool_calc_test.go new file mode 100644 index 0000000000..ee5306eed1 --- /dev/null +++ b/internal/frontend/repl/tool_calc_test.go @@ -0,0 +1,104 @@ +package repl + +import ( + "fmt" + "os" + "os/exec" + "path/filepath" + "strconv" + "sync" + "testing" + + "github.com/Open-MBEE/OpenSysML/internal/exec/analysis" + "github.com/Open-MBEE/OpenSysML/tests/testutil/gobuild" +) + +var ( + toolCalcOnce sync.Once + toolCalcPath string + toolCalcErr error +) + +// toolCalcBinary builds the tool stand-in of the analysis package's tests once per +// test binary. +func toolCalcBinary(t *testing.T) string { + t.Helper() + toolCalcOnce.Do(func() { + dir, err := os.MkdirTemp("", "toolcalc") + if err != nil { + toolCalcErr = err + return + } + toolCalcPath = filepath.Join(dir, "toolcalc") + build := exec.Command("go", gobuild.Args(toolCalcPath)...) + build.Dir = filepath.Join("..", "..", "exec", "analysis", "testdata", "toolcalc") + if out, err := build.CombinedOutput(); err != nil { + toolCalcErr = fmt.Errorf("go build: %v\n%s", err, out) + } + }) + if toolCalcErr != nil { + t.Fatalf("building the tool stand-in: %v", toolCalcErr) + } + return toolCalcPath +} + +// toolCalcModel is a calc the prompt computes through the tool its ToolExecution names. +const toolCalcModel = ` +package thermo { + private import ScalarValues::*; + private import AnalysisTooling::*; + private import ISQ::*; + + calc def Thermal { + metadata ToolExecution { toolName = "Thermo"; uri = "thermo://local"; } + in m : MassValue { @ToolVariable { name = "mass"; } } + in p : PowerValue { @ToolVariable { name = "power"; } } + out warn : Boolean { @ToolVariable { name = "warn"; } } + return : TemperatureValue { @ToolVariable { name = "Tmax"; } } + } +} +` + +// toolCalcManifest writes a manifest naming Thermo and returns its directory. +func toolCalcManifest(t *testing.T) string { + t.Helper() + dir := t.TempDir() + entry := `{"kind":"tool","toolName":"Thermo","version":"1.0.0",` + + `"executable":` + strconv.Quote(toolCalcBinary(t)) + `,` + + `"variables":["mass","power","warn","Tmax"]}` + if err := os.WriteFile(filepath.Join(dir, "thermo.json"), []byte(entry), 0o600); err != nil { + t.Fatal(err) + } + return dir +} + +// A %calc of a tool-computed calc prints the value the tool answers. +func TestToolCalcAtThePrompt(t *testing.T) { + s := loadSource(t, toolCalcModel) + t.Setenv(analysis.ToolsEnv, toolCalcManifest(t)) + engines, err := analysis.DefaultFromEnv() + if err != nil { + t.Fatalf("DefaultFromEnv: %v", err) + } + if err := s.SetEngines(engines); err != nil { + t.Fatalf("SetEngines: %v", err) + } + + wants(t, run(t, s, "%calc Thermal(2 [SI::kg], 10 [SI::W])"), "20 [SI::K]") +} + +// Without a manifest the same %calc fails as not registered, never evaluating a body. +func TestToolCalcUnregisteredAtThePrompt(t *testing.T) { + s := loadSource(t, toolCalcModel) + t.Setenv(analysis.ToolsEnv, t.TempDir()) + engines, err := analysis.DefaultFromEnv() + if err != nil { + t.Fatalf("DefaultFromEnv: %v", err) + } + if err := s.SetEngines(engines); err != nil { + t.Fatalf("SetEngines: %v", err) + } + + wants(t, run(t, s, "%calc Thermal(2 [SI::kg], 10 [SI::W])"), + "tool 'Thermo' is not registered; set OPENSYSML_TOOLS") +} diff --git a/internal/frontend/repl/view.go b/internal/frontend/repl/view.go index c878b09beb..3b02bd69ed 100644 --- a/internal/frontend/repl/view.go +++ b/internal/frontend/repl/view.go @@ -423,6 +423,7 @@ func (r *reportRuntime) runtime() (*runtime.Context, error) { return nil, err } r.session.applyDraws(ctx) + r.session.attachTools(ctx) r.ctx = ctx return ctx, nil } diff --git a/internal/frontend/usage/environment.go b/internal/frontend/usage/environment.go index c6d95d7163..f0b850b484 100644 --- a/internal/frontend/usage/environment.go +++ b/internal/frontend/usage/environment.go @@ -42,7 +42,7 @@ const BudgetScopeNote = "A budget bounds one run — one evaluation, one " + // performs actions annotated ToolExecution reads. func ToolEnvironment() []Item { return []Item{ - {"OPENSYSML_TOOLS", "Directory of the tool manifest: one JSON file per external tool (toolName, version, executable, variables), each registered as the engine tool: that performs actions annotated ToolExecution with that toolName. Unset registers no tool, and such an action is refused."}, - {"OPENSYSML_TOOL_TIMEOUT", "How long one tool invocation may take, as a Go duration. Default 10s, after which the performance fails."}, + {"OPENSYSML_TOOLS", "Directory of the tool manifest: one JSON file per external tool (toolName, version, executable, variables), each registered as the engine tool: that runs actions and calcs annotated ToolExecution with that toolName. Unset registers no tool, and such an action or calc is refused."}, + {"OPENSYSML_TOOL_TIMEOUT", "How long one tool invocation may take, as a Go duration. Default 10s, after which the performance or calculation fails."}, } } diff --git a/packaging/man/man1/sysml-grpc.1 b/packaging/man/man1/sysml-grpc.1 index 5ce2e5dba9..e5762113d9 100644 --- a/packaging/man/man1/sysml-grpc.1 +++ b/packaging/man/man1/sysml-grpc.1 @@ -139,12 +139,12 @@ CPUs. .B OPENSYSML_TOOLS Directory of the tool manifest: one JSON file per external tool (toolName, version, executable, variables), each registered as the engine tool: -that performs actions annotated ToolExecution with that toolName. Unset -registers no tool, and such an action is refused. +that runs actions and calcs annotated ToolExecution with that toolName. Unset +registers no tool, and such an action or calc is refused. .TP .B OPENSYSML_TOOL_TIMEOUT How long one tool invocation may take, as a Go duration. Default 10s, after -which the performance fails. +which the performance or calculation fails. .PP Each variable above also answers to its legacy SYSML_\-prefixed name (SYSML_MAX_STEPS for OPENSYSML_MAX_STEPS, and so on), which remains accepted. diff --git a/packaging/man/man1/sysml.1 b/packaging/man/man1/sysml.1 index 979ffe7f30..d9523911de 100644 --- a/packaging/man/man1/sysml.1 +++ b/packaging/man/man1/sysml.1 @@ -954,12 +954,12 @@ CPUs. .B OPENSYSML_TOOLS Directory of the tool manifest: one JSON file per external tool (toolName, version, executable, variables), each registered as the engine tool: -that performs actions annotated ToolExecution with that toolName. Unset -registers no tool, and such an action is refused. +that runs actions and calcs annotated ToolExecution with that toolName. Unset +registers no tool, and such an action or calc is refused. .TP .B OPENSYSML_TOOL_TIMEOUT How long one tool invocation may take, as a Go duration. Default 10s, after -which the performance fails. +which the performance or calculation fails. .TP .B OPENSYSML_SMT Executable the %check, %explain, %solve, %configure and %optimize commands diff --git a/tests/grpc/conformance_test.go b/tests/grpc/conformance_test.go index e2e75370f5..838770194e 100644 --- a/tests/grpc/conformance_test.go +++ b/tests/grpc/conformance_test.go @@ -5,14 +5,18 @@ import ( "encoding/json" "fmt" "os" + "os/exec" "path/filepath" "strconv" "strings" + "sync" "testing" pb "github.com/Open-MBEE/OpenSysML/api/proto" + "github.com/Open-MBEE/OpenSysML/internal/exec/analysis" "github.com/Open-MBEE/OpenSysML/internal/frontend/grpc" "github.com/Open-MBEE/OpenSysML/tests/fixtures" + "github.com/Open-MBEE/OpenSysML/tests/testutil/gobuild" ) // expectedValue is the fixture encoding of a pb.Value: the oneof field name @@ -53,9 +57,16 @@ type conformanceCase struct { ContextSymbolID string `json:"context_symbol_id,omitempty"` SubjectSymbolID string `json:"subject_symbol_id,omitempty"` - // GetSymbol, Instantiate, ExecuteAction, ExecuteState + // EvaluateCalc: positional arguments bound to the calc's parameters. + Arguments []expectedValue `json:"arguments,omitempty"` + + // GetSymbol, Instantiate, ExecuteAction, ExecuteState, EvaluateCalc SymbolID string `json:"symbol_id,omitempty"` + // Tools names stand-ins under internal/exec/analysis/testdata registered + // from a manifest written before the service is built. + Tools []string `json:"tools,omitempty"` + // ExecuteAction Inputs map[string]expectedValue `json:"inputs,omitempty"` @@ -149,6 +160,10 @@ func runGRPCConformanceCase(t *testing.T, dir, caseName string) { t.Fatalf("read model: %v", err) } + if len(tc.Tools) > 0 { + t.Setenv(analysis.ToolsEnv, conformanceToolManifest(t, tc.Tools)) + } + srv, err := grpc.NewService(4, "test") if err != nil { t.Fatalf("NewService: %v", err) @@ -183,6 +198,8 @@ func runGRPCConformanceCase(t *testing.T, dir, caseName string) { runApplyEditsCase(t, srv, ctx, parseResp.ModelHash, tc) case "RunDocumentQuery": runRunDocumentQueryCase(t, srv, ctx, parseResp.ModelHash, tc) + case "EvaluateCalc": + runEvaluateCalcCase(t, srv, ctx, parseResp.ModelHash, tc) default: t.Fatalf("unknown rpc %q", tc.RPC) } @@ -259,6 +276,105 @@ func runEvaluateCase(t *testing.T, srv *grpc.Service, ctx context.Context, model } } +// runEvaluateCalcCase invokes a calc through the service, as %calc does at the +// prompt: positional arguments bound, the result and the outputs a usage +// computes compared against the fixture. +func runEvaluateCalcCase(t *testing.T, srv *grpc.Service, ctx context.Context, modelHash string, tc conformanceCase) { + t.Helper() + + args := make([]*pb.Value, 0, len(tc.Arguments)) + for _, arg := range tc.Arguments { + args = append(args, toProtoValue(t, arg)) + } + + resp, err := srv.EvaluateCalc(ctx, &pb.EvaluateCalcRequest{ + ModelHash: modelHash, + SymbolId: tc.SymbolID, + Arguments: args, + }) + if err != nil { + t.Fatalf("EvaluateCalc: %v", err) + } + if checkExpectedError(t, tc, resp.Error) { + return + } + if tc.ExpectedResult != nil { + checkValue(t, "result", *tc.ExpectedResult, resp.Result) + } + if tc.ExpectedOutputs != nil { + got := make(map[string]*pb.Value, len(resp.Outputs)) + for _, out := range resp.Outputs { + got[out.GetName()] = out.GetValue() + } + for name, want := range tc.ExpectedOutputs { + value, ok := got[name] + if !ok { + t.Errorf("missing output %q", name) + continue + } + checkValue(t, "output "+name, want, value) + } + } +} + +// conformanceToolManifest builds each stand-in the fixture names and writes a +// manifest registering it under the tool name the models use. +func conformanceToolManifest(t *testing.T, tools []string) string { + t.Helper() + dir := t.TempDir() + for _, tool := range tools { + spec, ok := conformanceTools[tool] + if !ok { + t.Fatalf("unknown tool stand-in %q", tool) + } + entry := `{"kind":"tool","toolName":` + strconv.Quote(spec.toolName) + `,` + + `"executable":` + strconv.Quote(conformanceToolBinary(t, tool)) + `,` + + `"variables":[` + strings.Join(spec.variables, `,`) + `]}` + if err := os.WriteFile(filepath.Join(dir, spec.toolName+".json"), []byte(entry), 0o600); err != nil { + t.Fatal(err) + } + } + return dir +} + +// conformanceTools names the stand-ins a fixture can register: the toolName +// their ToolExecution names and the variables the manifest accepts. +var conformanceTools = map[string]struct { + toolName string + variables []string +}{ + "toolcalc": {"Thermo", []string{`"mass"`, `"power"`, `"warn"`, `"Tmax"`}}, +} + +var conformanceToolBuilds sync.Map // name -> string path or error + +// conformanceToolBinary builds a stand-in once per test binary. +func conformanceToolBinary(t *testing.T, name string) string { + t.Helper() + if built, ok := conformanceToolBuilds.Load(name); ok { + switch v := built.(type) { + case string: + return v + case error: + t.Fatalf("building the %s stand-in: %v", name, v) + } + } + dir, err := os.MkdirTemp("", "conformance-tool") + if err != nil { + t.Fatal(err) + } + path := filepath.Join(dir, name) + build := exec.Command("go", gobuild.Args(path)...) + build.Dir = filepath.Join("..", "..", "internal", "exec", "analysis", "testdata", name) + if out, err := build.CombinedOutput(); err != nil { + err = fmt.Errorf("go build: %v\n%s", err, out) + conformanceToolBuilds.Store(name, err) + t.Fatalf("building the %s stand-in: %v", name, err) + } + conformanceToolBuilds.Store(name, path) + return path +} + // runGetSymbolCase pins the static facts a symbol is reported with, and is how // the attribute set is kept from regressing to empty. func runGetSymbolCase(t *testing.T, srv *grpc.Service, ctx context.Context, modelHash string, tc conformanceCase) { diff --git a/tests/grpc/testdata/conformance/README.md b/tests/grpc/testdata/conformance/README.md index a7ba627f9b..fe3ddd4477 100644 --- a/tests/grpc/testdata/conformance/README.md +++ b/tests/grpc/testdata/conformance/README.md @@ -14,11 +14,13 @@ in this directory, so adding a case is a data-only change. | Field | Applies to | Meaning | |---|---|---| -| `rpc` | all | `GetSymbol`, `Evaluate`, `Instantiate`, `ExecuteAction`, `ExecuteState`, `ApplyEdits` or `RunDocumentQuery` | +| `rpc` | all | `GetSymbol`, `Evaluate`, `Instantiate`, `ExecuteAction`, `ExecuteState`, `ApplyEdits`, `RunDocumentQuery` or `EvaluateCalc` | | `expression` | Evaluate | expression source to evaluate | | `context_symbol_id` | Evaluate | optional FQN whose scope the expression is evaluated in | | `subject_symbol_id` | Evaluate | optional FQN of a part/usage instantiated and evaluated against, so features read its feature values | -| `symbol_id` | GetSymbol, Instantiate, ExecuteAction, ExecuteState | FQN of the subject | +| `symbol_id` | GetSymbol, Instantiate, ExecuteAction, ExecuteState, EvaluateCalc | FQN of the subject | +| `arguments` | EvaluateCalc | positional arguments bound to the calc's parameters | +| `tools` | EvaluateCalc | stand-ins under `internal/exec/analysis/testdata` built and registered from a manifest before the service is built | | `inputs` | ExecuteAction | parameter name → value, bound before execution | | `events` | ExecuteState | event names injected, in order | | `instantiate` | RunDocumentQuery | FQNs `Instantiate` creates objects of first, in order, so the query can bind and enumerate them | @@ -26,12 +28,12 @@ in this directory, so adding a case is a data-only change. | `bindings` | RunDocumentQuery | list of `{parameter, values}`, each value a document value bound to the parameter | | `expected_columns` | RunDocumentQuery | full ordered projected-column name list | | `expected_rows` | RunDocumentQuery | full ordered row list, each `{element, cells}` — the row's own document value and one list of document values per column | -| `expected_result` | Evaluate | expected `Value` | +| `expected_result` | Evaluate, EvaluateCalc | expected `Value` | | `expected_attribute_names` | GetSymbol | full ordered attribute name list, own then inherited | | `expected_attributes` | GetSymbol | attribute name → `{type, value_kind, value, unit}`; no `value_kind` requires no value | | `expected_feature_values` | Instantiate | feature name → `{materialized, value_kind, value, error}` | | `expected_instance_count` | Instantiate | number of reachable instances in the response graph | -| `expected_outputs` | ExecuteAction | output name → expected `Value` | +| `expected_outputs` | ExecuteAction, EvaluateCalc | output name → expected `Value`; for EvaluateCalc it is the outputs a calc usage computes when invoked without arguments | | `expected_states_visited` | ExecuteState | full ordered state-visit trace | | `expected_final_context` | ExecuteState | context entry name → expected `Value` | | `expected_error` | all | substring the RPC's in-band `error` must contain (for `RunDocumentQuery`, the status error the call fails with) | diff --git a/tests/grpc/testdata/conformance/evaluate_calc_tool.expected.json b/tests/grpc/testdata/conformance/evaluate_calc_tool.expected.json new file mode 100644 index 0000000000..7cb1acaf1d --- /dev/null +++ b/tests/grpc/testdata/conformance/evaluate_calc_tool.expected.json @@ -0,0 +1,13 @@ +{ + "rpc": "EvaluateCalc", + "tools": ["toolcalc"], + "symbol_id": "thermoCase::Thermal", + "arguments": [ + {"kind": "real_value", "value": 2}, + {"kind": "real_value", "value": 10} + ], + "expected_result": { + "kind": "quantity", + "value": "20 [SI::K] = SI::kelvin" + } +} diff --git a/tests/grpc/testdata/conformance/evaluate_calc_tool.sysml b/tests/grpc/testdata/conformance/evaluate_calc_tool.sysml new file mode 100644 index 0000000000..0a07a2d3ee --- /dev/null +++ b/tests/grpc/testdata/conformance/evaluate_calc_tool.sysml @@ -0,0 +1,13 @@ +package thermoCase { + private import ScalarValues::*; + private import AnalysisTooling::*; + private import ISQ::*; + + calc def Thermal { + metadata ToolExecution { toolName = "Thermo"; uri = "thermo://local"; } + in m : Real { @ToolVariable { name = "mass"; } } + in p : Real { @ToolVariable { name = "power"; } } + out warn : Boolean { @ToolVariable { name = "warn"; } } + return : TemperatureValue { @ToolVariable { name = "Tmax"; } } + } +} diff --git a/tests/grpc/testdata/conformance/evaluate_calc_tool_unregistered.expected.json b/tests/grpc/testdata/conformance/evaluate_calc_tool_unregistered.expected.json new file mode 100644 index 0000000000..668ec9bda4 --- /dev/null +++ b/tests/grpc/testdata/conformance/evaluate_calc_tool_unregistered.expected.json @@ -0,0 +1,9 @@ +{ + "rpc": "EvaluateCalc", + "symbol_id": "thermoCase::Thermal", + "arguments": [ + {"kind": "real_value", "value": 2}, + {"kind": "real_value", "value": 10} + ], + "expected_error": "is not registered" +} diff --git a/tests/grpc/testdata/conformance/evaluate_calc_tool_unregistered.sysml b/tests/grpc/testdata/conformance/evaluate_calc_tool_unregistered.sysml new file mode 100644 index 0000000000..0a07a2d3ee --- /dev/null +++ b/tests/grpc/testdata/conformance/evaluate_calc_tool_unregistered.sysml @@ -0,0 +1,13 @@ +package thermoCase { + private import ScalarValues::*; + private import AnalysisTooling::*; + private import ISQ::*; + + calc def Thermal { + metadata ToolExecution { toolName = "Thermo"; uri = "thermo://local"; } + in m : Real { @ToolVariable { name = "mass"; } } + in p : Real { @ToolVariable { name = "power"; } } + out warn : Boolean { @ToolVariable { name = "warn"; } } + return : TemperatureValue { @ToolVariable { name = "Tmax"; } } + } +} From fe5358984858fd883c989a2e6f0f21a9e1723b14 Mon Sep 17 00:00:00 2001 From: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Fri, 25 Sep 2026 06:56:45 +0000 Subject: [PATCH 02/10] fix(exec): return nothing for a tool calc declaring no result parameter calcToolCall reports resultKey empty when the calc has no result parameter, so an out-only tool calc demands no 'result' output of the reply; the calculation yields its designated output as a body that returns nothing does, and a usage's trace records CalcUsageExit rather than a returned value. Co-Authored-By: jason.han --- internal/exec/runtime/calc_usage.go | 10 ++++-- internal/exec/runtime/invoke_calc.go | 34 ++++++++++++++++--- internal/exec/runtime/tool_calc.go | 25 +++++++------- internal/exec/runtime/tool_calc_test.go | 44 +++++++++++++++++++++++++ 4 files changed, 95 insertions(+), 18 deletions(-) diff --git a/internal/exec/runtime/calc_usage.go b/internal/exec/runtime/calc_usage.go index 57dc3b79e7..d6af724d7c 100644 --- a/internal/exec/runtime/calc_usage.go +++ b/internal/exec/runtime/calc_usage.go @@ -896,19 +896,23 @@ func (ctx *Context) runCalcUsage(start *calcUsageStart) (*calcRun, error) { // result parameter's answer its result, and reading an unanswered one is the // reply's failure rather than a binding to evaluate. if shape.Tool != nil { - result, outputs, err := ctx.computeCalcByTool(shape, ec.scope, env.lookup) + result, returned, outputs, err := ctx.computeCalcByTool(shape, ec.scope, env.lookup) if err != nil { if ec.trace != nil { ec.trace.RecordCalculationExitError(shape.Kind, shape.Name, err) } return nil, calcFrame(shape.Kind, shape.Name, err) } - run.result, run.returned = result, true + run.result, run.returned = result, returned for name, value := range outputs { run.outputs[name] = value } if ec.trace != nil { - ec.trace.RecordCalculationExit(shape.Kind, shape.Name, result) + if returned { + ec.trace.RecordCalculationExit(shape.Kind, shape.Name, result) + } else { + ec.trace.RecordCalcUsageExit(shape.Kind, shape.Name) + } } return run, nil } diff --git a/internal/exec/runtime/invoke_calc.go b/internal/exec/runtime/invoke_calc.go index 8a17604900..1d57de4af0 100644 --- a/internal/exec/runtime/invoke_calc.go +++ b/internal/exec/runtime/invoke_calc.go @@ -256,11 +256,11 @@ func (ctx *Context) calcInterfaceOf(sym *symbols.Symbol) (*calcShape, error) { shape.BodyOutputs = assignedOutputs(shape.Steps, shape.Outputs, shape.Aliases) shape.Bindings = calcBindings(chain) shape.ResultExpr = resultBindingExpr(shape.Bindings) - if tool, err := ctx.toolExecutionOf(sym); err != nil { + tool, err := ctx.toolExecutionOf(sym) + if err != nil { return nil, err - } else { - shape.Tool = tool } + shape.Tool = tool // A calc computes nothing unless it returns or binds an output; a case also // computes through its steps, or answers with its verdicts alone, and a // library function the runtime implements natively computes through that. @@ -702,7 +702,12 @@ func (ctx *Context) invokeCalcShapeIn(shape *calcShape, args calcArgs, callerSco var result Value var err error if shape.Tool != nil { - result, _, err = ctx.computeCalcByTool(shape, ec.scope, locals.lookup) + var returned bool + var outputs map[string]Value + result, returned, outputs, err = ctx.computeCalcByTool(shape, ec.scope, locals.lookup) + if err == nil && !returned { + result, err = ctx.toolCalcResult(shape, frame, callerScope, self, activation, enclosing, outputs) + } } else { result, err = ctx.runCalcBody(shape, frame, callerScope, self, activation, enclosing) } @@ -823,6 +828,27 @@ func (ctx *Context) runCalcBody(shape *calcShape, frame *invocationFrame, caller return run.value(ctx, out) } +// toolCalcResult resolves what an invocation of a tool-computed calc yields when +// the calc declares no result parameter, as runCalcBody does for a body that +// returned nothing: the designated output's value, read from the tool's answers. +func (ctx *Context) toolCalcResult(shape *calcShape, frame *invocationFrame, callerScope *symbols.Scope, self *Instance, activation int64, enclosing []frame, outputs map[string]Value) (Value, error) { + out, err := shape.designatedOutput() + if err != nil { + return Value{}, err + } + run := newCalcRun(shape, callerScope, self, frame.locals()) + run.activation, run.perf = activation, frame.host.performance() + if len(enclosing) > 0 { + run.outer = &EvalContext{ctx: ctx, scope: callerScope, self: self, frames: enclosing, trace: ctx.trace, activation: activation} + } + // The invocation already holds this evaluation's nesting feature value. + run.onStack = true + for name, value := range outputs { + run.outputs[name] = value + } + return run.value(ctx, out) +} + // runCalcSteps runs the calc's lowered steps on engine, whose data holds the // calc's parameters on the way in and its locals on the way out, reporting // the value host took from a `return` and whether the body returned one. diff --git a/internal/exec/runtime/tool_calc.go b/internal/exec/runtime/tool_calc.go index fa96a0b11f..a1f562b493 100644 --- a/internal/exec/runtime/tool_calc.go +++ b/internal/exec/runtime/tool_calc.go @@ -19,7 +19,8 @@ import ( // and the result parameter is always an output — keyed by its ToolVariable name, else its // declared name, else `result`. The same refusals as a performance's call apply: an // unbound non-optional input is ErrUnboundParameter, a ToolVariable name two parameters -// carry is a ToolError. The name the result is bound under is returned with the call. +// carry is a ToolError. The name the result is bound under is returned with the call, +// empty when the calc declares no result parameter — an out-only calc asks for none. func (ctx *Context) calcToolCall(shape *calcShape, scope *symbols.Scope, held func(param string) (Value, bool)) (*ToolCall, string, error) { execution := shape.Tool tool := execution.tool @@ -70,8 +71,9 @@ func (ctx *Context) calcToolCall(shape *calcShape, scope *symbols.Scope, held fu call.Outputs = append(call.Outputs, ToolOutput{Variable: variable, Parameter: name, Declared: param.Symbol}) } } - resultKey := resultOutputName + var resultKey string if result != nil { + resultKey = resultOutputName variable := resultOutputName if result.Name != "" { resultKey, variable = result.Name, result.Name @@ -94,24 +96,25 @@ func (ctx *Context) calcToolCall(shape *calcShape, scope *symbols.Scope, held fu // computeCalcByTool computes a tool-annotated calc: the tool shape.Tool names is invoked // once with the inputs held answers, its outputs stand as the calc's outputs, and the -// result parameter's value is the calculation's result. The failures of a performance's -// tool apply unchanged: an annotation naming no tool, or a context with no runner, is -// not registered and the body never stands in. -func (ctx *Context) computeCalcByTool(shape *calcShape, scope *symbols.Scope, held func(param string) (Value, bool)) (Value, map[string]Value, error) { +// result parameter's value is the calculation's result — returned=false for a calc +// declaring no result parameter, as a body that returns nothing reports. The failures +// of a performance's tool apply unchanged: an annotation naming no tool, or a context +// with no runner, is not registered and the body never stands in. +func (ctx *Context) computeCalcByTool(shape *calcShape, scope *symbols.Scope, held func(param string) (Value, bool)) (result Value, returned bool, outputs map[string]Value, err error) { tool := shape.Tool.tool if tool == "" { - return Value{}, nil, &ToolNotRegisteredError{Tool: tool} + return Value{}, false, nil, &ToolNotRegisteredError{Tool: tool} } call, resultKey, err := ctx.calcToolCall(shape, scope, held) if err != nil { - return Value{}, nil, err + return Value{}, false, nil, err } if ctx.tools == nil { - return Value{}, nil, &ToolNotRegisteredError{Tool: tool} + return Value{}, false, nil, &ToolNotRegisteredError{Tool: tool} } answer, err := ctx.tools.RunTool(call) if err != nil { - return Value{}, nil, err + return Value{}, false, nil, err } if answer.Diverged { ctx.note(ToolDivergence{ @@ -121,5 +124,5 @@ func (ctx *Context) computeCalcByTool(shape *calcShape, scope *symbols.Scope, he Span: shape.Tool.on.DeclSpan, }) } - return answer.Outputs[resultKey], answer.Outputs, nil + return answer.Outputs[resultKey], resultKey != "", answer.Outputs, nil } diff --git a/internal/exec/runtime/tool_calc_test.go b/internal/exec/runtime/tool_calc_test.go index 8fdb62894a..39601ddb5a 100644 --- a/internal/exec/runtime/tool_calc_test.go +++ b/internal/exec/runtime/tool_calc_test.go @@ -43,6 +43,19 @@ const toolCalcModel = `package test { in p = 10 [SI::W]; } + calc def Warned { + metadata ToolExecution { toolName = "Thermo"; uri = "w"; } + in m : MassValue { @ToolVariable { name = "mass"; } } + in p : PowerValue { @ToolVariable { name = "power"; } } + out warn : Boolean { @ToolVariable { name = "warn"; } } + out rating : PowerValue { @ToolVariable { name = "rating"; } } + } + + calc w : Warned { + in m = 2 [SI::kg]; + in p = 10 [SI::W]; + } + part def Board { attribute mass : MassValue = 2 [SI::kg]; attribute power : PowerValue = 10 [SI::W]; @@ -164,6 +177,37 @@ func TestToolCalcUsageReadsItsOutputsFromTheTool(t *testing.T) { } } +// A calc declaring only `out` parameters asks the tool for those alone — no +// `result` output is demanded — and its usage reads them as any usage's. +func TestToolCalcOutOnlyAsksNoResult(t *testing.T) { + ctx, scope := analysisFixture(t, toolCalcModel) + runner := &recordingRunner{answer: map[string]ToolValue{ + "warn": {Value: semanticsBool(true)}, + "rating": {Value: toolReal(30), Unit: "W"}, + }} + ctx.SetToolRunner(runner) + + outputs, err := ctx.CalcUsageOutputs(calcNamed(t, scope, "w"), scope, nil) + if err != nil { + t.Fatalf("CalcUsageOutputs: %v", err) + } + if len(runner.calls) != 1 { + t.Fatalf("tool invoked %d times, want once", len(runner.calls)) + } + for _, out := range runner.calls[0].Outputs { + if out.Variable == "result" { + t.Fatalf("call demands a result output from an out-only calc: %+v", runner.calls[0].Outputs) + } + } + got := map[string]string{} + for _, out := range outputs { + got[out.Name] = FormatValue(out.Value) + } + if got["warn"] != "true" || got["rating"] != "30.0 [SI::W]" { + t.Fatalf("outputs %v, want warn = true and rating = 30 W", got) + } +} + // The answer's unit is converted to the result parameter's coherent unit; one of // another dimension is the reply's failure, a malformed ToolError. func TestToolCalcConvertsTheAnsweredUnit(t *testing.T) { From e5ff142cb837be5863409d66c232c77079a377b2 Mon Sep 17 00:00:00 2001 From: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Fri, 25 Sep 2026 08:10:45 +0000 Subject: [PATCH 03/10] fix(exec): designate a tool calc's bare out as its invocation result MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A calc declaring only ToolVariable-marked outputs has no declared value or body assignment for designatedOutput to see, so invoking it yielded ErrCalcNoReturn though the tool answered. designatedToolOutput picks among the outputs the tool actually bound — an answered result parameter winning outright, one answered output alone, none or several failing as designatedOutput reports — and its value is read from the reply, never from a body binding. Co-Authored-By: jason.han --- internal/exec/runtime/invoke_calc.go | 61 ++++++++++++++----- .../exec/runtime/robustness_toolcalc_test.go | 13 ++++ internal/exec/runtime/tool_calc_test.go | 61 +++++++++++++++++++ 3 files changed, 121 insertions(+), 14 deletions(-) diff --git a/internal/exec/runtime/invoke_calc.go b/internal/exec/runtime/invoke_calc.go index 1d57de4af0..5c9d2efe43 100644 --- a/internal/exec/runtime/invoke_calc.go +++ b/internal/exec/runtime/invoke_calc.go @@ -706,7 +706,7 @@ func (ctx *Context) invokeCalcShapeIn(shape *calcShape, args calcArgs, callerSco var outputs map[string]Value result, returned, outputs, err = ctx.computeCalcByTool(shape, ec.scope, locals.lookup) if err == nil && !returned { - result, err = ctx.toolCalcResult(shape, frame, callerScope, self, activation, enclosing, outputs) + result, err = shape.toolCalcResult(outputs) } } else { result, err = ctx.runCalcBody(shape, frame, callerScope, self, activation, enclosing) @@ -829,24 +829,57 @@ func (ctx *Context) runCalcBody(shape *calcShape, frame *invocationFrame, caller } // toolCalcResult resolves what an invocation of a tool-computed calc yields when -// the calc declares no result parameter, as runCalcBody does for a body that -// returned nothing: the designated output's value, read from the tool's answers. -func (ctx *Context) toolCalcResult(shape *calcShape, frame *invocationFrame, callerScope *symbols.Scope, self *Instance, activation int64, enclosing []frame, outputs map[string]Value) (Value, error) { - out, err := shape.designatedOutput() +// the calc declares no result parameter: the value the tool bound to the output +// it designates, as runCalcBody resolves a body that returned nothing — a result +// parameter the tool answered wins outright, the one output the tool bound wins +// alone, and several or none fail as designatedOutput fails. The tool's answers +// are the whole computation; no body binding is evaluated. +func (shape *calcShape) toolCalcResult(outputs map[string]Value) (Value, error) { + out, err := shape.designatedToolOutput(outputs) if err != nil { return Value{}, err } - run := newCalcRun(shape, callerScope, self, frame.locals()) - run.activation, run.perf = activation, frame.host.performance() - if len(enclosing) > 0 { - run.outer = &EvalContext{ctx: ctx, scope: callerScope, self: self, frames: enclosing, trace: ctx.trace, activation: activation} + key := out.Name + if key == "" { + key = resultOutputName } - // The invocation already holds this evaluation's nesting feature value. - run.onStack = true - for name, value := range outputs { - run.outputs[name] = value + return outputs[key], nil +} + +// designatedToolOutput returns the output an invocation of a tool-computed calc +// yields, in the spirit of designatedOutput: the candidates are the outputs the +// tool answered, a result parameter winning outright over the rest. +func (shape *calcShape) designatedToolOutput(outputs map[string]Value) (calcOutput, error) { + var valued []calcOutput + for _, out := range shape.Outputs { + key := out.Name + if out.IsResult && key == "" { + key = resultOutputName + } + if _, answered := outputs[key]; !answered { + continue + } + if out.IsResult { + return out, nil + } + valued = append(valued, out) + } + + switch len(valued) { + case 0: + return calcOutput{}, fmt.Errorf("%w: %s ended without a return", ErrCalcNoReturn, shape.Label) + case 1: + return valued[0], nil + default: + names := make([]string, 0, len(valued)) + for _, out := range valued { + names = append(names, out.Name) + } + return calcOutput{}, fmt.Errorf( + "%w: %s computes %d output features (%s) and designates no result; read them from a usage instead: %s", + ErrAmbiguousResult, shape.Label, len(valued), strings.Join(names, ", "), shape.usageSpelling(valued[0].Name), + ) } - return run.value(ctx, out) } // runCalcSteps runs the calc's lowered steps on engine, whose data holds the diff --git a/internal/exec/runtime/robustness_toolcalc_test.go b/internal/exec/runtime/robustness_toolcalc_test.go index a7907d0c58..9c4258ce56 100644 --- a/internal/exec/runtime/robustness_toolcalc_test.go +++ b/internal/exec/runtime/robustness_toolcalc_test.go @@ -59,6 +59,19 @@ func TestRuntimeRobustnessToolCalc(t *testing.T) { t.Fatalf("tool ran %d times under an ambiguous variable", len(runner.calls)) } }) + t.Run("ambiguous_result", func(t *testing.T) { + // Two bare outs answered by the tool designate no result, as two bound + // outputs do for a body that returned nothing. + ctx, scope := analysisFixture(t, toolCalcModel) + ctx.SetToolRunner(&recordingRunner{answer: map[string]ToolValue{ + "ok": {Value: semanticsBool(true)}, + "n": {Value: toolReal(3)}, + }}) + _, err := ctx.InvokeCalc(calcNamed(t, scope, "TwoOuts"), []Value{realOf(1)}, scope) + if !errors.Is(err, ErrAmbiguousResult) { + t.Fatalf("InvokeCalc = %v, want ErrAmbiguousResult", err) + } + }) t.Run("unanswered_output", func(t *testing.T) { // A reply the tool bound before the run ended reaches no binding: the // output read asks the run, which answers the reply's failure as a typed diff --git a/internal/exec/runtime/tool_calc_test.go b/internal/exec/runtime/tool_calc_test.go index 39601ddb5a..4b021b07e5 100644 --- a/internal/exec/runtime/tool_calc_test.go +++ b/internal/exec/runtime/tool_calc_test.go @@ -56,6 +56,23 @@ const toolCalcModel = `package test { in p = 10 [SI::W]; } + calc def Probe { + metadata ToolExecution { toolName = "Thermo"; uri = "p"; } + in a : Real { @ToolVariable { name = "a"; } } + out ok : Boolean { @ToolVariable { name = "ok"; } } + } + + calc def TwoOuts { + metadata ToolExecution { toolName = "Thermo"; uri = "t"; } + in a : Real { @ToolVariable { name = "a"; } } + out ok : Boolean { @ToolVariable { name = "ok"; } } + out n : Real { @ToolVariable { name = "n"; } } + } + + calc p : Probe { + in a = 1; + } + part def Board { attribute mass : MassValue = 2 [SI::kg]; attribute power : PowerValue = 10 [SI::W]; @@ -208,6 +225,50 @@ func TestToolCalcOutOnlyAsksNoResult(t *testing.T) { } } +// A calc declaring a bare `out` and no result parameter yields that output as +// its invocation's result, as a body returning nothing yields its designated +// output: the one output the tool answered, or the ambiguity the tool's answers +// leave when it bound several. +func TestToolCalcBareOutDesignatesTheAnsweredOutput(t *testing.T) { + t.Run("single out", func(t *testing.T) { + ctx, scope := analysisFixture(t, toolCalcModel) + ctx.SetToolRunner(&recordingRunner{answer: map[string]ToolValue{ + "ok": {Value: semanticsBool(true)}, + }}) + result, err := ctx.InvokeCalc(calcNamed(t, scope, "Probe"), []Value{realOf(1)}, scope) + if err != nil { + t.Fatalf("InvokeCalc: %v", err) + } + if got := FormatValue(result); got != "true" { + t.Fatalf("result = %s, want true", got) + } + }) + t.Run("two outs", func(t *testing.T) { + ctx, scope := analysisFixture(t, toolCalcModel) + ctx.SetToolRunner(&recordingRunner{answer: map[string]ToolValue{ + "ok": {Value: semanticsBool(true)}, + "n": {Value: toolReal(3)}, + }}) + _, err := ctx.InvokeCalc(calcNamed(t, scope, "TwoOuts"), []Value{realOf(1)}, scope) + if !errors.Is(err, ErrAmbiguousResult) { + t.Fatalf("InvokeCalc = %v, want ErrAmbiguousResult", err) + } + }) + t.Run("usage still reads the output", func(t *testing.T) { + ctx, scope := analysisFixture(t, toolCalcModel) + ctx.SetToolRunner(&recordingRunner{answer: map[string]ToolValue{ + "ok": {Value: semanticsBool(true)}, + }}) + value, err := ctx.CalcUsageOutput(calcNamed(t, scope, "p"), "ok", scope, nil) + if err != nil { + t.Fatalf("CalcUsageOutput: %v", err) + } + if got := FormatValue(value); got != "true" { + t.Fatalf("p.ok = %s, want true", got) + } + }) +} + // The answer's unit is converted to the result parameter's coherent unit; one of // another dimension is the reply's failure, a malformed ToolError. func TestToolCalcConvertsTheAnsweredUnit(t *testing.T) { From 04cf5d17526e674945c26a196dfe28c8f714bed5 Mon Sep 17 00:00:00 2001 From: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Fri, 25 Sep 2026 08:10:45 +0000 Subject: [PATCH 04/10] fix(grpc): bind the tool runner to the request context MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit newRuntimeContext bound the runner to context.Background(), so a cancelled RPC left a tool subprocess running — under the held-object lock for the cached population. The request context is threaded through newRuntime, newRuntimeOver, newRuntimeContext and newVerifyContext; model().Fresh and OpenSession bind to the background since plans rebind the runner on the worker and a session runtime outlives a call, and heldObjects rebinds its cached runtime's runner to the request inside lock(). Co-Authored-By: jason.han --- internal/frontend/grpc/analysis.go | 2 +- internal/frontend/grpc/budget_test.go | 3 +- .../frontend/grpc/convert_enum_valued_test.go | 2 +- .../frontend/grpc/convert_function_test.go | 6 +- internal/frontend/grpc/docquery.go | 4 +- .../frontend/grpc/evaluate_retention_test.go | 2 +- internal/frontend/grpc/objects.go | 22 +++++- internal/frontend/grpc/quantity_test.go | 2 +- internal/frontend/grpc/runtime_source_test.go | 2 +- internal/frontend/grpc/service.go | 37 +++++----- internal/frontend/grpc/session.go | 4 +- internal/frontend/grpc/sweep.go | 2 +- internal/frontend/grpc/validate.go | 2 +- internal/frontend/grpc/verify.go | 12 ++-- tests/grpc/toolcalc_cancel_test.go | 70 +++++++++++++++++++ 15 files changed, 133 insertions(+), 39 deletions(-) create mode 100644 tests/grpc/toolcalc_cancel_test.go diff --git a/internal/frontend/grpc/analysis.go b/internal/frontend/grpc/analysis.go index 85fbf56260..943e18ad4a 100644 --- a/internal/frontend/grpc/analysis.go +++ b/internal/frontend/grpc/analysis.go @@ -26,7 +26,7 @@ func (s *Service) RunAnalysis(ctx context.Context, req *pb.RunAnalysisRequest) ( if err := s.requireCapability(CapabilityVerification); err != nil { return nil, err } - v, err := s.newVerifyContext(req.ModelHash, req.Engine) + v, err := s.newVerifyContext(ctx, req.ModelHash, req.Engine) if err != nil { return nil, err } diff --git a/internal/frontend/grpc/budget_test.go b/internal/frontend/grpc/budget_test.go index 96c2f85897..553ec4a022 100644 --- a/internal/frontend/grpc/budget_test.go +++ b/internal/frontend/grpc/budget_test.go @@ -1,6 +1,7 @@ package grpc import ( + "context" "strings" "testing" @@ -44,7 +45,7 @@ func TestNewServiceResolvesBudgets(t *testing.T) { if err != nil { t.Fatalf("NewService: %v", err) } - ctx, _ := svc.newRuntime(&CachedModel{Index: symbols.NewIndex()}) + ctx, _ := svc.newRuntime(context.Background(), &CachedModel{Index: symbols.NewIndex()}) if got := ctx.Budgets(); got != svc.budgets { t.Errorf("context bounds = %+v, want the service's %+v", got, svc.budgets) } diff --git a/internal/frontend/grpc/convert_enum_valued_test.go b/internal/frontend/grpc/convert_enum_valued_test.go index f847fa583f..ea1ce23491 100644 --- a/internal/frontend/grpc/convert_enum_valued_test.go +++ b/internal/frontend/grpc/convert_enum_valued_test.go @@ -113,7 +113,7 @@ func TestScalarValuedEnumLiteralRoundTrip(t *testing.T) { if !ok { t.Fatal("model not cached") } - rt, _ := srv.newRuntime(cached) + rt, _ := srv.newRuntime(context.Background(), cached) idx := cached.Index sem := rt.Semantics() diff --git a/internal/frontend/grpc/convert_function_test.go b/internal/frontend/grpc/convert_function_test.go index 91e8fbdd8d..b758bd0818 100644 --- a/internal/frontend/grpc/convert_function_test.go +++ b/internal/frontend/grpc/convert_function_test.go @@ -114,7 +114,7 @@ func TestFunctionRoundTrip(t *testing.T) { t.Errorf("%s = %v, want calc_id %q closing over no object", expr, fn, want) } - rt, _ := srv.newRuntime(cached) + rt, _ := srv.newRuntime(context.Background(), cached) back, err := protoconv.ProtoToRuntimeValue(rt, pv, idx, sem) if err != nil { t.Fatalf("protoconv.ProtoToRuntimeValue(%s): %v", expr, err) @@ -141,7 +141,7 @@ func TestFunctionRoundTrip(t *testing.T) { "set in sequence": sequenceOf(setOf(sqCube...)), "sequence in set": setOf(sequenceOf(sqCube...)), } { - rt, _ := srv.newRuntime(cached) + rt, _ := srv.newRuntime(context.Background(), cached) back, err := protoconv.ProtoToRuntimeValue(rt, nested, idx, sem) if err != nil { t.Fatalf("protoconv.ProtoToRuntimeValue(functions in a %s): %v", name, err) @@ -287,7 +287,7 @@ func TestMalformedFunctionsAreRejected(t *testing.T) { modelHash := mustParse(t, srv, functionWireModel) cached, _ := srv.cache.Get(modelHash) idx, sem := cached.Index, NewSymbolContext(cached.Index).Semantics - rt, _ := srv.newRuntime(cached) + rt, _ := srv.newRuntime(context.Background(), cached) cases := []struct { name string diff --git a/internal/frontend/grpc/docquery.go b/internal/frontend/grpc/docquery.go index b4ce12804d..9f11581001 100644 --- a/internal/frontend/grpc/docquery.go +++ b/internal/frontend/grpc/docquery.go @@ -37,7 +37,7 @@ func (s *Service) RunDocumentQuery(ctx context.Context, req *pb.RunDocumentQuery // The query runs over the model's runtime and the objects it holds, as // %run-query runs over the session's; Objects answers no rows while none are held. held := s.objects(cached) - defer held.lock()() + defer held.lock(ctx)() qctx := held.queryContext() sym, err := documentSymbol(qctx.Index, req.QueryId) if err != nil { @@ -97,7 +97,7 @@ func (s *Service) RenderDocument(ctx context.Context, req *pb.RenderDocumentRequ // A document reads the objects the model holds, as -render-document reads // the ones -instantiate created beside it. held := s.objects(cached) - defer held.lock()() + defer held.lock(ctx)() qctx := held.queryContext() sym, err := documentSymbol(qctx.Index, req.DocumentId) if err != nil { diff --git a/internal/frontend/grpc/evaluate_retention_test.go b/internal/frontend/grpc/evaluate_retention_test.go index eb6f3dac2d..458d15db6c 100644 --- a/internal/frontend/grpc/evaluate_retention_test.go +++ b/internal/frontend/grpc/evaluate_retention_test.go @@ -204,7 +204,7 @@ package Demo { if got := worker.Semantics().MemoSize(); got != 0 { t.Fatalf("a new worker starts with %d selections, want none", got) } - rt, release := srv.newRuntime(cached) + rt, release := srv.newRuntime(context.Background(), cached) defer release() warm := rt.Semantics().MemoSize() if warm == 0 { diff --git a/internal/frontend/grpc/objects.go b/internal/frontend/grpc/objects.go index 4d78ac582c..d0588f29be 100644 --- a/internal/frontend/grpc/objects.go +++ b/internal/frontend/grpc/objects.go @@ -1,6 +1,7 @@ package grpc import ( + "context" "errors" "fmt" "math" @@ -15,6 +16,7 @@ import ( pb "github.com/Open-MBEE/OpenSysML/api/proto" "github.com/Open-MBEE/OpenSysML/internal/doc/queryexec" + "github.com/Open-MBEE/OpenSysML/internal/exec/analysis" "github.com/Open-MBEE/OpenSysML/internal/exec/objref" "github.com/Open-MBEE/OpenSysML/internal/exec/runtime" "github.com/Open-MBEE/OpenSysML/internal/semantic/symbols" @@ -72,6 +74,9 @@ type heldObjects struct { // displaced are objects a later Instantiate of their name displaced, still // roots of their own reached by id. displaced []*runtime.Instance + // runner binds the cached runtime's tool runner to the request under lock, + // so a request's end ends the tool it started through a held feature value. + runner func(context.Context) runtime.ToolRunner } // objects is the population held for cached, built by the first Instantiate @@ -81,23 +86,34 @@ func (s *Service) objects(cached *CachedModel) *heldObjects { defer cached.objectsMu.Unlock() if cached.objects == nil { model, _ := cached.Semantics() - rt := s.newRuntimeContext(model) + // The held runtime outlives any one call: lock rebinds the runner to the + // request holding the lock. + rt := s.newRuntimeContext(context.Background(), model) rt.SetMaxInstances(s.maxHeldObjects) // Events reads the population's run, so its records are kept from the start. rt.SetTrace(runtime.NewEventRecorder(s.maxHeldEvents)) + schedule := rt.Schedule() + if schedule == (runtime.SchedulePolicy{}) { + schedule = runtime.DefaultSchedulePolicy + } cached.objects = &heldObjects{ rt: rt, idx: cached.Index, named: make(map[string]*runtime.Instance), + runner: func(ctx context.Context) runtime.ToolRunner { + return s.engines.ToolRunner(ctx, rt, analysis.BudgetOf(s.budgets, schedule, analysis.Compute, s.jobs), analysis.Auto()) + }, } } return cached.objects } -// lock takes exclusive use of the population and its runtime, and returns the +// lock takes exclusive use of the population and its runtime, binding the +// runtime's tool runner to the request holding the lock, and returns the // function releasing it. -func (h *heldObjects) lock() func() { +func (h *heldObjects) lock(ctx context.Context) func() { h.mu.Lock() + h.rt.SetToolRunner(h.runner(ctx)) return h.mu.Unlock } diff --git a/internal/frontend/grpc/quantity_test.go b/internal/frontend/grpc/quantity_test.go index 978a922262..5a6110063f 100644 --- a/internal/frontend/grpc/quantity_test.go +++ b/internal/frontend/grpc/quantity_test.go @@ -205,7 +205,7 @@ func TestSetOfPointsReadForARuntime(t *testing.T) { &pb.Value{Kind: &pb.Value_Quantity{Quantity: celsius}}, ) - rt, _ := srv.newRuntime(cached) + rt, _ := srv.newRuntime(context.Background(), cached) if _, err := protoconv.ProtoToRuntimeValue(rt, sent, idx, sem); !errors.Is(err, protoconv.ErrSetElementRepeated) { t.Errorf("protoconv.ProtoToRuntimeValue({293.15 K, 20.0 °C_abs}) = %v, want %v", err, protoconv.ErrSetElementRepeated) } diff --git a/internal/frontend/grpc/runtime_source_test.go b/internal/frontend/grpc/runtime_source_test.go index 27d2c1feec..08b33dd154 100644 --- a/internal/frontend/grpc/runtime_source_test.go +++ b/internal/frontend/grpc/runtime_source_test.go @@ -36,7 +36,7 @@ package Demo { var last *runtime.Context for i := 0; i < 3; i++ { - ctx, _ := srv.newRuntime(cached) + ctx, _ := srv.newRuntime(context.Background(), cached) if last != nil && (ctx.Semantics() == last.Semantics() || ctx.Resolver() == last.Resolver()) { t.Fatalf("runtime %d: shares its resolver or semantic model with the one before", i) } diff --git a/internal/frontend/grpc/service.go b/internal/frontend/grpc/service.go index 6210b9f19b..76d1376f99 100644 --- a/internal/frontend/grpc/service.go +++ b/internal/frontend/grpc/service.go @@ -438,32 +438,33 @@ func (s *Service) requireValueCapabilities(pv *pb.Value) error { // newRuntime returns a runtime context under the service's budgets on a worker the request holds // alone, so concurrent requests share nothing mutable; the deferred release hands the worker on warm. -func (s *Service) newRuntime(cached *CachedModel) (*runtime.Context, func()) { +func (s *Service) newRuntime(ctx context.Context, cached *CachedModel) (*runtime.Context, func()) { w, release := cached.worker() - return s.newRuntimeOver(w), release + return s.newRuntimeOver(ctx, w), release } // newRuntimeOver builds a runtime context under the service's budgets on a worker; // every explored run gets one of its own. -func (s *Service) newRuntimeOver(w *analysis.Worker) *runtime.Context { - return s.newRuntimeContext(w.Model) +func (s *Service) newRuntimeOver(ctx context.Context, w *analysis.Worker) *runtime.Context { + return s.newRuntimeContext(ctx, w.Model) } -// newRuntimeContext builds a runtime context over model under the service's budgets. -func (s *Service) newRuntimeContext(model *runtime.Model) *runtime.Context { - ctx := runtime.NewContext(model, s.budgets.MaxSteps) - if err := ctx.SetBudgets(s.budgets); err != nil { +// newRuntimeContext builds a runtime context over model under the service's budgets, +// binding the tool runner to ctx so a request's end ends a tool it started. +func (s *Service) newRuntimeContext(ctx context.Context, model *runtime.Model) *runtime.Context { + rt := runtime.NewContext(model, s.budgets.MaxSteps) + if err := rt.SetBudgets(s.budgets); err != nil { // Unreachable: NewService validated these budgets. panic(fmt.Sprintf("grpc: invalid service budgets: %v", err)) } // The runner puts tool-computed actions and calcs of contexts held outside a plan — // feature values, documents, calc usages — to the engines as Compute questions. - schedule := ctx.Schedule() + schedule := rt.Schedule() if schedule == (runtime.SchedulePolicy{}) { schedule = runtime.DefaultSchedulePolicy } - ctx.SetToolRunner(s.engines.ToolRunner(context.Background(), ctx, analysis.BudgetOf(s.budgets, schedule, analysis.Compute, s.jobs), analysis.Auto())) - return ctx + rt.SetToolRunner(s.engines.ToolRunner(ctx, rt, analysis.BudgetOf(s.budgets, schedule, analysis.Compute, s.jobs), analysis.Auto())) + return rt } // model is the cached model as the engines reach it: a worker per plan over the shared @@ -471,7 +472,11 @@ func (s *Service) newRuntimeContext(model *runtime.Model) *runtime.Context { func (s *Service) model(cached *CachedModel) *analysis.Model { return &analysis.Model{ Semantics: cached.Semantics, - Fresh: func(w *analysis.Worker) (*runtime.Context, error) { return s.newRuntimeOver(w), nil }, + // Plans rebind the runner on the worker themselves, so the fresh context + // binds to the background. + Fresh: func(w *analysis.Worker) (*runtime.Context, error) { + return s.newRuntimeOver(context.Background(), w), nil + }, } } @@ -823,7 +828,7 @@ func (s *Service) Evaluate(ctx context.Context, req *pb.EvaluateRequest) (*pb.Ev scope = cached.PrimaryRoot() } - runtimeCtx, release := s.newRuntime(cached) + runtimeCtx, release := s.newRuntime(ctx, cached) defer release() var self *runtime.Instance @@ -890,7 +895,7 @@ func (s *Service) Instantiate(ctx context.Context, req *pb.InstantiateRequest) ( // The object outlives the request: a later RunDocumentQuery on the model // binds it by id or by the name it was created under. held := s.objects(cached) - defer held.lock()() + defer held.lock(ctx)() runtimeCtx := held.rt // Serializing the graph materializes the objects under the root, so it is @@ -948,7 +953,7 @@ func (s *Service) ExecuteAction(ctx context.Context, req *pb.ExecuteActionReques } action := syms[0] - runtimeCtx, release := s.newRuntime(cached) + runtimeCtx, release := s.newRuntime(ctx, cached) defer release() // Converted against the model's index, so a quantity input keeps the base @@ -1098,7 +1103,7 @@ func (s *Service) ExecuteState(ctx context.Context, req *pb.ExecuteStateRequest) return &pb.ExecuteStateResponse{Outcomes: x.outcomes, Exploration: x.status}, nil } - runtimeCtx, release := s.newRuntime(cached) + runtimeCtx, release := s.newRuntime(ctx, cached) defer release() if err := runtimeCtx.SetSchedule(schedule); err != nil { return nil, statusError(connect.CodeInvalidArgument, err.Error()) diff --git a/internal/frontend/grpc/session.go b/internal/frontend/grpc/session.go index 607c4f23dd..adfa5a7fed 100644 --- a/internal/frontend/grpc/session.go +++ b/internal/frontend/grpc/session.go @@ -1,6 +1,7 @@ package grpc import ( + "context" "errors" "fmt" "math" @@ -157,7 +158,8 @@ func (s *Service) OpenSession(modelHash string) (*Session, error) { return nil, statusErrorf(connect.CodeNotFound, msgModelNotFound, modelHash) } w, release := cached.worker() - rt := s.newRuntimeOver(w) + // A session runtime outlives any one call; plans rebind the runner on the worker. + rt := s.newRuntimeOver(context.Background(), w) rt.SetMaxInstances(s.maxHeldObjects) return &Session{svc: s, cached: cached, worker: w, release: release, rt: rt}, nil } diff --git a/internal/frontend/grpc/sweep.go b/internal/frontend/grpc/sweep.go index 123b8149e3..bc59b6d35e 100644 --- a/internal/frontend/grpc/sweep.go +++ b/internal/frontend/grpc/sweep.go @@ -28,7 +28,7 @@ func (s *Service) RunSweep(ctx context.Context, req *pb.RunSweepRequest) (*pb.Ru if err := s.requireCapability(CapabilityVerification); err != nil { return nil, err } - v, err := s.newVerifyContext(req.ModelHash, req.Engine) + v, err := s.newVerifyContext(ctx, req.ModelHash, req.Engine) if err != nil { return nil, err } diff --git a/internal/frontend/grpc/validate.go b/internal/frontend/grpc/validate.go index e32b3181ff..433ac13c86 100644 --- a/internal/frontend/grpc/validate.go +++ b/internal/frontend/grpc/validate.go @@ -19,7 +19,7 @@ func (s *Service) ValidateInstance(ctx context.Context, req *pb.ValidateInstance if err := s.requireCapability(CapabilityVerification); err != nil { return nil, err } - v, err := s.newVerifyContext(req.ModelHash, req.Engine) + v, err := s.newVerifyContext(ctx, req.ModelHash, req.Engine) if err != nil { return nil, err } diff --git a/internal/frontend/grpc/verify.go b/internal/frontend/grpc/verify.go index ecc609b572..fc9877fb5a 100644 --- a/internal/frontend/grpc/verify.go +++ b/internal/frontend/grpc/verify.go @@ -58,7 +58,7 @@ type verifyContext struct { // newVerifyContext reads the request's engine, looks the model up and builds a // runtime over it, the same way every other runtime RPC in this service does. -func (s *Service) newVerifyContext(modelHash, engine string) (*verifyContext, error) { +func (s *Service) newVerifyContext(ctx context.Context, modelHash, engine string) (*verifyContext, error) { selection, err := s.engineSelection(engine) if err != nil { return nil, err @@ -67,7 +67,7 @@ func (s *Service) newVerifyContext(modelHash, engine string) (*verifyContext, er if !ok { return nil, statusErrorf(connect.CodeNotFound, "model not found: %s", modelHash) } - rt, release := s.newRuntime(cached) + rt, release := s.newRuntime(ctx, cached) return &verifyContext{service: s, cached: cached, runtime: rt, engine: selection, release: release}, nil } @@ -262,7 +262,7 @@ func (s *Service) VerifyConstraint(ctx context.Context, req *pb.VerifyConstraint if err := s.requireCapability(CapabilityVerification); err != nil { return nil, err } - v, err := s.newVerifyContext(req.ModelHash, req.Engine) + v, err := s.newVerifyContext(ctx, req.ModelHash, req.Engine) if err != nil { return nil, err } @@ -295,7 +295,7 @@ func (s *Service) VerifyRequirement(ctx context.Context, req *pb.VerifyRequireme if err := s.requireCapability(CapabilityVerification); err != nil { return nil, err } - v, err := s.newVerifyContext(req.ModelHash, req.Engine) + v, err := s.newVerifyContext(ctx, req.ModelHash, req.Engine) if err != nil { return nil, err } @@ -332,7 +332,7 @@ func (s *Service) VerifySatisfaction(ctx context.Context, req *pb.VerifySatisfac if err := s.requireCapability(CapabilityVerification); err != nil { return nil, err } - v, err := s.newVerifyContext(req.ModelHash, req.Engine) + v, err := s.newVerifyContext(ctx, req.ModelHash, req.Engine) if err != nil { return nil, err } @@ -453,7 +453,7 @@ func (s *Service) EvaluateCalc(ctx context.Context, req *pb.EvaluateCalcRequest) if err := s.requireCapability(CapabilityVerification); err != nil { return nil, err } - v, err := s.newVerifyContext(req.ModelHash, req.Engine) + v, err := s.newVerifyContext(ctx, req.ModelHash, req.Engine) if err != nil { return nil, err } diff --git a/tests/grpc/toolcalc_cancel_test.go b/tests/grpc/toolcalc_cancel_test.go new file mode 100644 index 0000000000..90f9f7e965 --- /dev/null +++ b/tests/grpc/toolcalc_cancel_test.go @@ -0,0 +1,70 @@ +package grpc_test + +import ( + "context" + "os" + "path/filepath" + "strings" + "testing" + "time" + + pb "github.com/Open-MBEE/OpenSysML/api/proto" + "github.com/Open-MBEE/OpenSysML/internal/exec/analysis" + "github.com/Open-MBEE/OpenSysML/internal/frontend/grpc" +) + +// A request cancelled while a tool-computed calc's tool is running ends the +// tool rather than running its timeout out: the runner is bound to the +// request's context, not the background. +func TestGRPCToolCalcHonoursACancelledRequest(t *testing.T) { + t.Setenv("TOOL_CALC_MODE", "hang") + t.Setenv(analysis.ToolTimeoutEnv, "10s") + t.Setenv(analysis.ToolsEnv, conformanceToolManifest(t, []string{"toolcalc"})) + + srv, err := grpc.NewService(4, "test") + if err != nil { + t.Fatalf("NewService: %v", err) + } + modelData, err := os.ReadFile(filepath.Join("testdata", "conformance", "evaluate_calc_tool.sysml")) + if err != nil { + t.Fatalf("read model: %v", err) + } + parseResp, err := srv.ParseFile(context.Background(), &pb.ParseFileRequest{ + Source: &pb.ParseFileRequest_Content{Content: string(modelData)}, + ContentHash: "toolcalc-cancel", + }) + if err != nil { + t.Fatalf("ParseFile: %v", err) + } + for _, diag := range parseResp.Diagnostics { + if diag.Severity == "error" { + t.Fatalf("model has a diagnostic error: %s", diag.Message) + } + } + + ctx, cancel := context.WithCancel(context.Background()) + cancel() + start := time.Now() + resp, err := srv.EvaluateCalc(ctx, &pb.EvaluateCalcRequest{ + ModelHash: parseResp.ModelHash, + SymbolId: "thermoCase::Thermal", + Arguments: []*pb.Value{ + {Kind: &pb.Value_RealValue{RealValue: 2}}, + {Kind: &pb.Value_RealValue{RealValue: 10}}, + }, + }) + elapsed := time.Since(start) + if elapsed > 5*time.Second { + t.Fatalf("EvaluateCalc took %s on a cancelled request, want well under the 10s tool timeout", elapsed) + } + var failure string + switch { + case err != nil: + failure = err.Error() + case resp != nil: + failure = resp.Error + } + if !strings.Contains(failure, "cancel") { + t.Fatalf("EvaluateCalc failure = %q, want a cancellation error", failure) + } +} From c1a55384f35bf79c7557af102d11a352dac8d9d7 Mon Sep 17 00:00:00 2001 From: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Fri, 25 Sep 2026 08:31:41 +0000 Subject: [PATCH 05/10] fix(runtime): a tool-computed calc never performs its library general MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A bodiless calc def specializing a library function the runtime implements was answered by the built-in — the library-performance path ran before the shape was built and counted no tool. resolveLibraryPerformance treats a calc carrying ToolExecution as computing its own answer, so the shape path runs and the tool answers; a failed annotation read computes too, letting the error surface where the shape reports it. The same guard covers invocation expressions, which pick the same resolution. Co-Authored-By: jason.han --- internal/exec/runtime/invoke_calc.go | 5 +++ internal/exec/runtime/tool_calc_test.go | 54 +++++++++++++++++++++++++ 2 files changed, 59 insertions(+) diff --git a/internal/exec/runtime/invoke_calc.go b/internal/exec/runtime/invoke_calc.go index 5c9d2efe43..830786c47c 100644 --- a/internal/exec/runtime/invoke_calc.go +++ b/internal/exec/runtime/invoke_calc.go @@ -1061,6 +1061,11 @@ func (ctx *Context) resolveLibraryPerformance(sym *symbols.Symbol) *libraryPerfo if ctx.calcComputes(chain) { return nil } + // A tool-computed calc answers from its tool, not the library's implementation; + // a failed annotation read computes too, so the error surfaces on the shape path. + if tool, err := ctx.toolExecutionOf(sym); err != nil || tool != nil { + return nil + } lib := ctx.implementedLibraryCalc(sym) if lib == nil { return nil diff --git a/internal/exec/runtime/tool_calc_test.go b/internal/exec/runtime/tool_calc_test.go index 4b021b07e5..db3f5cb7fc 100644 --- a/internal/exec/runtime/tool_calc_test.go +++ b/internal/exec/runtime/tool_calc_test.go @@ -16,6 +16,7 @@ const toolCalcModel = `package test { private import ScalarValues::*; private import AnalysisTooling::*; private import ISQ::*; + private import RealFunctions::abs; calc def Thermal { metadata ToolExecution { toolName = "Thermo"; uri = "thermo://x"; } @@ -73,6 +74,15 @@ const toolCalcModel = `package test { in a = 1; } + calc def Custom :> abs { + metadata ToolExecution { toolName = "Thermo"; uri = "u"; } + in x :>> x { @ToolVariable { name = "x"; } } + } + + calc def UsesCustom { + return : Real = Custom(-4); + } + part def Board { attribute mass : MassValue = 2 [SI::kg]; attribute power : PowerValue = 10 [SI::W]; @@ -436,6 +446,50 @@ func TestToolCalcNotesDivergence(t *testing.T) { } } +// An annotated calc specializing a library function is computed by the tool, +// never by the library's implementation of what it specializes — whether the +// invocation goes through invokeCalc or an invocation expression — and a +// context with no runner refuses rather than answering from the library. +func TestToolCalcSpecializingALibraryFunctionComputesByTool(t *testing.T) { + t.Run("direct invocation", func(t *testing.T) { + ctx, scope := analysisFixture(t, toolCalcModel) + runner := &recordingRunner{answer: map[string]ToolValue{"result": {Value: toolReal(7)}}} + ctx.SetToolRunner(runner) + result, err := ctx.InvokeCalc(calcNamed(t, scope, "Custom"), []Value{realOf(-4)}, scope) + if err != nil { + t.Fatalf("InvokeCalc: %v", err) + } + if got := FormatValue(result); got != "7.0" { + t.Fatalf("result = %s, want the tool's 7.0, not abs's 4.0", got) + } + if len(runner.calls) != 1 { + t.Fatalf("tool invoked %d times, want once", len(runner.calls)) + } + if len(runner.calls[0].Inputs) != 1 || runner.calls[0].Inputs[0].Variable != "x" || + !nearly(runner.calls[0].Inputs[0].Value.Value, toolReal(-4)) { + t.Fatalf("inputs %+v, want x = -4", runner.calls[0].Inputs) + } + }) + t.Run("no runner", func(t *testing.T) { + ctx, scope := analysisFixture(t, toolCalcModel) + _, err := ctx.InvokeCalc(calcNamed(t, scope, "Custom"), []Value{realOf(-4)}, scope) + if !errors.Is(err, ErrToolNotRegistered) { + t.Fatalf("InvokeCalc = %v, want ErrToolNotRegistered, not abs's answer", err) + } + }) + t.Run("invocation expression", func(t *testing.T) { + ctx, scope := analysisFixture(t, toolCalcModel) + ctx.SetToolRunner(&recordingRunner{answer: map[string]ToolValue{"result": {Value: toolReal(7)}}}) + result, err := ctx.InvokeCalc(calcNamed(t, scope, "UsesCustom"), nil, scope) + if err != nil { + t.Fatalf("InvokeCalc: %v", err) + } + if got := FormatValue(result); got != "7.0" { + t.Fatalf("result = %s, want the tool's 7.0", got) + } + }) +} + // divergingCalcRunner answers the calc's call and reports the answer as changed. type divergingCalcRunner struct{} From c6e9588a4d63a59cad72c2efb3edae71d9b7d3f9 Mon Sep 17 00:00:00 2001 From: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Fri, 25 Sep 2026 08:32:11 +0000 Subject: [PATCH 06/10] fix(runtime): only a calc computes by tool; annotated cases keep their body calcInterfaceOf attached the annotation to every calc shape, so an analysis or verification case carrying ToolExecution would have skipped its body and steps for a tool. shape.Tool is recorded only for shapes whose kind is calc; cases run as before while the annotation is deferred. Co-Authored-By: jason.han --- internal/exec/runtime/invoke_calc.go | 11 +++++++---- internal/exec/runtime/tool_calc_test.go | 20 ++++++++++++++++++++ 2 files changed, 27 insertions(+), 4 deletions(-) diff --git a/internal/exec/runtime/invoke_calc.go b/internal/exec/runtime/invoke_calc.go index 830786c47c..8b0793ce05 100644 --- a/internal/exec/runtime/invoke_calc.go +++ b/internal/exec/runtime/invoke_calc.go @@ -256,11 +256,14 @@ func (ctx *Context) calcInterfaceOf(sym *symbols.Symbol) (*calcShape, error) { shape.BodyOutputs = assignedOutputs(shape.Steps, shape.Outputs, shape.Aliases) shape.Bindings = calcBindings(chain) shape.ResultExpr = resultBindingExpr(shape.Bindings) - tool, err := ctx.toolExecutionOf(sym) - if err != nil { - return nil, err + // Annotating cases is deferred; only a calc computes by tool. + if kind == "calc" { + tool, err := ctx.toolExecutionOf(sym) + if err != nil { + return nil, err + } + shape.Tool = tool } - shape.Tool = tool // A calc computes nothing unless it returns or binds an output; a case also // computes through its steps, or answers with its verdicts alone, and a // library function the runtime implements natively computes through that. diff --git a/internal/exec/runtime/tool_calc_test.go b/internal/exec/runtime/tool_calc_test.go index db3f5cb7fc..42821d5c4d 100644 --- a/internal/exec/runtime/tool_calc_test.go +++ b/internal/exec/runtime/tool_calc_test.go @@ -83,6 +83,13 @@ const toolCalcModel = `package test { return : Real = Custom(-4); } + analysis def Surveyed { + metadata ToolExecution { toolName = "Thermo"; uri = "u"; } + return r : Real = 42; + } + + analysis s : Surveyed; + part def Board { attribute mass : MassValue = 2 [SI::kg]; attribute power : PowerValue = 10 [SI::W]; @@ -490,6 +497,19 @@ func TestToolCalcSpecializingALibraryFunctionComputesByTool(t *testing.T) { }) } +// Annotating cases is deferred: an analysis case carrying ToolExecution runs +// its body and verdicts as before, needing no runner and refusing none. +func TestToolCalcAnnotatedCaseStillRunsItsBody(t *testing.T) { + ctx, scope := analysisFixture(t, toolCalcModel) + value, err := ctx.CalcUsageOutput(calcNamed(t, scope, "s"), "r", scope, nil) + if err != nil { + t.Fatalf("CalcUsageOutput: %v", err) + } + if got := FormatValue(value); got != "42" { + t.Fatalf("s.r = %s, want the body's 42", got) + } +} + // divergingCalcRunner answers the calc's call and reports the answer as changed. type divergingCalcRunner struct{} From 55095da4d5989f56f60cfac8a0b45e6e1a70b0a3 Mon Sep 17 00:00:00 2001 From: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Fri, 25 Sep 2026 08:32:45 +0000 Subject: [PATCH 07/10] docs(runtime): reword the tool-computed calc comment Co-Authored-By: jason.han --- internal/exec/runtime/invoke_calc.go | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/internal/exec/runtime/invoke_calc.go b/internal/exec/runtime/invoke_calc.go index 8b0793ce05..c331118523 100644 --- a/internal/exec/runtime/invoke_calc.go +++ b/internal/exec/runtime/invoke_calc.go @@ -256,7 +256,7 @@ func (ctx *Context) calcInterfaceOf(sym *symbols.Symbol) (*calcShape, error) { shape.BodyOutputs = assignedOutputs(shape.Steps, shape.Outputs, shape.Aliases) shape.Bindings = calcBindings(chain) shape.ResultExpr = resultBindingExpr(shape.Bindings) - // Annotating cases is deferred; only a calc computes by tool. + // Only a calc computes by tool; a case always runs its body. if kind == "calc" { tool, err := ctx.toolExecutionOf(sym) if err != nil { From fb0107fc60f2968ab98c84003c3f98d94dd07225 Mon Sep 17 00:00:00 2001 From: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Fri, 25 Sep 2026 19:38:09 +0000 Subject: [PATCH 08/10] fix(runtime): close the three bypasses around a tool-computed calc MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The compiled tier compiled a tool-annotated callee's body into callers that compilably invoked it, so the tool never ran: compileBatch.compile withdraws a tool shape as computed by tool, settle() then settles every caller to the evaluator, and invokeCalcShapeIn no longer needs the tool excluded from the compiled fast path — compiledCalcOf answers nil for the withdrawn shape. An optional input no argument binds arrived at the protocol as bound with a null value, which toolInput refuses: as an action's unbound optional sends nothing, a ToolVariable input whose held value is null on an optional parameter is simply not sent, while a required one keeps the typed refusal. A declared reader's context noted a runner's divergence on itself, where nothing reads it: a context seeded from another now forwards each note it makes to the one it was seeded from, so NewDeclaredReaderIn reports the divergence on the held runtime's Notes as a direct evaluation does. Co-Authored-By: jason.han --- internal/exec/runtime/choice.go | 3 + internal/exec/runtime/compile.go | 5 + internal/exec/runtime/context.go | 3 + internal/exec/runtime/declared.go | 5 +- internal/exec/runtime/invoke_calc.go | 2 +- .../exec/runtime/robustness_toolcalc_test.go | 11 +++ internal/exec/runtime/tool_calc.go | 7 +- internal/exec/runtime/tool_calc_test.go | 99 +++++++++++++++++++ 8 files changed, 130 insertions(+), 5 deletions(-) diff --git a/internal/exec/runtime/choice.go b/internal/exec/runtime/choice.go index f8f029c5fa..e4b2a4a9a9 100644 --- a/internal/exec/runtime/choice.go +++ b/internal/exec/runtime/choice.go @@ -275,6 +275,9 @@ func (ctx *Context) noteFrom(n RunNote, self *Instance, behavior *symbols.Symbol return } ctx.run.notes = append(ctx.run.notes, n) + if ctx.forwardNotes != nil { + ctx.forwardNotes(n) + } if c, ok := n.(ChoicePoint); ok { ctx.choices = append(ctx.choices, c.Choice()) } diff --git a/internal/exec/runtime/compile.go b/internal/exec/runtime/compile.go index 211a5829fb..43ed970e69 100644 --- a/internal/exec/runtime/compile.go +++ b/internal/exec/runtime/compile.go @@ -170,6 +170,11 @@ type compileBatch struct { // compile decides shape, compiling its callees first. func (b *compileBatch) compile(shape *calcShape) { + // A tool-computed calc has no body to compile; callers settle to the evaluator. + if shape.Tool != nil { + shape.withdraw("computed by tool " + shape.Tool.tool) + return + } shape.compileState = compileInProgress shape.compiled = &compiledCalc{kind: shape.Kind, name: shape.Name} c := &calcCompiler{batch: b, ctx: b.ctx, shape: shape} diff --git a/internal/exec/runtime/context.go b/internal/exec/runtime/context.go index 88690bf519..8385b7b338 100644 --- a/internal/exec/runtime/context.go +++ b/internal/exec/runtime/context.go @@ -90,6 +90,9 @@ type Context struct { metadataObjects map[metadataAnnotation]int64 // tools runs the external tool a ToolExecution names; nil refuses every such action. tools ToolRunner + // forwardNotes echoes each note to the context this one was seeded from + // (see DeclaredReader); nil keeps notes here alone. + forwardNotes func(RunNote) // variantObjects holds the object a variant stands for per owner that // selected it, so repeated reads of one selection read the same object. diff --git a/internal/exec/runtime/declared.go b/internal/exec/runtime/declared.go index 8bbf114d16..e69e60be9a 100644 --- a/internal/exec/runtime/declared.go +++ b/internal/exec/runtime/declared.go @@ -27,12 +27,13 @@ func NewDeclaredReader(model *semantics.Model, resolver *resolve.Resolver) *Decl // NewDeclaredReaderIn creates a reader on a fresh declarative context seeded // from held: its tool runner, so tool-computed calcs a derived feature calls -// answer through the same runner, and the parser reading the units they -// answer in. +// answer through the same runner, the parser reading the units they answer +// in, and the notes a run makes, which held's own Notes report. func NewDeclaredReaderIn(held *Context) *DeclaredReader { r := NewDeclaredReader(held.Semantics(), held.Resolver()) r.ctx.model.parse = held.model.parse r.ctx.SetToolRunner(held.ToolRunner()) + r.ctx.forwardNotes = held.note return r } diff --git a/internal/exec/runtime/invoke_calc.go b/internal/exec/runtime/invoke_calc.go index c331118523..cc4e791fc3 100644 --- a/internal/exec/runtime/invoke_calc.go +++ b/internal/exec/runtime/invoke_calc.go @@ -657,7 +657,7 @@ func (ctx *Context) invokeCalcShapeIn(shape *calcShape, args calcArgs, callerSco // sub-expression, an argument is not a scalar, a bound object may answer // a library constant the body reads before the library does, or the body // reads the bindings enclosing it. - if shape.Tool == nil && ctx.compileCalcs && ctx.trace == nil && len(enclosing) == 0 { + if ctx.compileCalcs && ctx.trace == nil && len(enclosing) == 0 { if compiled := ctx.compiledCalcOf(shape); compiled != nil && (self == nil || !compiled.readsLibrary) { if result, ran, err := compiled.invokeBoxed(ctx, args); ran { return result, err diff --git a/internal/exec/runtime/robustness_toolcalc_test.go b/internal/exec/runtime/robustness_toolcalc_test.go index 9c4258ce56..80c7f1c6c5 100644 --- a/internal/exec/runtime/robustness_toolcalc_test.go +++ b/internal/exec/runtime/robustness_toolcalc_test.go @@ -84,6 +84,17 @@ func TestRuntimeRobustnessToolCalc(t *testing.T) { t.Fatalf("CalcUsageOutput = %v, want ToolMissingOutput", err) } }) + t.Run("compiled_caller_no_runner", func(t *testing.T) { + // Under the compiled tier a caller of a tool calc settles to the + // evaluator, which refuses for want of a runner rather than running + // the compiled body it never made. + ctx, scope := analysisFixture(t, toolCalcModel) + ctx.SetCalcCompile(true) + _, err := ctx.InvokeCalc(calcNamed(t, scope, "Wrapper"), []Value{realOf(3)}, scope) + if !errors.Is(err, ErrToolNotRegistered) { + t.Fatalf("InvokeCalc = %v, want ErrToolNotRegistered", err) + } + }) } // dodgyCalcRunner answers the call's outputs except one, bypassing Bind's check, diff --git a/internal/exec/runtime/tool_calc.go b/internal/exec/runtime/tool_calc.go index a1f562b493..2570ff773f 100644 --- a/internal/exec/runtime/tool_calc.go +++ b/internal/exec/runtime/tool_calc.go @@ -55,11 +55,14 @@ func (ctx *Context) calcToolCall(shape *calcShape, scope *symbols.Scope, held fu writes := param.Direction == ast.DirOut || param.Direction == ast.DirInOut if reads { value, bound := held(name) - if !bound && !ctx.model.semantics.OptionalParameter(param.Symbol) { + optional := ctx.model.semantics.OptionalParameter(param.Symbol) + if !bound && !optional { return nil, "", fmt.Errorf("%w: %s: input parameter %s is bound by no argument", ErrUnboundParameter, shape.Label, name) } - if bound { + // An optional input bound to null is omitted: nothing is sent for it, + // as an action's unbound optional sends none. + if bound && (value.Kind != ValNull || !optional) { sent, err := toolInput(tool, param.Symbol, value) if err != nil { return nil, "", err diff --git a/internal/exec/runtime/tool_calc_test.go b/internal/exec/runtime/tool_calc_test.go index 42821d5c4d..9f9f894ffa 100644 --- a/internal/exec/runtime/tool_calc_test.go +++ b/internal/exec/runtime/tool_calc_test.go @@ -2,6 +2,7 @@ package runtime import ( "errors" + "fmt" "strings" "testing" @@ -83,6 +84,23 @@ const toolCalcModel = `package test { return : Real = Custom(-4); } + calc def External { + metadata ToolExecution { toolName = "Thermo"; uri = "u"; } + in x : Real { @ToolVariable { name = "x"; } } + return : Real = x * 2.0; + } + + calc def Wrapper { + in x : Real; + return : Real = External(x); + } + + calc def Biased { + metadata ToolExecution { toolName = "Thermo"; uri = "u"; } + in bias : Real [0..1] { @ToolVariable { name = "bias"; } } + return : Real; + } + analysis def Surveyed { metadata ToolExecution { toolName = "Thermo"; uri = "u"; } return r : Real = 42; @@ -497,6 +515,87 @@ func TestToolCalcSpecializingALibraryFunctionComputesByTool(t *testing.T) { }) } +// A compiled caller compiles its callees' bodies in, so a call of a calc a +// tool must compute compiles the body and runs it — unless compiling a +// tool-annotated shape withdraws it and settles the caller to the evaluator, +// which takes the call to the tool. Compiled or not, the answer is the tool's. +func TestToolCalcCompiledCallerGoesThroughTheTool(t *testing.T) { + for _, compile := range []bool{true, false} { + t.Run(fmt.Sprint("compile=", compile), func(t *testing.T) { + t.Run("answered", func(t *testing.T) { + ctx, scope := analysisFixture(t, toolCalcModel) + ctx.SetCalcCompile(compile) + runner := &recordingRunner{answer: map[string]ToolValue{"result": {Value: toolReal(42)}}} + ctx.SetToolRunner(runner) + result, err := ctx.InvokeCalc(calcNamed(t, scope, "Wrapper"), []Value{realOf(3)}, scope) + if err != nil { + t.Fatalf("InvokeCalc: %v", err) + } + if got := FormatValue(result); got != "42.0" { + t.Fatalf("Wrapper(3.0) = %s, want the tool's 42.0, not the body's 6.0", got) + } + if len(runner.calls) != 1 { + t.Fatalf("tool invoked %d times, want once", len(runner.calls)) + } + }) + t.Run("no runner", func(t *testing.T) { + ctx, scope := analysisFixture(t, toolCalcModel) + ctx.SetCalcCompile(compile) + _, err := ctx.InvokeCalc(calcNamed(t, scope, "Wrapper"), []Value{realOf(3)}, scope) + if !errors.Is(err, ErrToolNotRegistered) { + t.Fatalf("InvokeCalc = %v, want ErrToolNotRegistered, not the body's 6.0", err) + } + }) + }) + } +} + +// An optional input no argument binds is omitted from the call — the binder's +// null for it is not sent — while one bound by an argument is sent as usual. +func TestToolCalcOmitsAnOptionalNullInput(t *testing.T) { + t.Run("omitted", func(t *testing.T) { + ctx, scope := analysisFixture(t, toolCalcModel) + runner := &recordingRunner{answer: map[string]ToolValue{"result": {Value: toolReal(1)}}} + ctx.SetToolRunner(runner) + if _, err := ctx.InvokeCalc(calcNamed(t, scope, "Biased"), nil, scope); err != nil { + t.Fatalf("InvokeCalc: %v", err) + } + if len(runner.calls) != 1 || len(runner.calls[0].Inputs) != 0 { + t.Fatalf("calls %+v, want one call sending no input", runner.calls) + } + }) + t.Run("bound", func(t *testing.T) { + ctx, scope := analysisFixture(t, toolCalcModel) + runner := &recordingRunner{answer: map[string]ToolValue{"result": {Value: toolReal(1)}}} + ctx.SetToolRunner(runner) + if _, err := ctx.InvokeCalc(calcNamed(t, scope, "Biased"), []Value{realOf(2)}, scope); err != nil { + t.Fatalf("InvokeCalc: %v", err) + } + if len(runner.calls) != 1 || len(runner.calls[0].Inputs) != 1 || + runner.calls[0].Inputs[0].Variable != "bias" || !nearly(runner.calls[0].Inputs[0].Value.Value, toolReal(2)) { + t.Fatalf("calls %+v, want one call sending bias = 2", runner.calls) + } + }) +} + +// A derived feature the declared reader evaluates computes its tool calcs under +// the context the reader was seeded from, so the divergence a runner reports +// lands on that context's notes, not only the reader's own. +func TestToolCalcDeclaredReaderForwardsDivergence(t *testing.T) { + ctx, scope := analysisFixture(t, toolCalcModel) + ctx.SetToolRunner(divergingCalcRunner{}) + reader := NewDeclaredReaderIn(ctx) + if _, err := reader.Read(calcNamed(t, scope, "Board"), "Tmax"); err != nil { + t.Fatalf("Read: %v", err) + } + for _, note := range ctx.Notes() { + if d, ok := note.(ToolDivergence); ok && d.Tool == "Thermo" { + return + } + } + t.Fatalf("held notes = %v, want a ToolDivergence for Thermo", ctx.Notes()) +} + // Annotating cases is deferred: an analysis case carrying ToolExecution runs // its body and verdicts as before, needing no runner and refusing none. func TestToolCalcAnnotatedCaseStillRunsItsBody(t *testing.T) { From 69a9e10b9486523720f204e276d048866f60026c Mon Sep 17 00:00:00 2001 From: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Fri, 25 Sep 2026 19:59:27 +0000 Subject: [PATCH 09/10] fix(runtime): refuse a tool reply that leaves a calc output unanswered A runner answering without passing the reply through Bind could omit an output, and a direct invocation bound the zero Value for it; the missing output is refused as ToolMissingOutput, the same wording the usage path reports, before the reply binds anything. Co-Authored-By: jason.han --- .../exec/runtime/robustness_toolcalc_test.go | 14 ++++++++ internal/exec/runtime/tool_calc.go | 6 ++++ internal/exec/runtime/tool_calc_test.go | 36 +++++++++++++++++++ 3 files changed, 56 insertions(+) diff --git a/internal/exec/runtime/robustness_toolcalc_test.go b/internal/exec/runtime/robustness_toolcalc_test.go index 80c7f1c6c5..43581c3ccf 100644 --- a/internal/exec/runtime/robustness_toolcalc_test.go +++ b/internal/exec/runtime/robustness_toolcalc_test.go @@ -84,6 +84,20 @@ func TestRuntimeRobustnessToolCalc(t *testing.T) { t.Fatalf("CalcUsageOutput = %v, want ToolMissingOutput", err) } }) + t.Run("direct_result_unanswered", func(t *testing.T) { + // A runner answering the outputs and not the result, never passing + // Bind, is refused on the missing output: the direct invocation binds + // no zero value for what the tool left unanswered. + ctx, scope := analysisFixture(t, toolCalcRobustnessModel) + ctx.SetToolRunner(partialCalcRunner{outputs: map[string]Value{ + "warn": {Kind: ValConst, Const: semanticsBool(true)}, + }}) + _, err := ctx.InvokeCalc(calcNamed(t, scope, "Tw"), nil, scope) + var failure *ToolError + if !errors.As(err, &failure) || failure.Kind != ToolMissingOutput { + t.Fatalf("InvokeCalc = %v, want ToolMissingOutput", err) + } + }) t.Run("compiled_caller_no_runner", func(t *testing.T) { // Under the compiled tier a caller of a tool calc settles to the // evaluator, which refuses for want of a runner rather than running diff --git a/internal/exec/runtime/tool_calc.go b/internal/exec/runtime/tool_calc.go index 2570ff773f..5a3a2e9815 100644 --- a/internal/exec/runtime/tool_calc.go +++ b/internal/exec/runtime/tool_calc.go @@ -119,6 +119,12 @@ func (ctx *Context) computeCalcByTool(shape *calcShape, scope *symbols.Scope, he if err != nil { return Value{}, false, nil, err } + for _, out := range call.Outputs { + if _, answered := answer.Outputs[out.Parameter]; !answered { + return Value{}, false, nil, &ToolError{Tool: tool, Kind: ToolMissingOutput, + Detail: fmt.Sprintf("%s (%s of %s) was not answered", out.Parameter, out.Variable, shape.Label)} + } + } if answer.Diverged { ctx.note(ToolDivergence{ Tool: tool, diff --git a/internal/exec/runtime/tool_calc_test.go b/internal/exec/runtime/tool_calc_test.go index 9f9f894ffa..2e60ee9664 100644 --- a/internal/exec/runtime/tool_calc_test.go +++ b/internal/exec/runtime/tool_calc_test.go @@ -596,6 +596,42 @@ func TestToolCalcDeclaredReaderForwardsDivergence(t *testing.T) { t.Fatalf("held notes = %v, want a ToolDivergence for Thermo", ctx.Notes()) } +// A runner answering without Bind — as one is free to — that leaves one of the +// call's outputs unanswered is refused the same as a reply Bind would reject: +// the output it left out was answered by nothing, not bound to a zero value. +func TestToolCalcRefusesAnUnansweredResult(t *testing.T) { + t.Run("result unanswered", func(t *testing.T) { + ctx, scope := analysisFixture(t, toolCalcRobustnessModel) + ctx.SetToolRunner(partialCalcRunner{outputs: map[string]Value{ + "warn": {Kind: ValConst, Const: semanticsBool(true)}, + }}) + _, err := ctx.InvokeCalc(calcNamed(t, scope, "Tw"), nil, scope) + var failure *ToolError + if !errors.As(err, &failure) || failure.Kind != ToolMissingOutput { + t.Fatalf("InvokeCalc = %v, want ToolMissingOutput", err) + } + }) + t.Run("output unanswered", func(t *testing.T) { + ctx, scope := analysisFixture(t, toolCalcRobustnessModel) + ctx.SetToolRunner(partialCalcRunner{outputs: map[string]Value{ + "result": {Kind: ValConst, Const: toolReal(1)}, + }}) + _, err := ctx.InvokeCalc(calcNamed(t, scope, "Tw"), nil, scope) + var failure *ToolError + if !errors.As(err, &failure) || failure.Kind != ToolMissingOutput { + t.Fatalf("InvokeCalc = %v, want ToolMissingOutput", err) + } + }) +} + +// partialCalcRunner is a runner that answers exactly the outputs given, never +// passing them through the call's Bind check. +type partialCalcRunner struct{ outputs map[string]Value } + +func (r partialCalcRunner) RunTool(call *ToolCall) (ToolAnswer, error) { + return ToolAnswer{Outputs: r.outputs}, nil +} + // Annotating cases is deferred: an analysis case carrying ToolExecution runs // its body and verdicts as before, needing no runner and refusing none. func TestToolCalcAnnotatedCaseStillRunsItsBody(t *testing.T) { From d219b98401d968afbe446362fa4c297261645b62 Mon Sep 17 00:00:00 2001 From: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Sat, 26 Sep 2026 03:24:03 +0000 Subject: [PATCH 10/10] fix(runtime): carry held budgets into declared readers Document-derived calculations evaluated through NewDeclaredReaderIn now run under the held runtime's configured budgets instead of the defaults. Co-Authored-By: jason.han --- internal/exec/runtime/declared.go | 7 ++++++- internal/exec/runtime/tool_calc_test.go | 14 ++++++++++++++ 2 files changed, 20 insertions(+), 1 deletion(-) diff --git a/internal/exec/runtime/declared.go b/internal/exec/runtime/declared.go index e69e60be9a..c8ccad8213 100644 --- a/internal/exec/runtime/declared.go +++ b/internal/exec/runtime/declared.go @@ -28,10 +28,15 @@ func NewDeclaredReader(model *semantics.Model, resolver *resolve.Resolver) *Decl // NewDeclaredReaderIn creates a reader on a fresh declarative context seeded // from held: its tool runner, so tool-computed calcs a derived feature calls // answer through the same runner, the parser reading the units they answer -// in, and the notes a run makes, which held's own Notes report. +// in, the budgets it runs under, and the notes a run makes, which held's own +// Notes report. func NewDeclaredReaderIn(held *Context) *DeclaredReader { r := NewDeclaredReader(held.Semantics(), held.Resolver()) r.ctx.model.parse = held.model.parse + if err := r.ctx.SetBudgets(held.Budgets()); err != nil { + // held's budgets were validated when it was configured. + panic(err) + } r.ctx.SetToolRunner(held.ToolRunner()) r.ctx.forwardNotes = held.note return r diff --git a/internal/exec/runtime/tool_calc_test.go b/internal/exec/runtime/tool_calc_test.go index 2e60ee9664..ff4d031caf 100644 --- a/internal/exec/runtime/tool_calc_test.go +++ b/internal/exec/runtime/tool_calc_test.go @@ -596,6 +596,20 @@ func TestToolCalcDeclaredReaderForwardsDivergence(t *testing.T) { t.Fatalf("held notes = %v, want a ToolDivergence for Thermo", ctx.Notes()) } +// The reader runs under the budgets of the context it was seeded from. +func TestToolCalcDeclaredReaderCarriesBudgets(t *testing.T) { + ctx, _ := analysisFixture(t, toolCalcModel) + want := ctx.Budgets() + want.MaxSteps = 7 + want.MaxCalcDepth = 3 + if err := ctx.SetBudgets(want); err != nil { + t.Fatalf("SetBudgets: %v", err) + } + if got := NewDeclaredReaderIn(ctx).ctx.Budgets(); got != want { + t.Fatalf("reader budgets = %+v, want %+v", got, want) + } +} + // A runner answering without Bind — as one is free to — that leaves one of the // call's outputs unanswered is refused the same as a reply Bind would reject: // the output it left out was answered by nothing, not bound to a zero value.