@memberjunction/ai-agent-harness 0.0.0 → 6.1.0-edge.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +7 -0
- package/README.md +192 -27
- package/dist/HarnessAgentBase.d.ts +261 -0
- package/dist/HarnessAgentBase.d.ts.map +1 -0
- package/dist/HarnessAgentBase.js +822 -0
- package/dist/HarnessAgentBase.js.map +1 -0
- package/dist/HarnessAgentType.d.ts +39 -0
- package/dist/HarnessAgentType.d.ts.map +1 -0
- package/dist/HarnessAgentType.js +50 -0
- package/dist/HarnessAgentType.js.map +1 -0
- package/dist/adapters/BaseCliHarnessAdapter.d.ts +93 -0
- package/dist/adapters/BaseCliHarnessAdapter.d.ts.map +1 -0
- package/dist/adapters/BaseCliHarnessAdapter.js +184 -0
- package/dist/adapters/BaseCliHarnessAdapter.js.map +1 -0
- package/dist/adapters/BaseHarnessAdapter.d.ts +114 -0
- package/dist/adapters/BaseHarnessAdapter.d.ts.map +1 -0
- package/dist/adapters/BaseHarnessAdapter.js +86 -0
- package/dist/adapters/BaseHarnessAdapter.js.map +1 -0
- package/dist/adapters/ClaudeCodeCliAdapter.d.ts +104 -0
- package/dist/adapters/ClaudeCodeCliAdapter.d.ts.map +1 -0
- package/dist/adapters/ClaudeCodeCliAdapter.js +268 -0
- package/dist/adapters/ClaudeCodeCliAdapter.js.map +1 -0
- package/dist/adapters/CodexAdapter.d.ts +27 -0
- package/dist/adapters/CodexAdapter.d.ts.map +1 -0
- package/dist/adapters/CodexAdapter.js +117 -0
- package/dist/adapters/CodexAdapter.js.map +1 -0
- package/dist/adapters/GeminiCliAdapter.d.ts +25 -0
- package/dist/adapters/GeminiCliAdapter.d.ts.map +1 -0
- package/dist/adapters/GeminiCliAdapter.js +98 -0
- package/dist/adapters/GeminiCliAdapter.js.map +1 -0
- package/dist/adapters/OpenCodeAdapter.d.ts +23 -0
- package/dist/adapters/OpenCodeAdapter.d.ts.map +1 -0
- package/dist/adapters/OpenCodeAdapter.js +104 -0
- package/dist/adapters/OpenCodeAdapter.js.map +1 -0
- package/dist/adapters/PiAdapter.d.ts +73 -0
- package/dist/adapters/PiAdapter.d.ts.map +1 -0
- package/dist/adapters/PiAdapter.js +237 -0
- package/dist/adapters/PiAdapter.js.map +1 -0
- package/dist/adapters/StdioJsonAdapter.d.ts +43 -0
- package/dist/adapters/StdioJsonAdapter.d.ts.map +1 -0
- package/dist/adapters/StdioJsonAdapter.js +109 -0
- package/dist/adapters/StdioJsonAdapter.js.map +1 -0
- package/dist/index.d.ts +26 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +28 -0
- package/dist/index.js.map +1 -0
- package/dist/sandbox/ChildProcessExecutor.d.ts +41 -0
- package/dist/sandbox/ChildProcessExecutor.d.ts.map +1 -0
- package/dist/sandbox/ChildProcessExecutor.js +86 -0
- package/dist/sandbox/ChildProcessExecutor.js.map +1 -0
- package/dist/sandbox/DockerSandboxProvider.d.ts +64 -0
- package/dist/sandbox/DockerSandboxProvider.d.ts.map +1 -0
- package/dist/sandbox/DockerSandboxProvider.js +176 -0
- package/dist/sandbox/DockerSandboxProvider.js.map +1 -0
- package/dist/sandbox/ISandboxProvider.d.ts +71 -0
- package/dist/sandbox/ISandboxProvider.d.ts.map +1 -0
- package/dist/sandbox/ISandboxProvider.js +2 -0
- package/dist/sandbox/ISandboxProvider.js.map +1 -0
- package/dist/sandbox/LocalDirectorySandboxProvider.d.ts +37 -0
- package/dist/sandbox/LocalDirectorySandboxProvider.d.ts.map +1 -0
- package/dist/sandbox/LocalDirectorySandboxProvider.js +75 -0
- package/dist/sandbox/LocalDirectorySandboxProvider.js.map +1 -0
- package/dist/sandbox/SandboxExecutor.d.ts +47 -0
- package/dist/sandbox/SandboxExecutor.d.ts.map +1 -0
- package/dist/sandbox/SandboxExecutor.js +2 -0
- package/dist/sandbox/SandboxExecutor.js.map +1 -0
- package/dist/types.d.ts +180 -0
- package/dist/types.d.ts.map +1 -0
- package/dist/types.js +2 -0
- package/dist/types.js.map +1 -0
- package/package.json +35 -8
|
@@ -0,0 +1,822 @@
|
|
|
1
|
+
var __decorate = (this && this.__decorate) || function (decorators, target, key, desc) {
|
|
2
|
+
var c = arguments.length, r = c < 3 ? target : desc === null ? desc = Object.getOwnPropertyDescriptor(target, key) : desc, d;
|
|
3
|
+
if (typeof Reflect === "object" && typeof Reflect.decorate === "function") r = Reflect.decorate(decorators, target, key, desc);
|
|
4
|
+
else for (var i = decorators.length - 1; i >= 0; i--) if (d = decorators[i]) r = (c < 3 ? d(r) : c > 3 ? d(target, key, r) : d(target, key)) || r;
|
|
5
|
+
return c > 3 && r && Object.defineProperty(target, key, r), r;
|
|
6
|
+
};
|
|
7
|
+
import { MJGlobal, RegisterClass, UUIDsEqual } from '@memberjunction/global';
|
|
8
|
+
import { LogError, LogStatus, RunView } from '@memberjunction/core';
|
|
9
|
+
import { BaseAgent } from '@memberjunction/ai-agents';
|
|
10
|
+
import { TemplateEngineServer } from '@memberjunction/templates';
|
|
11
|
+
import { ChatResult, ModelUsage } from '@memberjunction/ai';
|
|
12
|
+
import { BaseHarnessAdapter } from './adapters/BaseHarnessAdapter.js';
|
|
13
|
+
import { LocalDirectorySandboxProvider } from './sandbox/LocalDirectorySandboxProvider.js';
|
|
14
|
+
import { DockerSandboxProvider } from './sandbox/DockerSandboxProvider.js';
|
|
15
|
+
/**
|
|
16
|
+
* Conventional environment variable each harness reads its own credential from.
|
|
17
|
+
*
|
|
18
|
+
* Only used by the zero-config vendor-key fallback. Explicit rather than derived because writing a
|
|
19
|
+
* key into the wrong variable name fails silently from MJ's side — the harness simply reports an
|
|
20
|
+
* auth error much later, with nothing pointing back at the mapping.
|
|
21
|
+
*/
|
|
22
|
+
const HARNESS_CREDENTIAL_ENV_VARS = {
|
|
23
|
+
ClaudeCodeCliAdapter: 'ANTHROPIC_API_KEY',
|
|
24
|
+
CodexAdapter: 'OPENAI_API_KEY',
|
|
25
|
+
GeminiCliAdapter: 'GEMINI_API_KEY',
|
|
26
|
+
};
|
|
27
|
+
/**
|
|
28
|
+
* Restated at the end of every turn input.
|
|
29
|
+
*
|
|
30
|
+
* Not redundant with the system prompt: a harness runs a full agentic loop inside its turn, so by
|
|
31
|
+
* the time it finishes working, the instruction it saw at the start is many internal steps behind
|
|
32
|
+
* it. Restating costs a few tokens; the alternative is a malformed-response retry, which costs a
|
|
33
|
+
* whole turn.
|
|
34
|
+
*/
|
|
35
|
+
const TURN_END_CONTRACT = [
|
|
36
|
+
'---',
|
|
37
|
+
'END OF TURN REQUIREMENT (this overrides any inclination to reply conversationally):',
|
|
38
|
+
'Respond with ONLY a single raw JSON object matching the response format you were given.',
|
|
39
|
+
'No prose, no markdown fences, no narration before or after. If the work is complete, say so',
|
|
40
|
+
'INSIDE the JSON. If you cannot proceed, say that inside the JSON too.',
|
|
41
|
+
].join('\n');
|
|
42
|
+
/**
|
|
43
|
+
* Executes an MJ agent whose reasoning substrate is an external harness.
|
|
44
|
+
*
|
|
45
|
+
* ## The single seam
|
|
46
|
+
*
|
|
47
|
+
* `BaseAgent` already runs an iterate → decide → execute-steps → iterate loop where the "decide"
|
|
48
|
+
* input is a prompt execution. This class overrides exactly one method — {@link executePrompt} — and
|
|
49
|
+
* substitutes a harness turn for that prompt call. Everything else is untouched `BaseAgent`: the
|
|
50
|
+
* loop, next-step validation, action and sub-agent execution, payload merging under ACLs, guardrail
|
|
51
|
+
* checks between iterations, and run-step recording.
|
|
52
|
+
*
|
|
53
|
+
* That is worth stating precisely because it is the whole architectural bet. `executePrompt` is a
|
|
54
|
+
* five-line protected method with a single call site, so substituting it changes what produces a
|
|
55
|
+
* decision without changing anything about how decisions are validated or enforced.
|
|
56
|
+
*
|
|
57
|
+
* ## Accounting is not optional
|
|
58
|
+
*
|
|
59
|
+
* A harness turn must produce a real `AIPromptRun` row. Run totals are DERIVED — `calculateTokenStats`
|
|
60
|
+
* sums `AIAgentRunStep.PromptRun` rollups — so a turn that records no prompt run contributes nothing,
|
|
61
|
+
* and the run reports zero tokens and zero cost forever. Combined with the cost guardrail, that means
|
|
62
|
+
* a runaway harness would never be interrupted: the ceiling would have nothing to compare against.
|
|
63
|
+
*
|
|
64
|
+
* `AIPromptRun.PromptID`, `.ModelID` and `.VendorID` are all NOT NULL, and every one is resolved to a
|
|
65
|
+
* REAL catalog row rather than a placeholder — see {@link resolveAccountingIds}.
|
|
66
|
+
*/
|
|
67
|
+
let HarnessAgentBase = class HarnessAgentBase extends BaseAgent {
|
|
68
|
+
constructor() {
|
|
69
|
+
super(...arguments);
|
|
70
|
+
this.adapter = null;
|
|
71
|
+
this.sandboxProvider = null;
|
|
72
|
+
this.sandboxHandle = null;
|
|
73
|
+
this.harnessRow = null;
|
|
74
|
+
this.turnIndex = 0;
|
|
75
|
+
}
|
|
76
|
+
/**
|
|
77
|
+
* Substitutes a harness turn for the prompt call a Loop agent would make.
|
|
78
|
+
*
|
|
79
|
+
* The returned {@link AIPromptRunResult} is shaped exactly as `AIPromptRunner` would shape one,
|
|
80
|
+
* because everything downstream — `DetermineNextStep`, the malformed-response retry machinery,
|
|
81
|
+
* step recording — reads it without knowing or caring that a harness produced it.
|
|
82
|
+
*/
|
|
83
|
+
async executePrompt(promptParams) {
|
|
84
|
+
const startTime = new Date();
|
|
85
|
+
try {
|
|
86
|
+
if (this.turnIndex === 0) {
|
|
87
|
+
await this.startHarnessSession(promptParams);
|
|
88
|
+
}
|
|
89
|
+
this.turnIndex++;
|
|
90
|
+
const isFirstTurn = this.turnIndex === 1;
|
|
91
|
+
const input = await this.buildTurnInput(promptParams, isFirstTurn);
|
|
92
|
+
const turn = await this.runTurn(input);
|
|
93
|
+
const promptRun = await this.recordPromptRun(promptParams, turn, startTime, input);
|
|
94
|
+
this.persistExternalSessionId(turn.SessionId);
|
|
95
|
+
return this.buildPromptResult(turn, promptRun, startTime);
|
|
96
|
+
}
|
|
97
|
+
catch (e) {
|
|
98
|
+
const message = describeError(e);
|
|
99
|
+
LogError(`Harness turn failed: ${message}`);
|
|
100
|
+
return this.buildPromptResult({ RawText: '', InputTokens: 0, OutputTokens: 0, ErrorMessage: message }, undefined, startTime);
|
|
101
|
+
}
|
|
102
|
+
}
|
|
103
|
+
/** Provisions the sandbox, resolves credentials and launches the harness session. */
|
|
104
|
+
async startHarnessSession(promptParams) {
|
|
105
|
+
const contextUser = promptParams.contextUser;
|
|
106
|
+
const config = this.readHarnessConfig();
|
|
107
|
+
const harness = await this.loadHarnessRow(config.harnessName, contextUser);
|
|
108
|
+
this.harnessRow = harness;
|
|
109
|
+
this.adapter = this.resolveAdapter(harness);
|
|
110
|
+
this.sandboxProvider = this.createSandboxProvider(config);
|
|
111
|
+
const key = {
|
|
112
|
+
Scope: config.sandbox?.workspaceScope ?? 'agent-user',
|
|
113
|
+
AgentId: this._agentRunAgentId(),
|
|
114
|
+
UserId: contextUser?.ID,
|
|
115
|
+
RunId: this._agentRunId(),
|
|
116
|
+
};
|
|
117
|
+
this.sandboxHandle = await this.sandboxProvider.Provision(key, {
|
|
118
|
+
NetworkPolicy: config.sandbox?.networkPolicy ?? 'mcp-only',
|
|
119
|
+
Image: config.sandbox?.image,
|
|
120
|
+
});
|
|
121
|
+
const environment = await this.resolveGrantedEnvironment(contextUser);
|
|
122
|
+
// Permissions are declared in agent metadata and applied BEFORE the session starts, so
|
|
123
|
+
// adapters can fold them into launch flags. Runtime overrides arrive through the same
|
|
124
|
+
// TypeConfiguration merge as every other harness setting, so a caller can loosen or tighten
|
|
125
|
+
// a single run without editing the agent.
|
|
126
|
+
const policy = this.resolvePermissionPolicy(config);
|
|
127
|
+
this.adapter.ApplyPermissionPolicy(policy);
|
|
128
|
+
this.warnOnUnenforceablePolicy(harness.Name, policy);
|
|
129
|
+
const resumeSessionId = await this.findResumableSession(harness, key.Scope, contextUser);
|
|
130
|
+
await this.adapter.StartSession({
|
|
131
|
+
ResumeSessionId: resumeSessionId,
|
|
132
|
+
PermissionPolicy: policy,
|
|
133
|
+
Executor: this.sandboxHandle.Executor,
|
|
134
|
+
WorkspacePath: this.sandboxHandle.WorkspacePath,
|
|
135
|
+
Environment: environment,
|
|
136
|
+
Model: harness.DefaultModel ?? undefined,
|
|
137
|
+
CancellationToken: promptParams.cancellationToken,
|
|
138
|
+
});
|
|
139
|
+
LogStatus(`Harness session started for '${harness.Name}' (${harness.DriverClass})`);
|
|
140
|
+
}
|
|
141
|
+
/**
|
|
142
|
+
* Resolves what the agent may do inside its sandbox.
|
|
143
|
+
*
|
|
144
|
+
* Defaults to `strict` when unset — the safe direction. An agent that has never been given a
|
|
145
|
+
* posture should be unable to mutate anything, rather than inheriting whatever the harness does
|
|
146
|
+
* by default, which for a coding agent is a great deal.
|
|
147
|
+
*/
|
|
148
|
+
resolvePermissionPolicy(config) {
|
|
149
|
+
return {
|
|
150
|
+
Posture: config.posture ?? 'strict',
|
|
151
|
+
AllowedTools: config.permissions?.allowedTools,
|
|
152
|
+
DisallowedTools: config.permissions?.disallowedTools,
|
|
153
|
+
};
|
|
154
|
+
}
|
|
155
|
+
/**
|
|
156
|
+
* Says out loud when a configured policy will not actually be enforced.
|
|
157
|
+
*
|
|
158
|
+
* Two distinct gaps, previously conflated behind one `PermissionHooks` check — which is why four
|
|
159
|
+
* adapters could ignore a policy entirely while the runtime warned about something else:
|
|
160
|
+
*
|
|
161
|
+
* 1. **`PermissionPolicy: false`** — the adapter never translated the policy into harness flags.
|
|
162
|
+
* The posture and allow/deny lists are inert; the harness runs on its own defaults. This is
|
|
163
|
+
* the serious one, because the agent's metadata reads as though something is gated.
|
|
164
|
+
* 2. **`PermissionHooks: false` under `strict`** — the policy applies, but there is no channel to
|
|
165
|
+
* route an approval through, so anything requiring one is denied rather than escalated.
|
|
166
|
+
*
|
|
167
|
+
* Warn, don't fail. Refusing the run would take every adapter without a verified flag vocabulary
|
|
168
|
+
* offline, and an unenforced policy on a properly-provisioned sandbox is still contained by the
|
|
169
|
+
* sandbox. What is not acceptable is the operator not knowing which situation they are in.
|
|
170
|
+
*/
|
|
171
|
+
warnOnUnenforceablePolicy(harnessName, policy) {
|
|
172
|
+
const capabilities = this.adapter?.Capabilities;
|
|
173
|
+
const policyWasConfigured = policy.Posture !== 'strict' ||
|
|
174
|
+
(policy.AllowedTools?.length ?? 0) > 0 ||
|
|
175
|
+
(policy.DisallowedTools?.length ?? 0) > 0;
|
|
176
|
+
if (!capabilities?.PermissionPolicy) {
|
|
177
|
+
LogError(`Harness '${harnessName}': its adapter does not apply MJ permission policies ` +
|
|
178
|
+
`(CapabilitySettings.PermissionPolicy is not true), so the ` +
|
|
179
|
+
`'${policy.Posture}' posture and any allow/deny lists are NOT enforced. The ` +
|
|
180
|
+
`harness runs on its own defaults — rely on the sandbox provider for containment.` +
|
|
181
|
+
(policyWasConfigured ? ' A policy IS configured on this agent and is being ignored.' : ''));
|
|
182
|
+
return;
|
|
183
|
+
}
|
|
184
|
+
if (policy.Posture === 'strict' && !capabilities.PermissionHooks) {
|
|
185
|
+
LogStatus(`Harness '${harnessName}' is set to STRICT posture but its adapter reports no ` +
|
|
186
|
+
`permission hooks. The harness's own prompts have nowhere to go headlessly, so ` +
|
|
187
|
+
`mutating tool calls will simply be denied rather than routed for approval.`);
|
|
188
|
+
}
|
|
189
|
+
}
|
|
190
|
+
/**
|
|
191
|
+
* Finds a prior harness session this run can continue, if the adapter can use one.
|
|
192
|
+
*
|
|
193
|
+
* ## Why this is worth doing
|
|
194
|
+
*
|
|
195
|
+
* Without it, every message in a conversation opens a COLD session and MJ replays the whole
|
|
196
|
+
* history into it. Measured on two consecutive messages in one conversation: the second cost
|
|
197
|
+
* $0.0448 against the first's $0.0155 — nearly 3x, spent entirely on re-reading context the
|
|
198
|
+
* harness had already been told once.
|
|
199
|
+
*
|
|
200
|
+
* ## Three gates, each guarding a different way this goes wrong
|
|
201
|
+
*
|
|
202
|
+
* 1. `SessionResume` capability — a harness that cannot resume must keep replaying. Offering a
|
|
203
|
+
* session id to an adapter that ignores it is harmless; ASSUMING it resumed is not, which is
|
|
204
|
+
* why the outcome is reported back rather than inferred.
|
|
205
|
+
* 2. Workspace scope must be durable. Harnesses key their session store by working directory, so
|
|
206
|
+
* a `run`-scoped workspace is a new directory every time and the session would never be
|
|
207
|
+
* found. Gating here keeps the failure at "no resume" rather than a silent miss.
|
|
208
|
+
* 3. Same conversation. That is the continuity boundary users already understand — a time-based
|
|
209
|
+
* cache would expire while someone is at lunch and, worse, leak stale context into an
|
|
210
|
+
* unrelated new conversation.
|
|
211
|
+
*/
|
|
212
|
+
async findResumableSession(harness, scope, contextUser) {
|
|
213
|
+
if (!this.adapter?.Capabilities?.SessionResume) {
|
|
214
|
+
return undefined;
|
|
215
|
+
}
|
|
216
|
+
if (scope === 'run') {
|
|
217
|
+
return undefined;
|
|
218
|
+
}
|
|
219
|
+
const conversationId = this._agentRunConversationId();
|
|
220
|
+
if (!conversationId) {
|
|
221
|
+
return undefined;
|
|
222
|
+
}
|
|
223
|
+
try {
|
|
224
|
+
const rv = new RunView();
|
|
225
|
+
const result = await rv.RunView({
|
|
226
|
+
EntityName: 'MJ: AI Agent Runs',
|
|
227
|
+
Fields: ['ExternalSessionID'],
|
|
228
|
+
ExtraFilter: `AgentID='${this._agentRunAgentId()}' AND ConversationID='${conversationId}' ` +
|
|
229
|
+
`AND ExternalSessionID IS NOT NULL`,
|
|
230
|
+
OrderBy: '__mj_CreatedAt DESC',
|
|
231
|
+
MaxRows: 1,
|
|
232
|
+
ResultType: 'simple',
|
|
233
|
+
}, contextUser);
|
|
234
|
+
if (!result.Success) {
|
|
235
|
+
LogError(`Failed to look up a resumable harness session: ${result.ErrorMessage}`);
|
|
236
|
+
return undefined;
|
|
237
|
+
}
|
|
238
|
+
const sessionId = result.Results?.[0]?.ExternalSessionID;
|
|
239
|
+
if (sessionId) {
|
|
240
|
+
LogStatus(`Resuming harness session ${sessionId} for '${harness.Name}'.`);
|
|
241
|
+
}
|
|
242
|
+
return sessionId;
|
|
243
|
+
}
|
|
244
|
+
catch (e) {
|
|
245
|
+
LogError(`Failed to look up a resumable harness session: ${describeError(e)}`);
|
|
246
|
+
return undefined;
|
|
247
|
+
}
|
|
248
|
+
}
|
|
249
|
+
/** Accumulates one turn's event stream into a single result. */
|
|
250
|
+
async runTurn(input) {
|
|
251
|
+
const adapter = this.adapter;
|
|
252
|
+
const turn = { RawText: '', InputTokens: 0, OutputTokens: 0 };
|
|
253
|
+
for await (const event of adapter.RunTurn(input)) {
|
|
254
|
+
switch (event.Type) {
|
|
255
|
+
case 'usage':
|
|
256
|
+
// Summed rather than replaced: a harness may report usage more than once per
|
|
257
|
+
// turn, and undercounting here silently weakens the cost guardrail.
|
|
258
|
+
turn.InputTokens += event.InputTokens;
|
|
259
|
+
turn.OutputTokens += event.OutputTokens;
|
|
260
|
+
turn.CostUsd = (turn.CostUsd ?? 0) + (event.CostUsd ?? 0);
|
|
261
|
+
break;
|
|
262
|
+
case 'turn-complete':
|
|
263
|
+
turn.RawText = event.RawText;
|
|
264
|
+
break;
|
|
265
|
+
case 'session-error':
|
|
266
|
+
turn.ErrorMessage = event.Error;
|
|
267
|
+
break;
|
|
268
|
+
default:
|
|
269
|
+
// assistant-text and sandbox-activity are live-view only. They are deliberately
|
|
270
|
+
// NOT persisted as run steps — the audit boundary is the turn, and widening it
|
|
271
|
+
// to in-sandbox activity is a documented non-goal, not an oversight.
|
|
272
|
+
break;
|
|
273
|
+
}
|
|
274
|
+
}
|
|
275
|
+
turn.SessionId = adapter.SessionId;
|
|
276
|
+
turn.ReportedModel = adapter.ReportedModel;
|
|
277
|
+
return turn;
|
|
278
|
+
}
|
|
279
|
+
/**
|
|
280
|
+
* Writes the `AIPromptRun` that carries this turn's usage.
|
|
281
|
+
*
|
|
282
|
+
* Returns undefined only when the row could not be created, which is logged loudly rather than
|
|
283
|
+
* swallowed: without it the run's cost and token totals stay at zero and its guardrails go blind.
|
|
284
|
+
*/
|
|
285
|
+
async recordPromptRun(promptParams, turn, startTime, input) {
|
|
286
|
+
try {
|
|
287
|
+
const ids = await this.resolveAccountingIds(promptParams);
|
|
288
|
+
if (!ids) {
|
|
289
|
+
LogError('Harness turn could not resolve a Prompt/Model/Vendor for accounting; this run will ' +
|
|
290
|
+
'under-report tokens and cost, and its cost guardrail will not fire. Set ' +
|
|
291
|
+
'AIAgentHarness.AIModelID and AIVendorID.');
|
|
292
|
+
return undefined;
|
|
293
|
+
}
|
|
294
|
+
const md = this.ProviderToUse;
|
|
295
|
+
const run = await md.GetEntityObject('MJ: AI Prompt Runs', promptParams.contextUser);
|
|
296
|
+
run.NewRecord();
|
|
297
|
+
run.PromptID = ids.PromptID;
|
|
298
|
+
run.ModelID = (await this.resolveReportedModelId(turn.ReportedModel, promptParams)) ?? ids.ModelID;
|
|
299
|
+
run.VendorID = ids.VendorID;
|
|
300
|
+
run.AgentID = this._agentRunAgentId();
|
|
301
|
+
run.RunAt = startTime;
|
|
302
|
+
run.CompletedAt = new Date();
|
|
303
|
+
run.Success = !turn.ErrorMessage;
|
|
304
|
+
run.Status = turn.ErrorMessage ? 'Failed' : 'Completed';
|
|
305
|
+
// Messages/Result are what the run-detail UI renders. AIPromptRunner populates them as
|
|
306
|
+
// a matter of course; synthesizing this row by hand reproduced the ACCOUNTING fields and
|
|
307
|
+
// dropped the OBSERVABILITY ones, so every harness prompt step showed blank input and
|
|
308
|
+
// output while its token and cost numbers were correct.
|
|
309
|
+
// JSON, not a raw string. AIPromptRun.Messages is documented as "the input messages sent
|
|
310
|
+
// to the model, typically in JSON format" and the run-detail UI parses it as such — a raw
|
|
311
|
+
// string parses to nothing, which is why the response rendered and the input did not.
|
|
312
|
+
run.Messages = JSON.stringify([{ role: 'user', content: input }]);
|
|
313
|
+
run.Result = turn.RawText;
|
|
314
|
+
run.TokensPrompt = turn.InputTokens;
|
|
315
|
+
run.TokensCompletion = turn.OutputTokens;
|
|
316
|
+
run.TokensUsed = turn.InputTokens + turn.OutputTokens;
|
|
317
|
+
// The ROLLUP columns are what BaseAgent.calculateTokenStats actually sums — the non-rollup
|
|
318
|
+
// ones are ignored by it entirely. Setting only TokensUsed left every harness run
|
|
319
|
+
// reporting zero tokens while its cost was correct, which is a confusing half-truth: it
|
|
320
|
+
// looks like a free run rather than an unaccounted one. A harness turn has no nested
|
|
321
|
+
// child prompt runs, so the rollup equals the turn's own usage.
|
|
322
|
+
run.TokensPromptRollup = turn.InputTokens;
|
|
323
|
+
run.TokensCompletionRollup = turn.OutputTokens;
|
|
324
|
+
run.TokensUsedRollup = turn.InputTokens + turn.OutputTokens;
|
|
325
|
+
if (turn.CostUsd !== undefined) {
|
|
326
|
+
run.TotalCost = turn.CostUsd;
|
|
327
|
+
}
|
|
328
|
+
if (turn.ErrorMessage) {
|
|
329
|
+
run.ErrorMessage = turn.ErrorMessage;
|
|
330
|
+
}
|
|
331
|
+
if (!(await run.Save())) {
|
|
332
|
+
LogError(`Failed to save harness AIPromptRun: ${run.LatestResult?.CompleteMessage ?? 'unknown error'}`);
|
|
333
|
+
return undefined;
|
|
334
|
+
}
|
|
335
|
+
return run;
|
|
336
|
+
}
|
|
337
|
+
catch (e) {
|
|
338
|
+
LogError(`Failed to record harness AIPromptRun: ${describeError(e)}`);
|
|
339
|
+
return undefined;
|
|
340
|
+
}
|
|
341
|
+
}
|
|
342
|
+
/**
|
|
343
|
+
* Resolves the three NOT NULL foreign keys on `AIPromptRun` to REAL catalog rows.
|
|
344
|
+
*
|
|
345
|
+
* None of these is a placeholder, which is the point — inventing catalog rows to satisfy a
|
|
346
|
+
* constraint would pollute the model and vendor catalogs with fictions that then show up in
|
|
347
|
+
* every cost report:
|
|
348
|
+
*
|
|
349
|
+
* · PromptID — the agent type's system prompt. The harness turn really was produced by that
|
|
350
|
+
* template; it is the same one the Loop type renders.
|
|
351
|
+
* · VendorID — `AIAgentHarness.AIVendorID`. Claude Code really does call Anthropic.
|
|
352
|
+
* · ModelID — `AIAgentHarness.AIModelID`. Ideally this would be the model the harness
|
|
353
|
+
* REPORTED for the turn, resolved by name; that refinement belongs here once adapters
|
|
354
|
+
* surface it, and falls back to the declared model meanwhile.
|
|
355
|
+
*/
|
|
356
|
+
async resolveAccountingIds(promptParams) {
|
|
357
|
+
const promptId = promptParams.prompt?.ID;
|
|
358
|
+
const modelId = this.harnessRow?.AIModelID;
|
|
359
|
+
const vendorId = this.harnessRow?.AIVendorID;
|
|
360
|
+
if (!promptId || !modelId || !vendorId) {
|
|
361
|
+
return null;
|
|
362
|
+
}
|
|
363
|
+
return { PromptID: promptId, ModelID: modelId, VendorID: vendorId };
|
|
364
|
+
}
|
|
365
|
+
/**
|
|
366
|
+
* Resolves the model the harness REPORTED using to an MJ catalog row.
|
|
367
|
+
*
|
|
368
|
+
* Recording the model we assumed rather than the one that ran is not a cosmetic problem: a
|
|
369
|
+
* harness picks its own model unless told otherwise, and Opus and Sonnet are not the same price,
|
|
370
|
+
* so the run's cost is attributed to the wrong model. Observed live — the harness ran
|
|
371
|
+
* `claude-opus-4-6` while the run recorded Claude Sonnet 5, purely because that was the harness
|
|
372
|
+
* row's declared anchor.
|
|
373
|
+
*
|
|
374
|
+
* Returns null when the reported name matches nothing, letting the caller fall back to the
|
|
375
|
+
* declared anchor. A miss is expected for a model newer than the catalog and must not fail the
|
|
376
|
+
* run — an approximate attribution still beats no AIPromptRun at all.
|
|
377
|
+
*/
|
|
378
|
+
async resolveReportedModelId(reportedModel, promptParams) {
|
|
379
|
+
if (!reportedModel) {
|
|
380
|
+
return null;
|
|
381
|
+
}
|
|
382
|
+
try {
|
|
383
|
+
// APIName lives on AIModelVendor, NOT AIModel — the first version of this queried
|
|
384
|
+
// AIModel.APIName, which does not exist. RunView returns Success:false for an invalid
|
|
385
|
+
// column rather than throwing, so the resolver failed SILENTLY on every turn and quietly
|
|
386
|
+
// fell back to the declared anchor. That is the failure shape this codebase keeps
|
|
387
|
+
// producing: a wrong answer that looks like a right one.
|
|
388
|
+
//
|
|
389
|
+
// Vendors report dated variants (`claude-sonnet-4-5-20250929`) where the catalog holds
|
|
390
|
+
// the base name (`claude-sonnet-4-5`), so an exact match is tried first and a prefix
|
|
391
|
+
// match second.
|
|
392
|
+
const escaped = reportedModel.replace(/'/g, "''");
|
|
393
|
+
const base = escaped.replace(/-\d{8}$/, '');
|
|
394
|
+
const rv = new RunView();
|
|
395
|
+
const result = await rv.RunView({
|
|
396
|
+
EntityName: 'MJ: AI Model Vendors',
|
|
397
|
+
Fields: ['ModelID'],
|
|
398
|
+
ExtraFilter: `APIName='${escaped}' OR APIName='${base}'`,
|
|
399
|
+
ResultType: 'simple',
|
|
400
|
+
}, promptParams.contextUser);
|
|
401
|
+
if (!result.Success) {
|
|
402
|
+
LogError(`Reported-model lookup failed for '${reportedModel}': ${result.ErrorMessage}`);
|
|
403
|
+
return null;
|
|
404
|
+
}
|
|
405
|
+
const modelId = result.Results?.[0]?.ModelID ?? null;
|
|
406
|
+
if (!modelId) {
|
|
407
|
+
LogStatus(`Harness reported model '${reportedModel}', which is not in the catalog — ` +
|
|
408
|
+
`attributing this turn to the harness row's declared model instead.`);
|
|
409
|
+
}
|
|
410
|
+
return modelId;
|
|
411
|
+
}
|
|
412
|
+
catch (e) {
|
|
413
|
+
LogError(`Failed to resolve reported harness model '${reportedModel}': ${describeError(e)}`);
|
|
414
|
+
return null;
|
|
415
|
+
}
|
|
416
|
+
}
|
|
417
|
+
/**
|
|
418
|
+
* Records the harness session on the run.
|
|
419
|
+
*
|
|
420
|
+
* AIAgentRun.ExternalSessionID exists precisely so an MJ run can be correlated with the vendor's
|
|
421
|
+
* own session logs when diagnosing in-sandbox behaviour — the one place MJ's audit trail
|
|
422
|
+
* deliberately stops. It was added, documented, and then never populated, so answering "did this
|
|
423
|
+
* run resume its session?" meant reading the vendor's files off disk instead of the run record.
|
|
424
|
+
*/
|
|
425
|
+
persistExternalSessionId(sessionId) {
|
|
426
|
+
const run = this._agentRun;
|
|
427
|
+
if (run && sessionId && !run.ExternalSessionID) {
|
|
428
|
+
run.ExternalSessionID = sessionId;
|
|
429
|
+
}
|
|
430
|
+
}
|
|
431
|
+
/**
|
|
432
|
+
* Builds the environment injected into the sandbox.
|
|
433
|
+
*
|
|
434
|
+
* ## Secrets travel as process environment, never as prompt text
|
|
435
|
+
*
|
|
436
|
+
* Everything resolved here is handed to the sandbox executor and becomes the harness PROCESS's
|
|
437
|
+
* environment. None of it is rendered into the turn prompt, so a credential never enters the
|
|
438
|
+
* model's context and cannot be echoed back, logged as conversation, or persisted to a run step.
|
|
439
|
+
* It lives exactly as long as the process does.
|
|
440
|
+
*
|
|
441
|
+
* ## Resolution order — credentials first, env as the documented fallback
|
|
442
|
+
*
|
|
443
|
+
* Mirrors how MJ's AI layer already resolves vendor keys, because operators should not have to
|
|
444
|
+
* learn a second scheme:
|
|
445
|
+
*
|
|
446
|
+
* 1. `MJ: AI Agent Credentials` grants for this agent, read from `MJ: Credentials`. The
|
|
447
|
+
* governed path — auditable, revocable, per-agent.
|
|
448
|
+
* 2. The server's own `process.env[EnvVariableName]`. If a harness needs ANTHROPIC_API_KEY and
|
|
449
|
+
* no credential row grants one, the MJAPI process's own value is used.
|
|
450
|
+
* 3. When the agent has no grants at all, the harness vendor's key under the existing
|
|
451
|
+
* `AI_VENDOR_API_KEY__<DRIVER>` convention — the zero-config path.
|
|
452
|
+
*
|
|
453
|
+
* Preferring credentials matters: env vars are process-wide, so falling back means an agent gets
|
|
454
|
+
* whatever the server holds rather than only what it was granted. That is the pragmatic path for
|
|
455
|
+
* dev and single-tenant installs, and the reason multi-tenant deployments should grant
|
|
456
|
+
* explicitly. The distinction is logged, not silent.
|
|
457
|
+
*/
|
|
458
|
+
async resolveGrantedEnvironment(contextUser) {
|
|
459
|
+
const environment = {};
|
|
460
|
+
try {
|
|
461
|
+
const rv = new RunView();
|
|
462
|
+
const grants = await rv.RunView({
|
|
463
|
+
EntityName: 'MJ: AI Agent Credentials',
|
|
464
|
+
ExtraFilter: `AgentID='${this._agentRunAgentId()}' AND Status='Active'`,
|
|
465
|
+
OrderBy: 'Priority ASC',
|
|
466
|
+
ResultType: 'entity_object',
|
|
467
|
+
}, contextUser);
|
|
468
|
+
if (!grants.Success) {
|
|
469
|
+
LogError(`Failed to load harness credential grants: ${grants.ErrorMessage}`);
|
|
470
|
+
}
|
|
471
|
+
const rows = grants.Success ? (grants.Results ?? []) : [];
|
|
472
|
+
for (const grant of rows) {
|
|
473
|
+
if (!grant.EnvVariableName) {
|
|
474
|
+
// No variable name means the adapter decides how to surface it; nothing to
|
|
475
|
+
// inject generically.
|
|
476
|
+
continue;
|
|
477
|
+
}
|
|
478
|
+
const secret = await this.loadCredentialValue(grant.CredentialID, contextUser);
|
|
479
|
+
if (secret) {
|
|
480
|
+
environment[grant.EnvVariableName] = secret;
|
|
481
|
+
continue;
|
|
482
|
+
}
|
|
483
|
+
const fromEnv = process.env[grant.EnvVariableName];
|
|
484
|
+
if (fromEnv) {
|
|
485
|
+
LogStatus(`Harness credential '${grant.EnvVariableName}' not resolvable from MJ: Credentials; ` +
|
|
486
|
+
'falling back to the server environment.');
|
|
487
|
+
environment[grant.EnvVariableName] = fromEnv;
|
|
488
|
+
}
|
|
489
|
+
}
|
|
490
|
+
if (Object.keys(environment).length === 0) {
|
|
491
|
+
this.applyVendorKeyFallback(environment);
|
|
492
|
+
}
|
|
493
|
+
}
|
|
494
|
+
catch (e) {
|
|
495
|
+
LogError(`Failed to resolve harness environment: ${describeError(e)}`);
|
|
496
|
+
}
|
|
497
|
+
return environment;
|
|
498
|
+
}
|
|
499
|
+
/**
|
|
500
|
+
* Zero-config path: use the harness vendor's key from the environment when the agent has no
|
|
501
|
+
* explicit grants.
|
|
502
|
+
*
|
|
503
|
+
* Uses the same `AI_VENDOR_API_KEY__<DRIVER>` convention the AI layer already uses, so a
|
|
504
|
+
* developer who has MJ talking to Anthropic already has Claude Code working without seeding a
|
|
505
|
+
* credential row.
|
|
506
|
+
*/
|
|
507
|
+
applyVendorKeyFallback(environment) {
|
|
508
|
+
const harness = this.harnessRow;
|
|
509
|
+
if (!harness?.DriverClass) {
|
|
510
|
+
return;
|
|
511
|
+
}
|
|
512
|
+
const envKey = `AI_VENDOR_API_KEY__${harness.DriverClass.toUpperCase()}`;
|
|
513
|
+
const value = process.env[envKey];
|
|
514
|
+
if (!value) {
|
|
515
|
+
return;
|
|
516
|
+
}
|
|
517
|
+
// The variable the harness itself expects. Kept as an explicit map rather than guessed,
|
|
518
|
+
// because writing a key into the wrong variable name is a silent no-op the harness reports
|
|
519
|
+
// only as an auth failure. A harness not listed here must be granted explicitly through
|
|
520
|
+
// MJ: AI Agent Credentials.
|
|
521
|
+
const target = HARNESS_CREDENTIAL_ENV_VARS[harness.DriverClass];
|
|
522
|
+
if (!target) {
|
|
523
|
+
LogStatus(`Harness '${harness.Name}' has no credential grants and no known env-var convention for ` +
|
|
524
|
+
`driver '${harness.DriverClass}'. Grant one via MJ: AI Agent Credentials.`);
|
|
525
|
+
return;
|
|
526
|
+
}
|
|
527
|
+
environment[target] = value;
|
|
528
|
+
LogStatus(`Harness '${harness.Name}' using vendor key fallback from ${envKey}.`);
|
|
529
|
+
}
|
|
530
|
+
/**
|
|
531
|
+
* Reads a credential's value.
|
|
532
|
+
*
|
|
533
|
+
* Custody stays in `MJ: Credentials` — this only reads what the agent was granted, and does not
|
|
534
|
+
* cache it beyond the session.
|
|
535
|
+
*/
|
|
536
|
+
async loadCredentialValue(credentialId, contextUser) {
|
|
537
|
+
try {
|
|
538
|
+
const md = this.ProviderToUse;
|
|
539
|
+
const credential = await md.GetEntityObject('MJ: Credentials', contextUser);
|
|
540
|
+
if (!(await credential.Load(credentialId))) {
|
|
541
|
+
return null;
|
|
542
|
+
}
|
|
543
|
+
return credential.Values ?? null;
|
|
544
|
+
}
|
|
545
|
+
catch (e) {
|
|
546
|
+
LogError(`Failed to load credential ${credentialId}: ${describeError(e)}`);
|
|
547
|
+
return null;
|
|
548
|
+
}
|
|
549
|
+
}
|
|
550
|
+
/** Loads the harness registry row this agent selected by name. */
|
|
551
|
+
async loadHarnessRow(harnessName, contextUser) {
|
|
552
|
+
if (!harnessName) {
|
|
553
|
+
throw new Error("Agent is of type 'Harness' but its TypeConfiguration does not name a harness. " +
|
|
554
|
+
'Set { "harnessName": "..." } matching a row in MJ: AI Agent Harnesses.');
|
|
555
|
+
}
|
|
556
|
+
const rv = new RunView();
|
|
557
|
+
const result = await rv.RunView({
|
|
558
|
+
EntityName: 'MJ: AI Agent Harnesses',
|
|
559
|
+
ExtraFilter: `Name='${harnessName.replace(/'/g, "''")}' AND Status='Active'`,
|
|
560
|
+
ResultType: 'entity_object',
|
|
561
|
+
}, contextUser);
|
|
562
|
+
if (!result.Success) {
|
|
563
|
+
throw new Error(`Failed to load harness '${harnessName}': ${result.ErrorMessage}`);
|
|
564
|
+
}
|
|
565
|
+
const row = result.Results?.[0];
|
|
566
|
+
if (!row) {
|
|
567
|
+
throw new Error(`No Active harness named '${harnessName}' in MJ: AI Agent Harnesses.`);
|
|
568
|
+
}
|
|
569
|
+
return row;
|
|
570
|
+
}
|
|
571
|
+
/** Resolves the adapter class named by the harness row. */
|
|
572
|
+
resolveAdapter(harness) {
|
|
573
|
+
const instance = MJGlobal.Instance.ClassFactory.CreateInstance(BaseHarnessAdapter, harness.DriverClass);
|
|
574
|
+
if (!instance) {
|
|
575
|
+
throw new Error(`Harness '${harness.Name}' names DriverClass '${harness.DriverClass}', which is not registered. ` +
|
|
576
|
+
'Ensure the adapter package is loaded (see LoadAgentHarnessAdapters).');
|
|
577
|
+
}
|
|
578
|
+
if (harness.ExecutablePath && 'SetExecutable' in instance) {
|
|
579
|
+
instance.SetExecutable(harness.ExecutablePath);
|
|
580
|
+
}
|
|
581
|
+
return instance;
|
|
582
|
+
}
|
|
583
|
+
/** Chooses a sandbox provider from the agent's configuration. */
|
|
584
|
+
createSandboxProvider(config) {
|
|
585
|
+
return config.sandbox?.provider === 'docker'
|
|
586
|
+
? new DockerSandboxProvider({ defaultImage: config.sandbox?.image })
|
|
587
|
+
: new LocalDirectorySandboxProvider();
|
|
588
|
+
}
|
|
589
|
+
/** Reads and parses the harness block from the agent's TypeConfiguration. */
|
|
590
|
+
readHarnessConfig() {
|
|
591
|
+
const raw = this._agentTypeConfiguration();
|
|
592
|
+
if (!raw) {
|
|
593
|
+
return {};
|
|
594
|
+
}
|
|
595
|
+
try {
|
|
596
|
+
return JSON.parse(raw);
|
|
597
|
+
}
|
|
598
|
+
catch (e) {
|
|
599
|
+
LogError(`AIAgent.TypeConfiguration is not valid JSON: ${describeError(e)}`);
|
|
600
|
+
return {};
|
|
601
|
+
}
|
|
602
|
+
}
|
|
603
|
+
/**
|
|
604
|
+
* The text handed to the harness for this turn.
|
|
605
|
+
*
|
|
606
|
+
* ## Turn 1 carries the RENDERED system prompt — this is not optional
|
|
607
|
+
*
|
|
608
|
+
* The agent-type system prompt template holds the turn-end contract AND, critically, the
|
|
609
|
+
* `_OUTPUT_EXAMPLE` placeholder that shows the harness the exact JSON envelope shape. In the
|
|
610
|
+
* normal Loop path `AIPromptRunner` renders that template; a harness turn bypasses
|
|
611
|
+
* AIPromptRunner, so without rendering it here the harness never sees the schema at all.
|
|
612
|
+
*
|
|
613
|
+
* The failure that caused is worth recording, because it did not look like a missing prompt.
|
|
614
|
+
* The harness emitted well-formed JSON and simply GUESSED the vocabulary — `nextStep.type` came
|
|
615
|
+
* back as `complete`, then `respond`, then `undefined`, none of which are Loop step names. Five
|
|
616
|
+
* turns were burned while BaseAgent's retry feedback taught it the contract one rejection at a
|
|
617
|
+
* time, turning a one-turn question into a two-minute run. A model inventing plausible values
|
|
618
|
+
* for a schema it was never shown reads as a sloppy model; it is actually a missing prompt.
|
|
619
|
+
*
|
|
620
|
+
* Later turns send only the conversation: the harness has the contract from turn 1 and, where
|
|
621
|
+
* `SessionResume` is true, still has it in session context.
|
|
622
|
+
*/
|
|
623
|
+
async buildTurnInput(promptParams, isFirstTurn) {
|
|
624
|
+
// When the adapter genuinely resumed, the harness already holds the conversation — send only
|
|
625
|
+
// the newest message, exactly as a user typing the next line would. Replaying history on top
|
|
626
|
+
// of a resumed session hands it everything twice. Keyed off DidResumeSession (what happened)
|
|
627
|
+
// rather than the capability flag (what is possible): a pruned session makes those disagree.
|
|
628
|
+
const resumed = isFirstTurn && this.adapter?.DidResumeSession === true;
|
|
629
|
+
const allMessages = promptParams.conversationMessages ?? [];
|
|
630
|
+
const conversation = (resumed ? allMessages.slice(-1) : allMessages)
|
|
631
|
+
.map((m) => {
|
|
632
|
+
const content = typeof m.content === 'string' ? m.content : JSON.stringify(m.content ?? '');
|
|
633
|
+
return `[${m.role}]\n${content}`;
|
|
634
|
+
})
|
|
635
|
+
.join('\n\n');
|
|
636
|
+
const contract = this.buildTurnEndContract(promptParams);
|
|
637
|
+
if (isFirstTurn) {
|
|
638
|
+
// Route MJ's system prompt to the harness's SYSTEM channel where the adapter supports it.
|
|
639
|
+
// Sent as user text it competes with the harness's own system prompt and loses. The
|
|
640
|
+
// contract stays in the turn input as well, so adapters without a system channel are
|
|
641
|
+
// unaffected.
|
|
642
|
+
const systemPrompt = await this.renderSystemPrompt(promptParams);
|
|
643
|
+
if (systemPrompt.trim()) {
|
|
644
|
+
this.adapter?.SetSystemPrompt(`${systemPrompt}\n\n${contract}`);
|
|
645
|
+
}
|
|
646
|
+
}
|
|
647
|
+
return [conversation, contract].filter((p) => p.trim().length > 0).join('\n\n');
|
|
648
|
+
}
|
|
649
|
+
/**
|
|
650
|
+
* The turn-end contract, carrying the ACTUAL envelope schema.
|
|
651
|
+
*
|
|
652
|
+
* Deliberately does not depend on template rendering succeeding. The schema reaches the harness
|
|
653
|
+
* from `AIPrompt.OutputExample` directly, because the first attempt at this relied on the
|
|
654
|
+
* agent-type template rendering `_OUTPUT_EXAMPLE` — and when that silently fell back to raw
|
|
655
|
+
* template text, the harness received the literal string `{{ _OUTPUT_EXAMPLE }}` and was no
|
|
656
|
+
* better off than before. It then invented step names (`complete`, `result`, `undefined`) across
|
|
657
|
+
* five wasted turns.
|
|
658
|
+
*
|
|
659
|
+
* The step vocabulary is listed explicitly too. A harness that knows the SHAPE but guesses the
|
|
660
|
+
* VALUES still fails validation, and that is precisely the failure mode observed: well-formed
|
|
661
|
+
* JSON, invented `nextStep.type`.
|
|
662
|
+
*/
|
|
663
|
+
buildTurnEndContract(promptParams) {
|
|
664
|
+
const example = promptParams.prompt?.OutputExample?.trim();
|
|
665
|
+
const lines = [
|
|
666
|
+
'---',
|
|
667
|
+
'END OF TURN REQUIREMENT (this overrides any inclination to reply conversationally):',
|
|
668
|
+
'Respond with ONLY a single raw JSON object. No prose, no markdown fences, no narration',
|
|
669
|
+
'before or after.',
|
|
670
|
+
'',
|
|
671
|
+
'To FINISH and return a final answer:',
|
|
672
|
+
' {"taskComplete": true, "message": "<your answer>", "reasoning": "<why you are done>"}',
|
|
673
|
+
'',
|
|
674
|
+
'To CONTINUE, set taskComplete false and supply nextStep. The field is nextStep.TYPE',
|
|
675
|
+
'(not "step"), and it must be exactly one of:',
|
|
676
|
+
' Actions | Sub-Agent | Chat | Retry | ClientTools | ForEach | While | Pipeline | Skill | Plan',
|
|
677
|
+
'There is no "Success" or "complete" type — completion is taskComplete: true, above.',
|
|
678
|
+
];
|
|
679
|
+
if (example) {
|
|
680
|
+
lines.push('', 'Full response shape:', example);
|
|
681
|
+
}
|
|
682
|
+
return lines.join('\n');
|
|
683
|
+
}
|
|
684
|
+
/**
|
|
685
|
+
* Renders the agent type's system prompt through the same template engine AIPromptRunner uses,
|
|
686
|
+
* so the harness receives exactly what a Loop model would — including the output example.
|
|
687
|
+
*
|
|
688
|
+
* Falls back to the raw template text if rendering fails. A partially-substituted prompt still
|
|
689
|
+
* carries the envelope shape and lets the run proceed; throwing here would fail a run over a
|
|
690
|
+
* template warning, which is the worse trade.
|
|
691
|
+
*/
|
|
692
|
+
async renderSystemPrompt(promptParams) {
|
|
693
|
+
const prompt = promptParams.prompt;
|
|
694
|
+
if (!prompt) {
|
|
695
|
+
return '';
|
|
696
|
+
}
|
|
697
|
+
try {
|
|
698
|
+
await TemplateEngineServer.Instance.Config(false, promptParams.contextUser);
|
|
699
|
+
// Look the template up by ID, not name. TemplateEngineBase.FindTemplate takes a NAME —
|
|
700
|
+
// passing prompt.TemplateID matched nothing on every call, so every render fell through
|
|
701
|
+
// to the fallback and the harness never received the agent's own instructions. It failed
|
|
702
|
+
// quietly because a fallback existed, which is exactly what made it survive two rounds of
|
|
703
|
+
// fixing this same symptom.
|
|
704
|
+
const template = TemplateEngineServer.Instance.Templates.find((t) => UUIDsEqual(t.ID, prompt.TemplateID));
|
|
705
|
+
const content = template?.GetHighestPriorityContent();
|
|
706
|
+
if (template && content) {
|
|
707
|
+
const rendered = await TemplateEngineServer.Instance.RenderTemplate(template, content, { ...(promptParams.data ?? {}), ...(promptParams.templateData ?? {}) }, true, true);
|
|
708
|
+
if (rendered.Success && rendered.Output?.trim()) {
|
|
709
|
+
return rendered.Output;
|
|
710
|
+
}
|
|
711
|
+
LogError(`Harness system prompt render returned no output: ${rendered.Message ?? 'unknown'}`);
|
|
712
|
+
}
|
|
713
|
+
else {
|
|
714
|
+
LogError(`Harness system prompt template not found for TemplateID '${prompt.TemplateID}' ` +
|
|
715
|
+
`(prompt '${prompt.Name}').`);
|
|
716
|
+
}
|
|
717
|
+
}
|
|
718
|
+
catch (e) {
|
|
719
|
+
LogError(`Harness system prompt render failed, falling back to raw template: ${describeError(e)}`);
|
|
720
|
+
}
|
|
721
|
+
// Returning the RAW template would hand the harness literal `{{ placeholder }}` strings,
|
|
722
|
+
// which is worse than sending nothing: it looks like a prompt, reads as noise, and the
|
|
723
|
+
// turn-end contract below is then the only thing carrying real information. Return empty
|
|
724
|
+
// and let the explicit contract do the work.
|
|
725
|
+
LogError('Harness system prompt could not be rendered; proceeding with the explicit turn-end ' +
|
|
726
|
+
'contract only. The harness will not see agent-specific instructions this run.');
|
|
727
|
+
return '';
|
|
728
|
+
}
|
|
729
|
+
/** Shapes a harness turn as the prompt result the rest of BaseAgent expects. */
|
|
730
|
+
buildPromptResult(turn, promptRun, startTime) {
|
|
731
|
+
const endTime = new Date();
|
|
732
|
+
const chatResult = new ChatResult(!turn.ErrorMessage, startTime, endTime);
|
|
733
|
+
chatResult.data = {
|
|
734
|
+
choices: [
|
|
735
|
+
{
|
|
736
|
+
message: { role: 'assistant', content: turn.RawText },
|
|
737
|
+
finish_reason: turn.ErrorMessage ? 'error' : 'stop',
|
|
738
|
+
index: 0,
|
|
739
|
+
},
|
|
740
|
+
],
|
|
741
|
+
usage: new ModelUsage(turn.InputTokens, turn.OutputTokens),
|
|
742
|
+
};
|
|
743
|
+
chatResult.statusText = turn.ErrorMessage ?? 'OK';
|
|
744
|
+
if (turn.ErrorMessage) {
|
|
745
|
+
chatResult.errorMessage = turn.ErrorMessage;
|
|
746
|
+
}
|
|
747
|
+
return {
|
|
748
|
+
success: !turn.ErrorMessage,
|
|
749
|
+
// `result` is what LoopAgentType.parseJSONResponse reads — the harness's turn-end
|
|
750
|
+
// envelope goes in exactly where a model's response would.
|
|
751
|
+
result: turn.RawText,
|
|
752
|
+
rawResult: turn.RawText,
|
|
753
|
+
chatResult,
|
|
754
|
+
errorMessage: turn.ErrorMessage,
|
|
755
|
+
promptRun,
|
|
756
|
+
};
|
|
757
|
+
}
|
|
758
|
+
/**
|
|
759
|
+
* Wires {@link EndHarnessSession} into `BaseAgent.Execute()`'s per-run teardown. `Execute()`
|
|
760
|
+
* calls this exactly once per call, from its top-level `finally` block, regardless of outcome —
|
|
761
|
+
* without it, `startHarnessSession()`'s provisioned sandbox (a live Docker container or a
|
|
762
|
+
* workspace directory) is never finalized and leaks for the life of the host process.
|
|
763
|
+
*/
|
|
764
|
+
async finalizeRun(outcome) {
|
|
765
|
+
await this.EndHarnessSession(outcome);
|
|
766
|
+
}
|
|
767
|
+
/** Tears the session and sandbox down on every exit path. */
|
|
768
|
+
async EndHarnessSession(outcome) {
|
|
769
|
+
try {
|
|
770
|
+
await this.adapter?.EndSession();
|
|
771
|
+
}
|
|
772
|
+
catch (e) {
|
|
773
|
+
LogError(`Harness adapter teardown failed: ${describeError(e)}`);
|
|
774
|
+
}
|
|
775
|
+
try {
|
|
776
|
+
if (this.sandboxProvider && this.sandboxHandle) {
|
|
777
|
+
await this.sandboxProvider.Finalize(this.sandboxHandle, outcome);
|
|
778
|
+
}
|
|
779
|
+
}
|
|
780
|
+
catch (e) {
|
|
781
|
+
LogError(`Harness sandbox finalize failed: ${describeError(e)}`);
|
|
782
|
+
}
|
|
783
|
+
this.adapter = null;
|
|
784
|
+
this.sandboxHandle = null;
|
|
785
|
+
this.sandboxProvider = null;
|
|
786
|
+
this.turnIndex = 0;
|
|
787
|
+
}
|
|
788
|
+
// ---- narrow accessors over BaseAgent internals -------------------------------------------
|
|
789
|
+
// Kept as small named methods so the coupling to BaseAgent's private state is visible in one
|
|
790
|
+
// place rather than scattered through the class.
|
|
791
|
+
_agentRunId() {
|
|
792
|
+
return this._agentRun?.ID ?? 'unknown-run';
|
|
793
|
+
}
|
|
794
|
+
_agentRunAgentId() {
|
|
795
|
+
return this._executeAgentParams()?.agent?.ID ?? '';
|
|
796
|
+
}
|
|
797
|
+
/**
|
|
798
|
+
* The agent's type-specific configuration.
|
|
799
|
+
*
|
|
800
|
+
* Read from BaseAgent's `_executeParams`, NOT from `_agentConfig`: the latter is an
|
|
801
|
+
* AgentConfiguration (agentType / systemPrompt / childPrompt) and carries no agent entity, so
|
|
802
|
+
* reaching for `.agent` there silently yields undefined and every run fails with "does not name
|
|
803
|
+
* a harness" no matter how it is configured.
|
|
804
|
+
*/
|
|
805
|
+
_agentRunConversationId() {
|
|
806
|
+
return this._agentRun?.ConversationID ?? null;
|
|
807
|
+
}
|
|
808
|
+
_agentTypeConfiguration() {
|
|
809
|
+
return this._executeAgentParams()?.agent?.TypeConfiguration ?? null;
|
|
810
|
+
}
|
|
811
|
+
_executeAgentParams() {
|
|
812
|
+
return this._executeParams;
|
|
813
|
+
}
|
|
814
|
+
};
|
|
815
|
+
HarnessAgentBase = __decorate([
|
|
816
|
+
RegisterClass(BaseAgent, 'HarnessAgentType')
|
|
817
|
+
], HarnessAgentBase);
|
|
818
|
+
export { HarnessAgentBase };
|
|
819
|
+
function describeError(e) {
|
|
820
|
+
return e instanceof Error ? e.message : String(e);
|
|
821
|
+
}
|
|
822
|
+
//# sourceMappingURL=HarnessAgentBase.js.map
|