@memberjunction/ai-agent-harness 0.0.0 → 6.1.0-edge.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. package/LICENSE +7 -0
  2. package/README.md +192 -27
  3. package/dist/HarnessAgentBase.d.ts +254 -0
  4. package/dist/HarnessAgentBase.d.ts.map +1 -0
  5. package/dist/HarnessAgentBase.js +813 -0
  6. package/dist/HarnessAgentBase.js.map +1 -0
  7. package/dist/HarnessAgentType.d.ts +39 -0
  8. package/dist/HarnessAgentType.d.ts.map +1 -0
  9. package/dist/HarnessAgentType.js +50 -0
  10. package/dist/HarnessAgentType.js.map +1 -0
  11. package/dist/adapters/BaseCliHarnessAdapter.d.ts +93 -0
  12. package/dist/adapters/BaseCliHarnessAdapter.d.ts.map +1 -0
  13. package/dist/adapters/BaseCliHarnessAdapter.js +184 -0
  14. package/dist/adapters/BaseCliHarnessAdapter.js.map +1 -0
  15. package/dist/adapters/BaseHarnessAdapter.d.ts +114 -0
  16. package/dist/adapters/BaseHarnessAdapter.d.ts.map +1 -0
  17. package/dist/adapters/BaseHarnessAdapter.js +86 -0
  18. package/dist/adapters/BaseHarnessAdapter.js.map +1 -0
  19. package/dist/adapters/ClaudeCodeCliAdapter.d.ts +104 -0
  20. package/dist/adapters/ClaudeCodeCliAdapter.d.ts.map +1 -0
  21. package/dist/adapters/ClaudeCodeCliAdapter.js +268 -0
  22. package/dist/adapters/ClaudeCodeCliAdapter.js.map +1 -0
  23. package/dist/adapters/CodexAdapter.d.ts +27 -0
  24. package/dist/adapters/CodexAdapter.d.ts.map +1 -0
  25. package/dist/adapters/CodexAdapter.js +117 -0
  26. package/dist/adapters/CodexAdapter.js.map +1 -0
  27. package/dist/adapters/GeminiCliAdapter.d.ts +25 -0
  28. package/dist/adapters/GeminiCliAdapter.d.ts.map +1 -0
  29. package/dist/adapters/GeminiCliAdapter.js +98 -0
  30. package/dist/adapters/GeminiCliAdapter.js.map +1 -0
  31. package/dist/adapters/OpenCodeAdapter.d.ts +23 -0
  32. package/dist/adapters/OpenCodeAdapter.d.ts.map +1 -0
  33. package/dist/adapters/OpenCodeAdapter.js +104 -0
  34. package/dist/adapters/OpenCodeAdapter.js.map +1 -0
  35. package/dist/adapters/PiAdapter.d.ts +73 -0
  36. package/dist/adapters/PiAdapter.d.ts.map +1 -0
  37. package/dist/adapters/PiAdapter.js +237 -0
  38. package/dist/adapters/PiAdapter.js.map +1 -0
  39. package/dist/adapters/StdioJsonAdapter.d.ts +43 -0
  40. package/dist/adapters/StdioJsonAdapter.d.ts.map +1 -0
  41. package/dist/adapters/StdioJsonAdapter.js +109 -0
  42. package/dist/adapters/StdioJsonAdapter.js.map +1 -0
  43. package/dist/index.d.ts +26 -0
  44. package/dist/index.d.ts.map +1 -0
  45. package/dist/index.js +28 -0
  46. package/dist/index.js.map +1 -0
  47. package/dist/sandbox/ChildProcessExecutor.d.ts +41 -0
  48. package/dist/sandbox/ChildProcessExecutor.d.ts.map +1 -0
  49. package/dist/sandbox/ChildProcessExecutor.js +86 -0
  50. package/dist/sandbox/ChildProcessExecutor.js.map +1 -0
  51. package/dist/sandbox/DockerSandboxProvider.d.ts +64 -0
  52. package/dist/sandbox/DockerSandboxProvider.d.ts.map +1 -0
  53. package/dist/sandbox/DockerSandboxProvider.js +176 -0
  54. package/dist/sandbox/DockerSandboxProvider.js.map +1 -0
  55. package/dist/sandbox/ISandboxProvider.d.ts +71 -0
  56. package/dist/sandbox/ISandboxProvider.d.ts.map +1 -0
  57. package/dist/sandbox/ISandboxProvider.js +2 -0
  58. package/dist/sandbox/ISandboxProvider.js.map +1 -0
  59. package/dist/sandbox/LocalDirectorySandboxProvider.d.ts +37 -0
  60. package/dist/sandbox/LocalDirectorySandboxProvider.d.ts.map +1 -0
  61. package/dist/sandbox/LocalDirectorySandboxProvider.js +75 -0
  62. package/dist/sandbox/LocalDirectorySandboxProvider.js.map +1 -0
  63. package/dist/sandbox/SandboxExecutor.d.ts +47 -0
  64. package/dist/sandbox/SandboxExecutor.d.ts.map +1 -0
  65. package/dist/sandbox/SandboxExecutor.js +2 -0
  66. package/dist/sandbox/SandboxExecutor.js.map +1 -0
  67. package/dist/types.d.ts +180 -0
  68. package/dist/types.d.ts.map +1 -0
  69. package/dist/types.js +2 -0
  70. package/dist/types.js.map +1 -0
  71. package/package.json +35 -8
@@ -0,0 +1,813 @@
1
+ var __decorate = (this && this.__decorate) || function (decorators, target, key, desc) {
2
+ var c = arguments.length, r = c < 3 ? target : desc === null ? desc = Object.getOwnPropertyDescriptor(target, key) : desc, d;
3
+ if (typeof Reflect === "object" && typeof Reflect.decorate === "function") r = Reflect.decorate(decorators, target, key, desc);
4
+ else for (var i = decorators.length - 1; i >= 0; i--) if (d = decorators[i]) r = (c < 3 ? d(r) : c > 3 ? d(target, key, r) : d(target, key)) || r;
5
+ return c > 3 && r && Object.defineProperty(target, key, r), r;
6
+ };
7
+ import { MJGlobal, RegisterClass, UUIDsEqual } from '@memberjunction/global';
8
+ import { LogError, LogStatus, RunView } from '@memberjunction/core';
9
+ import { BaseAgent } from '@memberjunction/ai-agents';
10
+ import { TemplateEngineServer } from '@memberjunction/templates';
11
+ import { ChatResult, ModelUsage } from '@memberjunction/ai';
12
+ import { BaseHarnessAdapter } from './adapters/BaseHarnessAdapter.js';
13
+ import { LocalDirectorySandboxProvider } from './sandbox/LocalDirectorySandboxProvider.js';
14
+ import { DockerSandboxProvider } from './sandbox/DockerSandboxProvider.js';
15
+ /**
16
+ * Conventional environment variable each harness reads its own credential from.
17
+ *
18
+ * Only used by the zero-config vendor-key fallback. Explicit rather than derived because writing a
19
+ * key into the wrong variable name fails silently from MJ's side — the harness simply reports an
20
+ * auth error much later, with nothing pointing back at the mapping.
21
+ */
22
+ const HARNESS_CREDENTIAL_ENV_VARS = {
23
+ ClaudeCodeCliAdapter: 'ANTHROPIC_API_KEY',
24
+ CodexAdapter: 'OPENAI_API_KEY',
25
+ GeminiCliAdapter: 'GEMINI_API_KEY',
26
+ };
27
+ /**
28
+ * Restated at the end of every turn input.
29
+ *
30
+ * Not redundant with the system prompt: a harness runs a full agentic loop inside its turn, so by
31
+ * the time it finishes working, the instruction it saw at the start is many internal steps behind
32
+ * it. Restating costs a few tokens; the alternative is a malformed-response retry, which costs a
33
+ * whole turn.
34
+ */
35
+ const TURN_END_CONTRACT = [
36
+ '---',
37
+ 'END OF TURN REQUIREMENT (this overrides any inclination to reply conversationally):',
38
+ 'Respond with ONLY a single raw JSON object matching the response format you were given.',
39
+ 'No prose, no markdown fences, no narration before or after. If the work is complete, say so',
40
+ 'INSIDE the JSON. If you cannot proceed, say that inside the JSON too.',
41
+ ].join('\n');
42
+ /**
43
+ * Executes an MJ agent whose reasoning substrate is an external harness.
44
+ *
45
+ * ## The single seam
46
+ *
47
+ * `BaseAgent` already runs an iterate → decide → execute-steps → iterate loop where the "decide"
48
+ * input is a prompt execution. This class overrides exactly one method — {@link executePrompt} — and
49
+ * substitutes a harness turn for that prompt call. Everything else is untouched `BaseAgent`: the
50
+ * loop, next-step validation, action and sub-agent execution, payload merging under ACLs, guardrail
51
+ * checks between iterations, and run-step recording.
52
+ *
53
+ * That is worth stating precisely because it is the whole architectural bet. `executePrompt` is a
54
+ * five-line protected method with a single call site, so substituting it changes what produces a
55
+ * decision without changing anything about how decisions are validated or enforced.
56
+ *
57
+ * ## Accounting is not optional
58
+ *
59
+ * A harness turn must produce a real `AIPromptRun` row. Run totals are DERIVED — `calculateTokenStats`
60
+ * sums `AIAgentRunStep.PromptRun` rollups — so a turn that records no prompt run contributes nothing,
61
+ * and the run reports zero tokens and zero cost forever. Combined with the cost guardrail, that means
62
+ * a runaway harness would never be interrupted: the ceiling would have nothing to compare against.
63
+ *
64
+ * `AIPromptRun.PromptID`, `.ModelID` and `.VendorID` are all NOT NULL, and every one is resolved to a
65
+ * REAL catalog row rather than a placeholder — see {@link resolveAccountingIds}.
66
+ */
67
+ let HarnessAgentBase = class HarnessAgentBase extends BaseAgent {
68
+ constructor() {
69
+ super(...arguments);
70
+ this.adapter = null;
71
+ this.sandboxProvider = null;
72
+ this.sandboxHandle = null;
73
+ this.harnessRow = null;
74
+ this.turnIndex = 0;
75
+ }
76
+ /**
77
+ * Substitutes a harness turn for the prompt call a Loop agent would make.
78
+ *
79
+ * The returned {@link AIPromptRunResult} is shaped exactly as `AIPromptRunner` would shape one,
80
+ * because everything downstream — `DetermineNextStep`, the malformed-response retry machinery,
81
+ * step recording — reads it without knowing or caring that a harness produced it.
82
+ */
83
+ async executePrompt(promptParams) {
84
+ const startTime = new Date();
85
+ try {
86
+ if (this.turnIndex === 0) {
87
+ await this.startHarnessSession(promptParams);
88
+ }
89
+ this.turnIndex++;
90
+ const isFirstTurn = this.turnIndex === 1;
91
+ const input = await this.buildTurnInput(promptParams, isFirstTurn);
92
+ const turn = await this.runTurn(input);
93
+ const promptRun = await this.recordPromptRun(promptParams, turn, startTime, input);
94
+ this.persistExternalSessionId(turn.SessionId);
95
+ return this.buildPromptResult(turn, promptRun, startTime);
96
+ }
97
+ catch (e) {
98
+ const message = describeError(e);
99
+ LogError(`Harness turn failed: ${message}`);
100
+ return this.buildPromptResult({ RawText: '', InputTokens: 0, OutputTokens: 0, ErrorMessage: message }, undefined, startTime);
101
+ }
102
+ }
103
+ /** Provisions the sandbox, resolves credentials and launches the harness session. */
104
+ async startHarnessSession(promptParams) {
105
+ const contextUser = promptParams.contextUser;
106
+ const config = this.readHarnessConfig();
107
+ const harness = await this.loadHarnessRow(config.harnessName, contextUser);
108
+ this.harnessRow = harness;
109
+ this.adapter = this.resolveAdapter(harness);
110
+ this.sandboxProvider = this.createSandboxProvider(config);
111
+ const key = {
112
+ Scope: config.sandbox?.workspaceScope ?? 'agent-user',
113
+ AgentId: this._agentRunAgentId(),
114
+ UserId: contextUser?.ID,
115
+ RunId: this._agentRunId(),
116
+ };
117
+ this.sandboxHandle = await this.sandboxProvider.Provision(key, {
118
+ NetworkPolicy: config.sandbox?.networkPolicy ?? 'mcp-only',
119
+ Image: config.sandbox?.image,
120
+ });
121
+ const environment = await this.resolveGrantedEnvironment(contextUser);
122
+ // Permissions are declared in agent metadata and applied BEFORE the session starts, so
123
+ // adapters can fold them into launch flags. Runtime overrides arrive through the same
124
+ // TypeConfiguration merge as every other harness setting, so a caller can loosen or tighten
125
+ // a single run without editing the agent.
126
+ const policy = this.resolvePermissionPolicy(config);
127
+ this.adapter.ApplyPermissionPolicy(policy);
128
+ this.warnOnUnenforceablePolicy(harness.Name, policy);
129
+ const resumeSessionId = await this.findResumableSession(harness, key.Scope, contextUser);
130
+ await this.adapter.StartSession({
131
+ ResumeSessionId: resumeSessionId,
132
+ PermissionPolicy: policy,
133
+ Executor: this.sandboxHandle.Executor,
134
+ WorkspacePath: this.sandboxHandle.WorkspacePath,
135
+ Environment: environment,
136
+ Model: harness.DefaultModel ?? undefined,
137
+ CancellationToken: promptParams.cancellationToken,
138
+ });
139
+ LogStatus(`Harness session started for '${harness.Name}' (${harness.DriverClass})`);
140
+ }
141
+ /**
142
+ * Resolves what the agent may do inside its sandbox.
143
+ *
144
+ * Defaults to `strict` when unset — the safe direction. An agent that has never been given a
145
+ * posture should be unable to mutate anything, rather than inheriting whatever the harness does
146
+ * by default, which for a coding agent is a great deal.
147
+ */
148
+ resolvePermissionPolicy(config) {
149
+ return {
150
+ Posture: config.posture ?? 'strict',
151
+ AllowedTools: config.permissions?.allowedTools,
152
+ DisallowedTools: config.permissions?.disallowedTools,
153
+ };
154
+ }
155
+ /**
156
+ * Says out loud when a configured policy will not actually be enforced.
157
+ *
158
+ * Two distinct gaps, previously conflated behind one `PermissionHooks` check — which is why four
159
+ * adapters could ignore a policy entirely while the runtime warned about something else:
160
+ *
161
+ * 1. **`PermissionPolicy: false`** — the adapter never translated the policy into harness flags.
162
+ * The posture and allow/deny lists are inert; the harness runs on its own defaults. This is
163
+ * the serious one, because the agent's metadata reads as though something is gated.
164
+ * 2. **`PermissionHooks: false` under `strict`** — the policy applies, but there is no channel to
165
+ * route an approval through, so anything requiring one is denied rather than escalated.
166
+ *
167
+ * Warn, don't fail. Refusing the run would take every adapter without a verified flag vocabulary
168
+ * offline, and an unenforced policy on a properly-provisioned sandbox is still contained by the
169
+ * sandbox. What is not acceptable is the operator not knowing which situation they are in.
170
+ */
171
+ warnOnUnenforceablePolicy(harnessName, policy) {
172
+ const capabilities = this.adapter?.Capabilities;
173
+ const policyWasConfigured = policy.Posture !== 'strict' ||
174
+ (policy.AllowedTools?.length ?? 0) > 0 ||
175
+ (policy.DisallowedTools?.length ?? 0) > 0;
176
+ if (!capabilities?.PermissionPolicy) {
177
+ LogError(`Harness '${harnessName}': its adapter does not apply MJ permission policies ` +
178
+ `(CapabilitySettings.PermissionPolicy is not true), so the ` +
179
+ `'${policy.Posture}' posture and any allow/deny lists are NOT enforced. The ` +
180
+ `harness runs on its own defaults — rely on the sandbox provider for containment.` +
181
+ (policyWasConfigured ? ' A policy IS configured on this agent and is being ignored.' : ''));
182
+ return;
183
+ }
184
+ if (policy.Posture === 'strict' && !capabilities.PermissionHooks) {
185
+ LogStatus(`Harness '${harnessName}' is set to STRICT posture but its adapter reports no ` +
186
+ `permission hooks. The harness's own prompts have nowhere to go headlessly, so ` +
187
+ `mutating tool calls will simply be denied rather than routed for approval.`);
188
+ }
189
+ }
190
+ /**
191
+ * Finds a prior harness session this run can continue, if the adapter can use one.
192
+ *
193
+ * ## Why this is worth doing
194
+ *
195
+ * Without it, every message in a conversation opens a COLD session and MJ replays the whole
196
+ * history into it. Measured on two consecutive messages in one conversation: the second cost
197
+ * $0.0448 against the first's $0.0155 — nearly 3x, spent entirely on re-reading context the
198
+ * harness had already been told once.
199
+ *
200
+ * ## Three gates, each guarding a different way this goes wrong
201
+ *
202
+ * 1. `SessionResume` capability — a harness that cannot resume must keep replaying. Offering a
203
+ * session id to an adapter that ignores it is harmless; ASSUMING it resumed is not, which is
204
+ * why the outcome is reported back rather than inferred.
205
+ * 2. Workspace scope must be durable. Harnesses key their session store by working directory, so
206
+ * a `run`-scoped workspace is a new directory every time and the session would never be
207
+ * found. Gating here keeps the failure at "no resume" rather than a silent miss.
208
+ * 3. Same conversation. That is the continuity boundary users already understand — a time-based
209
+ * cache would expire while someone is at lunch and, worse, leak stale context into an
210
+ * unrelated new conversation.
211
+ */
212
+ async findResumableSession(harness, scope, contextUser) {
213
+ if (!this.adapter?.Capabilities?.SessionResume) {
214
+ return undefined;
215
+ }
216
+ if (scope === 'run') {
217
+ return undefined;
218
+ }
219
+ const conversationId = this._agentRunConversationId();
220
+ if (!conversationId) {
221
+ return undefined;
222
+ }
223
+ try {
224
+ const rv = new RunView();
225
+ const result = await rv.RunView({
226
+ EntityName: 'MJ: AI Agent Runs',
227
+ Fields: ['ExternalSessionID'],
228
+ ExtraFilter: `AgentID='${this._agentRunAgentId()}' AND ConversationID='${conversationId}' ` +
229
+ `AND ExternalSessionID IS NOT NULL`,
230
+ OrderBy: '__mj_CreatedAt DESC',
231
+ MaxRows: 1,
232
+ ResultType: 'simple',
233
+ }, contextUser);
234
+ if (!result.Success) {
235
+ LogError(`Failed to look up a resumable harness session: ${result.ErrorMessage}`);
236
+ return undefined;
237
+ }
238
+ const sessionId = result.Results?.[0]?.ExternalSessionID;
239
+ if (sessionId) {
240
+ LogStatus(`Resuming harness session ${sessionId} for '${harness.Name}'.`);
241
+ }
242
+ return sessionId;
243
+ }
244
+ catch (e) {
245
+ LogError(`Failed to look up a resumable harness session: ${describeError(e)}`);
246
+ return undefined;
247
+ }
248
+ }
249
+ /** Accumulates one turn's event stream into a single result. */
250
+ async runTurn(input) {
251
+ const adapter = this.adapter;
252
+ const turn = { RawText: '', InputTokens: 0, OutputTokens: 0 };
253
+ for await (const event of adapter.RunTurn(input)) {
254
+ switch (event.Type) {
255
+ case 'usage':
256
+ // Summed rather than replaced: a harness may report usage more than once per
257
+ // turn, and undercounting here silently weakens the cost guardrail.
258
+ turn.InputTokens += event.InputTokens;
259
+ turn.OutputTokens += event.OutputTokens;
260
+ turn.CostUsd = (turn.CostUsd ?? 0) + (event.CostUsd ?? 0);
261
+ break;
262
+ case 'turn-complete':
263
+ turn.RawText = event.RawText;
264
+ break;
265
+ case 'session-error':
266
+ turn.ErrorMessage = event.Error;
267
+ break;
268
+ default:
269
+ // assistant-text and sandbox-activity are live-view only. They are deliberately
270
+ // NOT persisted as run steps — the audit boundary is the turn, and widening it
271
+ // to in-sandbox activity is a documented non-goal, not an oversight.
272
+ break;
273
+ }
274
+ }
275
+ turn.SessionId = adapter.SessionId;
276
+ turn.ReportedModel = adapter.ReportedModel;
277
+ return turn;
278
+ }
279
+ /**
280
+ * Writes the `AIPromptRun` that carries this turn's usage.
281
+ *
282
+ * Returns undefined only when the row could not be created, which is logged loudly rather than
283
+ * swallowed: without it the run's cost and token totals stay at zero and its guardrails go blind.
284
+ */
285
+ async recordPromptRun(promptParams, turn, startTime, input) {
286
+ try {
287
+ const ids = await this.resolveAccountingIds(promptParams);
288
+ if (!ids) {
289
+ LogError('Harness turn could not resolve a Prompt/Model/Vendor for accounting; this run will ' +
290
+ 'under-report tokens and cost, and its cost guardrail will not fire. Set ' +
291
+ 'AIAgentHarness.AIModelID and AIVendorID.');
292
+ return undefined;
293
+ }
294
+ const md = this.ProviderToUse;
295
+ const run = await md.GetEntityObject('MJ: AI Prompt Runs', promptParams.contextUser);
296
+ run.NewRecord();
297
+ run.PromptID = ids.PromptID;
298
+ run.ModelID = (await this.resolveReportedModelId(turn.ReportedModel, promptParams)) ?? ids.ModelID;
299
+ run.VendorID = ids.VendorID;
300
+ run.AgentID = this._agentRunAgentId();
301
+ run.RunAt = startTime;
302
+ run.CompletedAt = new Date();
303
+ run.Success = !turn.ErrorMessage;
304
+ run.Status = turn.ErrorMessage ? 'Failed' : 'Completed';
305
+ // Messages/Result are what the run-detail UI renders. AIPromptRunner populates them as
306
+ // a matter of course; synthesizing this row by hand reproduced the ACCOUNTING fields and
307
+ // dropped the OBSERVABILITY ones, so every harness prompt step showed blank input and
308
+ // output while its token and cost numbers were correct.
309
+ // JSON, not a raw string. AIPromptRun.Messages is documented as "the input messages sent
310
+ // to the model, typically in JSON format" and the run-detail UI parses it as such — a raw
311
+ // string parses to nothing, which is why the response rendered and the input did not.
312
+ run.Messages = JSON.stringify([{ role: 'user', content: input }]);
313
+ run.Result = turn.RawText;
314
+ run.TokensPrompt = turn.InputTokens;
315
+ run.TokensCompletion = turn.OutputTokens;
316
+ run.TokensUsed = turn.InputTokens + turn.OutputTokens;
317
+ // The ROLLUP columns are what BaseAgent.calculateTokenStats actually sums — the non-rollup
318
+ // ones are ignored by it entirely. Setting only TokensUsed left every harness run
319
+ // reporting zero tokens while its cost was correct, which is a confusing half-truth: it
320
+ // looks like a free run rather than an unaccounted one. A harness turn has no nested
321
+ // child prompt runs, so the rollup equals the turn's own usage.
322
+ run.TokensPromptRollup = turn.InputTokens;
323
+ run.TokensCompletionRollup = turn.OutputTokens;
324
+ run.TokensUsedRollup = turn.InputTokens + turn.OutputTokens;
325
+ if (turn.CostUsd !== undefined) {
326
+ run.TotalCost = turn.CostUsd;
327
+ }
328
+ if (turn.ErrorMessage) {
329
+ run.ErrorMessage = turn.ErrorMessage;
330
+ }
331
+ if (!(await run.Save())) {
332
+ LogError(`Failed to save harness AIPromptRun: ${run.LatestResult?.CompleteMessage ?? 'unknown error'}`);
333
+ return undefined;
334
+ }
335
+ return run;
336
+ }
337
+ catch (e) {
338
+ LogError(`Failed to record harness AIPromptRun: ${describeError(e)}`);
339
+ return undefined;
340
+ }
341
+ }
342
+ /**
343
+ * Resolves the three NOT NULL foreign keys on `AIPromptRun` to REAL catalog rows.
344
+ *
345
+ * None of these is a placeholder, which is the point — inventing catalog rows to satisfy a
346
+ * constraint would pollute the model and vendor catalogs with fictions that then show up in
347
+ * every cost report:
348
+ *
349
+ * · PromptID — the agent type's system prompt. The harness turn really was produced by that
350
+ * template; it is the same one the Loop type renders.
351
+ * · VendorID — `AIAgentHarness.AIVendorID`. Claude Code really does call Anthropic.
352
+ * · ModelID — `AIAgentHarness.AIModelID`. Ideally this would be the model the harness
353
+ * REPORTED for the turn, resolved by name; that refinement belongs here once adapters
354
+ * surface it, and falls back to the declared model meanwhile.
355
+ */
356
+ async resolveAccountingIds(promptParams) {
357
+ const promptId = promptParams.prompt?.ID;
358
+ const modelId = this.harnessRow?.AIModelID;
359
+ const vendorId = this.harnessRow?.AIVendorID;
360
+ if (!promptId || !modelId || !vendorId) {
361
+ return null;
362
+ }
363
+ return { PromptID: promptId, ModelID: modelId, VendorID: vendorId };
364
+ }
365
+ /**
366
+ * Resolves the model the harness REPORTED using to an MJ catalog row.
367
+ *
368
+ * Recording the model we assumed rather than the one that ran is not a cosmetic problem: a
369
+ * harness picks its own model unless told otherwise, and Opus and Sonnet are not the same price,
370
+ * so the run's cost is attributed to the wrong model. Observed live — the harness ran
371
+ * `claude-opus-4-6` while the run recorded Claude Sonnet 5, purely because that was the harness
372
+ * row's declared anchor.
373
+ *
374
+ * Returns null when the reported name matches nothing, letting the caller fall back to the
375
+ * declared anchor. A miss is expected for a model newer than the catalog and must not fail the
376
+ * run — an approximate attribution still beats no AIPromptRun at all.
377
+ */
378
+ async resolveReportedModelId(reportedModel, promptParams) {
379
+ if (!reportedModel) {
380
+ return null;
381
+ }
382
+ try {
383
+ // APIName lives on AIModelVendor, NOT AIModel — the first version of this queried
384
+ // AIModel.APIName, which does not exist. RunView returns Success:false for an invalid
385
+ // column rather than throwing, so the resolver failed SILENTLY on every turn and quietly
386
+ // fell back to the declared anchor. That is the failure shape this codebase keeps
387
+ // producing: a wrong answer that looks like a right one.
388
+ //
389
+ // Vendors report dated variants (`claude-sonnet-4-5-20250929`) where the catalog holds
390
+ // the base name (`claude-sonnet-4-5`), so an exact match is tried first and a prefix
391
+ // match second.
392
+ const escaped = reportedModel.replace(/'/g, "''");
393
+ const base = escaped.replace(/-\d{8}$/, '');
394
+ const rv = new RunView();
395
+ const result = await rv.RunView({
396
+ EntityName: 'MJ: AI Model Vendors',
397
+ Fields: ['ModelID'],
398
+ ExtraFilter: `APIName='${escaped}' OR APIName='${base}'`,
399
+ ResultType: 'simple',
400
+ }, promptParams.contextUser);
401
+ if (!result.Success) {
402
+ LogError(`Reported-model lookup failed for '${reportedModel}': ${result.ErrorMessage}`);
403
+ return null;
404
+ }
405
+ const modelId = result.Results?.[0]?.ModelID ?? null;
406
+ if (!modelId) {
407
+ LogStatus(`Harness reported model '${reportedModel}', which is not in the catalog — ` +
408
+ `attributing this turn to the harness row's declared model instead.`);
409
+ }
410
+ return modelId;
411
+ }
412
+ catch (e) {
413
+ LogError(`Failed to resolve reported harness model '${reportedModel}': ${describeError(e)}`);
414
+ return null;
415
+ }
416
+ }
417
+ /**
418
+ * Records the harness session on the run.
419
+ *
420
+ * AIAgentRun.ExternalSessionID exists precisely so an MJ run can be correlated with the vendor's
421
+ * own session logs when diagnosing in-sandbox behaviour — the one place MJ's audit trail
422
+ * deliberately stops. It was added, documented, and then never populated, so answering "did this
423
+ * run resume its session?" meant reading the vendor's files off disk instead of the run record.
424
+ */
425
+ persistExternalSessionId(sessionId) {
426
+ const run = this._agentRun;
427
+ if (run && sessionId && !run.ExternalSessionID) {
428
+ run.ExternalSessionID = sessionId;
429
+ }
430
+ }
431
+ /**
432
+ * Builds the environment injected into the sandbox.
433
+ *
434
+ * ## Secrets travel as process environment, never as prompt text
435
+ *
436
+ * Everything resolved here is handed to the sandbox executor and becomes the harness PROCESS's
437
+ * environment. None of it is rendered into the turn prompt, so a credential never enters the
438
+ * model's context and cannot be echoed back, logged as conversation, or persisted to a run step.
439
+ * It lives exactly as long as the process does.
440
+ *
441
+ * ## Resolution order — credentials first, env as the documented fallback
442
+ *
443
+ * Mirrors how MJ's AI layer already resolves vendor keys, because operators should not have to
444
+ * learn a second scheme:
445
+ *
446
+ * 1. `MJ: AI Agent Credentials` grants for this agent, read from `MJ: Credentials`. The
447
+ * governed path — auditable, revocable, per-agent.
448
+ * 2. The server's own `process.env[EnvVariableName]`. If a harness needs ANTHROPIC_API_KEY and
449
+ * no credential row grants one, the MJAPI process's own value is used.
450
+ * 3. When the agent has no grants at all, the harness vendor's key under the existing
451
+ * `AI_VENDOR_API_KEY__<DRIVER>` convention — the zero-config path.
452
+ *
453
+ * Preferring credentials matters: env vars are process-wide, so falling back means an agent gets
454
+ * whatever the server holds rather than only what it was granted. That is the pragmatic path for
455
+ * dev and single-tenant installs, and the reason multi-tenant deployments should grant
456
+ * explicitly. The distinction is logged, not silent.
457
+ */
458
+ async resolveGrantedEnvironment(contextUser) {
459
+ const environment = {};
460
+ try {
461
+ const rv = new RunView();
462
+ const grants = await rv.RunView({
463
+ EntityName: 'MJ: AI Agent Credentials',
464
+ ExtraFilter: `AgentID='${this._agentRunAgentId()}' AND Status='Active'`,
465
+ OrderBy: 'Priority ASC',
466
+ ResultType: 'entity_object',
467
+ }, contextUser);
468
+ if (!grants.Success) {
469
+ LogError(`Failed to load harness credential grants: ${grants.ErrorMessage}`);
470
+ }
471
+ const rows = grants.Success ? (grants.Results ?? []) : [];
472
+ for (const grant of rows) {
473
+ if (!grant.EnvVariableName) {
474
+ // No variable name means the adapter decides how to surface it; nothing to
475
+ // inject generically.
476
+ continue;
477
+ }
478
+ const secret = await this.loadCredentialValue(grant.CredentialID, contextUser);
479
+ if (secret) {
480
+ environment[grant.EnvVariableName] = secret;
481
+ continue;
482
+ }
483
+ const fromEnv = process.env[grant.EnvVariableName];
484
+ if (fromEnv) {
485
+ LogStatus(`Harness credential '${grant.EnvVariableName}' not resolvable from MJ: Credentials; ` +
486
+ 'falling back to the server environment.');
487
+ environment[grant.EnvVariableName] = fromEnv;
488
+ }
489
+ }
490
+ if (Object.keys(environment).length === 0) {
491
+ this.applyVendorKeyFallback(environment);
492
+ }
493
+ }
494
+ catch (e) {
495
+ LogError(`Failed to resolve harness environment: ${describeError(e)}`);
496
+ }
497
+ return environment;
498
+ }
499
+ /**
500
+ * Zero-config path: use the harness vendor's key from the environment when the agent has no
501
+ * explicit grants.
502
+ *
503
+ * Uses the same `AI_VENDOR_API_KEY__<DRIVER>` convention the AI layer already uses, so a
504
+ * developer who has MJ talking to Anthropic already has Claude Code working without seeding a
505
+ * credential row.
506
+ */
507
+ applyVendorKeyFallback(environment) {
508
+ const harness = this.harnessRow;
509
+ if (!harness?.DriverClass) {
510
+ return;
511
+ }
512
+ const envKey = `AI_VENDOR_API_KEY__${harness.DriverClass.toUpperCase()}`;
513
+ const value = process.env[envKey];
514
+ if (!value) {
515
+ return;
516
+ }
517
+ // The variable the harness itself expects. Kept as an explicit map rather than guessed,
518
+ // because writing a key into the wrong variable name is a silent no-op the harness reports
519
+ // only as an auth failure. A harness not listed here must be granted explicitly through
520
+ // MJ: AI Agent Credentials.
521
+ const target = HARNESS_CREDENTIAL_ENV_VARS[harness.DriverClass];
522
+ if (!target) {
523
+ LogStatus(`Harness '${harness.Name}' has no credential grants and no known env-var convention for ` +
524
+ `driver '${harness.DriverClass}'. Grant one via MJ: AI Agent Credentials.`);
525
+ return;
526
+ }
527
+ environment[target] = value;
528
+ LogStatus(`Harness '${harness.Name}' using vendor key fallback from ${envKey}.`);
529
+ }
530
+ /**
531
+ * Reads a credential's value.
532
+ *
533
+ * Custody stays in `MJ: Credentials` — this only reads what the agent was granted, and does not
534
+ * cache it beyond the session.
535
+ */
536
+ async loadCredentialValue(credentialId, contextUser) {
537
+ try {
538
+ const md = this.ProviderToUse;
539
+ const credential = await md.GetEntityObject('MJ: Credentials', contextUser);
540
+ if (!(await credential.Load(credentialId))) {
541
+ return null;
542
+ }
543
+ return credential.Values ?? null;
544
+ }
545
+ catch (e) {
546
+ LogError(`Failed to load credential ${credentialId}: ${describeError(e)}`);
547
+ return null;
548
+ }
549
+ }
550
+ /** Loads the harness registry row this agent selected by name. */
551
+ async loadHarnessRow(harnessName, contextUser) {
552
+ if (!harnessName) {
553
+ throw new Error("Agent is of type 'Harness' but its TypeConfiguration does not name a harness. " +
554
+ 'Set { "harnessName": "..." } matching a row in MJ: AI Agent Harnesses.');
555
+ }
556
+ const rv = new RunView();
557
+ const result = await rv.RunView({
558
+ EntityName: 'MJ: AI Agent Harnesses',
559
+ ExtraFilter: `Name='${harnessName.replace(/'/g, "''")}' AND Status='Active'`,
560
+ ResultType: 'entity_object',
561
+ }, contextUser);
562
+ if (!result.Success) {
563
+ throw new Error(`Failed to load harness '${harnessName}': ${result.ErrorMessage}`);
564
+ }
565
+ const row = result.Results?.[0];
566
+ if (!row) {
567
+ throw new Error(`No Active harness named '${harnessName}' in MJ: AI Agent Harnesses.`);
568
+ }
569
+ return row;
570
+ }
571
+ /** Resolves the adapter class named by the harness row. */
572
+ resolveAdapter(harness) {
573
+ const instance = MJGlobal.Instance.ClassFactory.CreateInstance(BaseHarnessAdapter, harness.DriverClass);
574
+ if (!instance) {
575
+ throw new Error(`Harness '${harness.Name}' names DriverClass '${harness.DriverClass}', which is not registered. ` +
576
+ 'Ensure the adapter package is loaded (see LoadAgentHarnessAdapters).');
577
+ }
578
+ if (harness.ExecutablePath && 'SetExecutable' in instance) {
579
+ instance.SetExecutable(harness.ExecutablePath);
580
+ }
581
+ return instance;
582
+ }
583
+ /** Chooses a sandbox provider from the agent's configuration. */
584
+ createSandboxProvider(config) {
585
+ return config.sandbox?.provider === 'docker'
586
+ ? new DockerSandboxProvider({ defaultImage: config.sandbox?.image })
587
+ : new LocalDirectorySandboxProvider();
588
+ }
589
+ /** Reads and parses the harness block from the agent's TypeConfiguration. */
590
+ readHarnessConfig() {
591
+ const raw = this._agentTypeConfiguration();
592
+ if (!raw) {
593
+ return {};
594
+ }
595
+ try {
596
+ return JSON.parse(raw);
597
+ }
598
+ catch (e) {
599
+ LogError(`AIAgent.TypeConfiguration is not valid JSON: ${describeError(e)}`);
600
+ return {};
601
+ }
602
+ }
603
+ /**
604
+ * The text handed to the harness for this turn.
605
+ *
606
+ * ## Turn 1 carries the RENDERED system prompt — this is not optional
607
+ *
608
+ * The agent-type system prompt template holds the turn-end contract AND, critically, the
609
+ * `_OUTPUT_EXAMPLE` placeholder that shows the harness the exact JSON envelope shape. In the
610
+ * normal Loop path `AIPromptRunner` renders that template; a harness turn bypasses
611
+ * AIPromptRunner, so without rendering it here the harness never sees the schema at all.
612
+ *
613
+ * The failure that caused is worth recording, because it did not look like a missing prompt.
614
+ * The harness emitted well-formed JSON and simply GUESSED the vocabulary — `nextStep.type` came
615
+ * back as `complete`, then `respond`, then `undefined`, none of which are Loop step names. Five
616
+ * turns were burned while BaseAgent's retry feedback taught it the contract one rejection at a
617
+ * time, turning a one-turn question into a two-minute run. A model inventing plausible values
618
+ * for a schema it was never shown reads as a sloppy model; it is actually a missing prompt.
619
+ *
620
+ * Later turns send only the conversation: the harness has the contract from turn 1 and, where
621
+ * `SessionResume` is true, still has it in session context.
622
+ */
623
+ async buildTurnInput(promptParams, isFirstTurn) {
624
+ // When the adapter genuinely resumed, the harness already holds the conversation — send only
625
+ // the newest message, exactly as a user typing the next line would. Replaying history on top
626
+ // of a resumed session hands it everything twice. Keyed off DidResumeSession (what happened)
627
+ // rather than the capability flag (what is possible): a pruned session makes those disagree.
628
+ const resumed = isFirstTurn && this.adapter?.DidResumeSession === true;
629
+ const allMessages = promptParams.conversationMessages ?? [];
630
+ const conversation = (resumed ? allMessages.slice(-1) : allMessages)
631
+ .map((m) => {
632
+ const content = typeof m.content === 'string' ? m.content : JSON.stringify(m.content ?? '');
633
+ return `[${m.role}]\n${content}`;
634
+ })
635
+ .join('\n\n');
636
+ const contract = this.buildTurnEndContract(promptParams);
637
+ if (isFirstTurn) {
638
+ // Route MJ's system prompt to the harness's SYSTEM channel where the adapter supports it.
639
+ // Sent as user text it competes with the harness's own system prompt and loses. The
640
+ // contract stays in the turn input as well, so adapters without a system channel are
641
+ // unaffected.
642
+ const systemPrompt = await this.renderSystemPrompt(promptParams);
643
+ if (systemPrompt.trim()) {
644
+ this.adapter?.SetSystemPrompt(`${systemPrompt}\n\n${contract}`);
645
+ }
646
+ }
647
+ return [conversation, contract].filter((p) => p.trim().length > 0).join('\n\n');
648
+ }
649
+ /**
650
+ * The turn-end contract, carrying the ACTUAL envelope schema.
651
+ *
652
+ * Deliberately does not depend on template rendering succeeding. The schema reaches the harness
653
+ * from `AIPrompt.OutputExample` directly, because the first attempt at this relied on the
654
+ * agent-type template rendering `_OUTPUT_EXAMPLE` — and when that silently fell back to raw
655
+ * template text, the harness received the literal string `{{ _OUTPUT_EXAMPLE }}` and was no
656
+ * better off than before. It then invented step names (`complete`, `result`, `undefined`) across
657
+ * five wasted turns.
658
+ *
659
+ * The step vocabulary is listed explicitly too. A harness that knows the SHAPE but guesses the
660
+ * VALUES still fails validation, and that is precisely the failure mode observed: well-formed
661
+ * JSON, invented `nextStep.type`.
662
+ */
663
+ buildTurnEndContract(promptParams) {
664
+ const example = promptParams.prompt?.OutputExample?.trim();
665
+ const lines = [
666
+ '---',
667
+ 'END OF TURN REQUIREMENT (this overrides any inclination to reply conversationally):',
668
+ 'Respond with ONLY a single raw JSON object. No prose, no markdown fences, no narration',
669
+ 'before or after.',
670
+ '',
671
+ 'To FINISH and return a final answer:',
672
+ ' {"taskComplete": true, "message": "<your answer>", "reasoning": "<why you are done>"}',
673
+ '',
674
+ 'To CONTINUE, set taskComplete false and supply nextStep. The field is nextStep.TYPE',
675
+ '(not "step"), and it must be exactly one of:',
676
+ ' Actions | Sub-Agent | Chat | Retry | ClientTools | ForEach | While | Pipeline | Skill | Plan',
677
+ 'There is no "Success" or "complete" type — completion is taskComplete: true, above.',
678
+ ];
679
+ if (example) {
680
+ lines.push('', 'Full response shape:', example);
681
+ }
682
+ return lines.join('\n');
683
+ }
684
+ /**
685
+ * Renders the agent type's system prompt through the same template engine AIPromptRunner uses,
686
+ * so the harness receives exactly what a Loop model would — including the output example.
687
+ *
688
+ * Falls back to the raw template text if rendering fails. A partially-substituted prompt still
689
+ * carries the envelope shape and lets the run proceed; throwing here would fail a run over a
690
+ * template warning, which is the worse trade.
691
+ */
692
+ async renderSystemPrompt(promptParams) {
693
+ const prompt = promptParams.prompt;
694
+ if (!prompt) {
695
+ return '';
696
+ }
697
+ try {
698
+ await TemplateEngineServer.Instance.Config(false, promptParams.contextUser);
699
+ // Look the template up by ID, not name. TemplateEngineBase.FindTemplate takes a NAME —
700
+ // passing prompt.TemplateID matched nothing on every call, so every render fell through
701
+ // to the fallback and the harness never received the agent's own instructions. It failed
702
+ // quietly because a fallback existed, which is exactly what made it survive two rounds of
703
+ // fixing this same symptom.
704
+ const template = TemplateEngineServer.Instance.Templates.find((t) => UUIDsEqual(t.ID, prompt.TemplateID));
705
+ const content = template?.GetHighestPriorityContent();
706
+ if (template && content) {
707
+ const rendered = await TemplateEngineServer.Instance.RenderTemplate(template, content, { ...(promptParams.data ?? {}), ...(promptParams.templateData ?? {}) }, true, true);
708
+ if (rendered.Success && rendered.Output?.trim()) {
709
+ return rendered.Output;
710
+ }
711
+ LogError(`Harness system prompt render returned no output: ${rendered.Message ?? 'unknown'}`);
712
+ }
713
+ else {
714
+ LogError(`Harness system prompt template not found for TemplateID '${prompt.TemplateID}' ` +
715
+ `(prompt '${prompt.Name}').`);
716
+ }
717
+ }
718
+ catch (e) {
719
+ LogError(`Harness system prompt render failed, falling back to raw template: ${describeError(e)}`);
720
+ }
721
+ // Returning the RAW template would hand the harness literal `{{ placeholder }}` strings,
722
+ // which is worse than sending nothing: it looks like a prompt, reads as noise, and the
723
+ // turn-end contract below is then the only thing carrying real information. Return empty
724
+ // and let the explicit contract do the work.
725
+ LogError('Harness system prompt could not be rendered; proceeding with the explicit turn-end ' +
726
+ 'contract only. The harness will not see agent-specific instructions this run.');
727
+ return '';
728
+ }
729
+ /** Shapes a harness turn as the prompt result the rest of BaseAgent expects. */
730
+ buildPromptResult(turn, promptRun, startTime) {
731
+ const endTime = new Date();
732
+ const chatResult = new ChatResult(!turn.ErrorMessage, startTime, endTime);
733
+ chatResult.data = {
734
+ choices: [
735
+ {
736
+ message: { role: 'assistant', content: turn.RawText },
737
+ finish_reason: turn.ErrorMessage ? 'error' : 'stop',
738
+ index: 0,
739
+ },
740
+ ],
741
+ usage: new ModelUsage(turn.InputTokens, turn.OutputTokens),
742
+ };
743
+ chatResult.statusText = turn.ErrorMessage ?? 'OK';
744
+ if (turn.ErrorMessage) {
745
+ chatResult.errorMessage = turn.ErrorMessage;
746
+ }
747
+ return {
748
+ success: !turn.ErrorMessage,
749
+ // `result` is what LoopAgentType.parseJSONResponse reads — the harness's turn-end
750
+ // envelope goes in exactly where a model's response would.
751
+ result: turn.RawText,
752
+ rawResult: turn.RawText,
753
+ chatResult,
754
+ errorMessage: turn.ErrorMessage,
755
+ promptRun,
756
+ };
757
+ }
758
+ /** Tears the session and sandbox down on every exit path. */
759
+ async EndHarnessSession(outcome) {
760
+ try {
761
+ await this.adapter?.EndSession();
762
+ }
763
+ catch (e) {
764
+ LogError(`Harness adapter teardown failed: ${describeError(e)}`);
765
+ }
766
+ try {
767
+ if (this.sandboxProvider && this.sandboxHandle) {
768
+ await this.sandboxProvider.Finalize(this.sandboxHandle, outcome);
769
+ }
770
+ }
771
+ catch (e) {
772
+ LogError(`Harness sandbox finalize failed: ${describeError(e)}`);
773
+ }
774
+ this.adapter = null;
775
+ this.sandboxHandle = null;
776
+ this.sandboxProvider = null;
777
+ this.turnIndex = 0;
778
+ }
779
+ // ---- narrow accessors over BaseAgent internals -------------------------------------------
780
+ // Kept as small named methods so the coupling to BaseAgent's private state is visible in one
781
+ // place rather than scattered through the class.
782
+ _agentRunId() {
783
+ return this._agentRun?.ID ?? 'unknown-run';
784
+ }
785
+ _agentRunAgentId() {
786
+ return this._executeAgentParams()?.agent?.ID ?? '';
787
+ }
788
+ /**
789
+ * The agent's type-specific configuration.
790
+ *
791
+ * Read from BaseAgent's `_executeParams`, NOT from `_agentConfig`: the latter is an
792
+ * AgentConfiguration (agentType / systemPrompt / childPrompt) and carries no agent entity, so
793
+ * reaching for `.agent` there silently yields undefined and every run fails with "does not name
794
+ * a harness" no matter how it is configured.
795
+ */
796
+ _agentRunConversationId() {
797
+ return this._agentRun?.ConversationID ?? null;
798
+ }
799
+ _agentTypeConfiguration() {
800
+ return this._executeAgentParams()?.agent?.TypeConfiguration ?? null;
801
+ }
802
+ _executeAgentParams() {
803
+ return this._executeParams;
804
+ }
805
+ };
806
+ HarnessAgentBase = __decorate([
807
+ RegisterClass(BaseAgent, 'HarnessAgentType')
808
+ ], HarnessAgentBase);
809
+ export { HarnessAgentBase };
810
+ function describeError(e) {
811
+ return e instanceof Error ? e.message : String(e);
812
+ }
813
+ //# sourceMappingURL=HarnessAgentBase.js.map