@memberjunction/ai-agent-harness 0.0.0 → 6.1.0-edge.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. package/LICENSE +7 -0
  2. package/README.md +192 -27
  3. package/dist/HarnessAgentBase.d.ts +261 -0
  4. package/dist/HarnessAgentBase.d.ts.map +1 -0
  5. package/dist/HarnessAgentBase.js +822 -0
  6. package/dist/HarnessAgentBase.js.map +1 -0
  7. package/dist/HarnessAgentType.d.ts +39 -0
  8. package/dist/HarnessAgentType.d.ts.map +1 -0
  9. package/dist/HarnessAgentType.js +50 -0
  10. package/dist/HarnessAgentType.js.map +1 -0
  11. package/dist/adapters/BaseCliHarnessAdapter.d.ts +93 -0
  12. package/dist/adapters/BaseCliHarnessAdapter.d.ts.map +1 -0
  13. package/dist/adapters/BaseCliHarnessAdapter.js +184 -0
  14. package/dist/adapters/BaseCliHarnessAdapter.js.map +1 -0
  15. package/dist/adapters/BaseHarnessAdapter.d.ts +114 -0
  16. package/dist/adapters/BaseHarnessAdapter.d.ts.map +1 -0
  17. package/dist/adapters/BaseHarnessAdapter.js +86 -0
  18. package/dist/adapters/BaseHarnessAdapter.js.map +1 -0
  19. package/dist/adapters/ClaudeCodeCliAdapter.d.ts +104 -0
  20. package/dist/adapters/ClaudeCodeCliAdapter.d.ts.map +1 -0
  21. package/dist/adapters/ClaudeCodeCliAdapter.js +268 -0
  22. package/dist/adapters/ClaudeCodeCliAdapter.js.map +1 -0
  23. package/dist/adapters/CodexAdapter.d.ts +27 -0
  24. package/dist/adapters/CodexAdapter.d.ts.map +1 -0
  25. package/dist/adapters/CodexAdapter.js +117 -0
  26. package/dist/adapters/CodexAdapter.js.map +1 -0
  27. package/dist/adapters/GeminiCliAdapter.d.ts +25 -0
  28. package/dist/adapters/GeminiCliAdapter.d.ts.map +1 -0
  29. package/dist/adapters/GeminiCliAdapter.js +98 -0
  30. package/dist/adapters/GeminiCliAdapter.js.map +1 -0
  31. package/dist/adapters/OpenCodeAdapter.d.ts +23 -0
  32. package/dist/adapters/OpenCodeAdapter.d.ts.map +1 -0
  33. package/dist/adapters/OpenCodeAdapter.js +104 -0
  34. package/dist/adapters/OpenCodeAdapter.js.map +1 -0
  35. package/dist/adapters/PiAdapter.d.ts +73 -0
  36. package/dist/adapters/PiAdapter.d.ts.map +1 -0
  37. package/dist/adapters/PiAdapter.js +237 -0
  38. package/dist/adapters/PiAdapter.js.map +1 -0
  39. package/dist/adapters/StdioJsonAdapter.d.ts +43 -0
  40. package/dist/adapters/StdioJsonAdapter.d.ts.map +1 -0
  41. package/dist/adapters/StdioJsonAdapter.js +109 -0
  42. package/dist/adapters/StdioJsonAdapter.js.map +1 -0
  43. package/dist/index.d.ts +26 -0
  44. package/dist/index.d.ts.map +1 -0
  45. package/dist/index.js +28 -0
  46. package/dist/index.js.map +1 -0
  47. package/dist/sandbox/ChildProcessExecutor.d.ts +41 -0
  48. package/dist/sandbox/ChildProcessExecutor.d.ts.map +1 -0
  49. package/dist/sandbox/ChildProcessExecutor.js +86 -0
  50. package/dist/sandbox/ChildProcessExecutor.js.map +1 -0
  51. package/dist/sandbox/DockerSandboxProvider.d.ts +64 -0
  52. package/dist/sandbox/DockerSandboxProvider.d.ts.map +1 -0
  53. package/dist/sandbox/DockerSandboxProvider.js +176 -0
  54. package/dist/sandbox/DockerSandboxProvider.js.map +1 -0
  55. package/dist/sandbox/ISandboxProvider.d.ts +71 -0
  56. package/dist/sandbox/ISandboxProvider.d.ts.map +1 -0
  57. package/dist/sandbox/ISandboxProvider.js +2 -0
  58. package/dist/sandbox/ISandboxProvider.js.map +1 -0
  59. package/dist/sandbox/LocalDirectorySandboxProvider.d.ts +37 -0
  60. package/dist/sandbox/LocalDirectorySandboxProvider.d.ts.map +1 -0
  61. package/dist/sandbox/LocalDirectorySandboxProvider.js +75 -0
  62. package/dist/sandbox/LocalDirectorySandboxProvider.js.map +1 -0
  63. package/dist/sandbox/SandboxExecutor.d.ts +47 -0
  64. package/dist/sandbox/SandboxExecutor.d.ts.map +1 -0
  65. package/dist/sandbox/SandboxExecutor.js +2 -0
  66. package/dist/sandbox/SandboxExecutor.js.map +1 -0
  67. package/dist/types.d.ts +180 -0
  68. package/dist/types.d.ts.map +1 -0
  69. package/dist/types.js +2 -0
  70. package/dist/types.js.map +1 -0
  71. package/package.json +35 -8
package/LICENSE ADDED
@@ -0,0 +1,7 @@
1
+ ISC License
2
+
3
+ Copyright (c) 2023 MemberJunction
4
+
5
+ Permission to use, copy, modify, and/or distribute this software for any purpose with or without fee is hereby granted, provided that the above copyright notice and this permission notice appear in all copies.
6
+
7
+ THE SOFTWARE IS PROVIDED "AS IS" AND THE AUTHOR DISCLAIMS ALL WARRANTIES WITH REGARD TO THIS SOFTWARE INCLUDING ALL IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS. IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR ANY SPECIAL, DIRECT, INDIRECT, OR CONSEQUENTIAL DAMAGES OR ANY DAMAGES WHATSOEVER RESULTING FROM LOSS OF USE, DATA OR PROFITS, WHETHER IN AN ACTION OF CONTRACT, NEGLIGENCE OR OTHER TORTIOUS ACTION, ARISING OUT OF OR IN CONNECTION WITH THE USE OR PERFORMANCE OF THIS SOFTWARE.
package/README.md CHANGED
@@ -1,45 +1,210 @@
1
1
  # @memberjunction/ai-agent-harness
2
2
 
3
- ## ⚠️ IMPORTANT NOTICE ⚠️
3
+ Run an **external agent harness** — Claude Code, Codex CLI, OpenCode, Gemini CLI, Pi — as the
4
+ reasoning substrate for a MemberJunction agent, while MJ keeps identity, permissions, governed data
5
+ access, payload contracts, HITL, cost control and run-level audit.
4
6
 
5
- **This package is created solely for the purpose of setting up OIDC (OpenID Connect) trusted publishing with npm.**
7
+ > **Design principle: the harness is a substrate, not a peer.** MJ owns the run record, the
8
+ > credentials, the tool surface and the approval flow. The harness owns the reasoning inside a turn.
6
9
 
7
- This is **NOT** a functional package and contains **NO** code or functionality beyond the OIDC setup configuration.
10
+ Design plan: [`plans/external-agent-harness.md`](../../../plans/external-agent-harness.md)
8
11
 
9
- ## Purpose
12
+ ---
13
+
14
+ ## The core idea: a harness turn *is* a Loop iteration
15
+
16
+ `BaseAgent` already runs an iterate → decide → execute-steps → iterate loop where the "decide" input
17
+ is a prompt execution. This package substitutes a **harness turn** for that prompt call and changes
18
+ nothing else.
19
+
20
+ ```
21
+ ┌─ BaseAgent loop (unchanged) ──────────────────────────────────┐
22
+ │ │
23
+ │ executePrompt() ──► [ HarnessAgentBase override ] │
24
+ │ │ │ │
25
+ │ │ ├─ adapter.RunTurn(input) │
26
+ │ │ ├─ accumulate usage │
27
+ │ │ └─ write AIPromptRun │
28
+ │ ▼ │
29
+ │ DetermineNextStep (inherited from LoopAgentType) │
30
+ │ ▼ │
31
+ │ validate ─► execute actions / sub-agents / skills │
32
+ │ ▼ │
33
+ │ checkExecutionGuardrails ─► next turn │
34
+ └───────────────────────────────────────────────────────────────┘
35
+ ```
36
+
37
+ The harness ends each turn by emitting the **Loop next-step JSON envelope**. MJ then executes any
38
+ actions, sub-agents or skills through its own validated machinery and resumes the session with the
39
+ results.
40
+
41
+ **Why this matters:** every guardrail, payload ACL, HITL gate and accounting path already written for
42
+ Loop agents applies to harness agents with no new enforcement code — and there is exactly **one**
43
+ authority channel to audit, not two.
44
+
45
+ ---
46
+
47
+ ## Two registrations, one metadata column
48
+
49
+ ClassFactory registrations are namespaced per base class, so `'HarnessAgentType'` is registered
50
+ twice against different roots:
51
+
52
+ | Registered under | Class | Resolved by | Gives you |
53
+ |---|---|---|---|
54
+ | `BaseAgentType` | `HarnessAgentType` | `BaseAgentType.GetAgentTypeInstance` | the turn protocol (inherits Loop) |
55
+ | `BaseAgent` | `HarnessAgentBase` | `AgentRunner` (`AgentRunner.ts:101`) | the execution driver |
56
+
57
+ Both read `AIAgentType.DriverClass`, so **one metadata value selects both halves**. This is the
58
+ mechanism working as designed — `AgentRunner` already treats the type's `DriverClass` as a
59
+ `BaseAgent` key and falls back to plain `BaseAgent` when unregistered, which is why every Loop agent
60
+ gets the base execution class today.
61
+
62
+ ---
63
+
64
+ ## Adapters
65
+
66
+ | Harness | DriverClass | Mechanism | Notes |
67
+ |---|---|---|---|
68
+ | Claude Code | `ClaudeCodeCliAdapter` | CLI, `stream-json` | Session resume; permission hook pending |
69
+ | Codex | `CodexAdapter` | `codex exec --json` | Session resume |
70
+ | OpenCode | `OpenCodeAdapter` | `opencode run` JSON | Session resume |
71
+ | Gemini CLI | `GeminiCliAdapter` | `gemini --output-format json` | **No** resume — context replayed |
72
+ | Pi | `PiAdapter` | stdio-JSON contract | Requires `ExecutablePath` |
73
+ | *anything* | `StdioJsonAdapter` | documented JSON contract | Escape hatch — zero MJ code |
10
74
 
11
- This package exists to:
12
- 1. Configure OIDC trusted publishing for the package name `@memberjunction/ai-agent-harness`
13
- 2. Enable secure, token-less publishing from CI/CD workflows
14
- 3. Establish provenance for packages published under this name
75
+ Register your own with `@RegisterClass(BaseHarnessAdapter, 'MyAdapter')` and point a harness row's
76
+ `DriverClass` at it. No core changes required.
15
77
 
16
- ## What is OIDC Trusted Publishing?
78
+ ### Capability honesty
17
79
 
18
- OIDC trusted publishing allows package maintainers to publish packages directly from their CI/CD workflows without needing to manage npm access tokens. Instead, it uses OpenID Connect to establish trust between the CI/CD provider (like GitHub Actions) and npm.
80
+ `AIAgentHarness.CapabilitySettings` (typed as `IHarnessCapabilitySettings`) declares what an adapter
81
+ **actually implements**, because the runtime *emulates what is missing*. Gemini CLI reports
82
+ `SessionResume: false`, so context is replayed each turn and those extra tokens are budgeted against
83
+ the run's guardrails rather than quietly absorbed.
19
84
 
20
- ## Setup Instructions
85
+ Claiming a capability that is not wired up produces a silent behavioural gap, not an error. Report
86
+ `false` and let the runtime compensate.
21
87
 
22
- To properly configure OIDC trusted publishing for this package:
88
+ ---
89
+
90
+ ## Sandboxes: the provider owns process placement
91
+
92
+ Adapters never call `spawn`. They run everything through `SandboxExecutor`, obtained from the
93
+ handle the provider returns.
94
+
95
+ | Provider | Execution | Isolation |
96
+ |---|---|---|
97
+ | `LocalDirectorySandboxProvider` | direct spawn | **None** — dev only |
98
+ | `DockerSandboxProvider` | `docker exec`, container per run | Real FS boundary; `networkPolicy: 'none'` enforced |
99
+
100
+ > ⚠️ The local provider scopes a *directory*; it does **not** contain the *process*. `networkPolicy`
101
+ > is advisory there. Anyone who believes `'none'` is enforced locally has a false sense of
102
+ > containment, which is worse than knowing the boundary is soft.
103
+
104
+ `HarnessProcess` is deliberately **stream-based**, not `ChildProcess`-based — a Kubernetes exec is
105
+ streams over a websocket and a remote runner is HTTP, and neither could honestly implement a
106
+ `ChildProcess` contract. It also makes adapters unit-testable with a fake executor: no binary, no
107
+ container, no network.
23
108
 
24
- 1. Go to [npmjs.com](https://www.npmjs.com/) and navigate to your package settings
25
- 2. Configure the trusted publisher (e.g., GitHub Actions)
26
- 3. Specify the repository and workflow that should be allowed to publish
27
- 4. Use the configured workflow to publish your actual package
109
+ ### `WorkspacePath` means "as the harness sees it"
28
110
 
29
- ## DO NOT USE THIS PACKAGE
111
+ A host path under the local provider; a **container-internal** path under Docker. Pass it to harness
112
+ processes — do **not** open it with `fs` unless you know you are on the local provider.
30
113
 
31
- This package is a placeholder for OIDC configuration only. It:
32
- - Contains no executable code
33
- - Provides no functionality
34
- - Should not be installed as a dependency
35
- - Exists only for administrative purposes
114
+ ---
115
+
116
+ ## Configuration
117
+
118
+ Per-agent, in `AIAgent.TypeConfiguration`, validated against `AIAgentType.ConfigSchema`:
36
119
 
37
- ## More Information
120
+ ```jsonc
121
+ {
122
+ "harnessName": "Claude Code", // lookup into MJ: AI Agent Harnesses
123
+ "sandbox": {
124
+ "provider": "local", // local | docker
125
+ "image": "ghcr.io/memberjunction/harness-sandbox:latest",
126
+ "workspaceScope": "agent-user", // run | agent | agent-user
127
+ "networkPolicy": "mcp-only"
128
+ }
129
+ }
130
+ ```
38
131
 
39
- For more details about npm's trusted publishing feature, see:
40
- - [npm Trusted Publishing Documentation](https://docs.npmjs.com/generating-provenance-statements)
41
- - [GitHub Actions OIDC Documentation](https://docs.github.com/en/actions/deployment/security-hardening-your-deployments/about-security-hardening-with-openid-connect)
132
+ **Workspace scope** decides how long files live: `run` is discarded, `agent` is shared across every
133
+ run of that agent, `agent-user` (default) is per agent per user — continuity without one user's
134
+ working files leaking into another's session.
42
135
 
43
136
  ---
44
137
 
45
- **Maintained for OIDC setup purposes only**
138
+ ## Accounting — why every turn writes an `AIPromptRun`
139
+
140
+ Run totals are **derived**: `calculateTokenStats` sums `AIAgentRunStep.PromptRun` rollups. A turn
141
+ that records no prompt run contributes nothing, so the run reports zero tokens and zero cost
142
+ *forever* — and its cost ceiling has nothing to compare against.
143
+
144
+ `AIPromptRun.PromptID`, `.ModelID` and `.VendorID` are all **NOT NULL**, and each resolves to a
145
+ **real** catalog row rather than a placeholder:
146
+
147
+ | Column | Resolves to | Why it is not a fiction |
148
+ |---|---|---|
149
+ | `PromptID` | the agent type's system prompt | that template really did produce the turn |
150
+ | `VendorID` | `AIAgentHarness.AIVendorID` | Claude Code really does call Anthropic |
151
+ | `ModelID` | `AIAgentHarness.AIModelID` | the harness really does run that model |
152
+
153
+ If none resolves, the runtime **fails loudly** rather than skipping the row. A silent skip is exactly
154
+ how a cost ceiling stops protecting anything.
155
+
156
+ ---
157
+
158
+ ## Credentials
159
+
160
+ `MJ: AI Agent Credentials` records the **grant edge** — which credentials an agent carries into its
161
+ sandbox. Custody stays in `MJ: Credentials` / `CredentialEngine`.
162
+
163
+ Environment injection is the **only** channel by which a secret reaches the harness, and it carries
164
+ exactly what was granted — never the MJAPI process environment, never DB credentials, never a user
165
+ token.
166
+
167
+ Distinct from `MJ: AI Credential Bindings`, which is inference-selection plumbing for
168
+ `AIPromptRunner` failover when *MJ itself* executes a prompt.
169
+
170
+ ---
171
+
172
+ ## The audit boundary
173
+
174
+ MJ records what **crosses the boundary**: MCP loopback reads and the turn-end step. Activity *inside*
175
+ the sandbox — file edits, shell commands — streams to `onProgress` for live view but is **not**
176
+ persisted as run steps.
177
+
178
+ This "opaque super-step" granularity is **intentional**. In-sandbox behaviour is governed by posture
179
+ policy, not by run steps. Widening it is a design change, not a bug fix.
180
+
181
+ ---
182
+
183
+ ## Deployment
184
+
185
+ | Environment | Provider | Notes |
186
+ |---|---|---|
187
+ | Local dev | `local` | Fast; uses the dev's own installed CLI and auth |
188
+ | Local parity | `docker` | Same path as production |
189
+ | AWS / Azure | `docker` → ECS/Fargate or ACI | Sandbox image versioned **separately** from MJAPI |
190
+
191
+ Do **not** bake harness binaries into the MJAPI image. A harness running inside the API container
192
+ inherits that container's network reach and IAM role — the wrong blast radius for a process
193
+ executing an autonomous agent's shell commands.
194
+
195
+ ---
196
+
197
+ ## Known gaps
198
+
199
+ - **`PermissionHooks: false` on every adapter.** The `strict` posture needs an MCP permission-prompt
200
+ tool that does not exist yet. Reported honestly so the runtime cannot assume interception it lacks.
201
+ - **`networkPolicy` `mcp-only` / `allowlist` are not enforced at the packet level** under Docker —
202
+ documented as such rather than aliased to `open`.
203
+ - **MCP loopback is not yet wired.** `HarnessSessionConfig` carries the fields; the server and
204
+ per-run scoped credential are still to come.
205
+ - **`ModelID` uses the declared model**, not the model the harness reported for the turn. The
206
+ refinement belongs in `resolveAccountingIds` once adapters surface it.
207
+
208
+ ## License
209
+
210
+ ISC
@@ -0,0 +1,261 @@
1
+ import { BaseAgent } from '@memberjunction/ai-agents';
2
+ import { AIPromptParams, AIPromptRunResult } from '@memberjunction/ai-core-plus';
3
+ /**
4
+ * Executes an MJ agent whose reasoning substrate is an external harness.
5
+ *
6
+ * ## The single seam
7
+ *
8
+ * `BaseAgent` already runs an iterate → decide → execute-steps → iterate loop where the "decide"
9
+ * input is a prompt execution. This class overrides exactly one method — {@link executePrompt} — and
10
+ * substitutes a harness turn for that prompt call. Everything else is untouched `BaseAgent`: the
11
+ * loop, next-step validation, action and sub-agent execution, payload merging under ACLs, guardrail
12
+ * checks between iterations, and run-step recording.
13
+ *
14
+ * That is worth stating precisely because it is the whole architectural bet. `executePrompt` is a
15
+ * five-line protected method with a single call site, so substituting it changes what produces a
16
+ * decision without changing anything about how decisions are validated or enforced.
17
+ *
18
+ * ## Accounting is not optional
19
+ *
20
+ * A harness turn must produce a real `AIPromptRun` row. Run totals are DERIVED — `calculateTokenStats`
21
+ * sums `AIAgentRunStep.PromptRun` rollups — so a turn that records no prompt run contributes nothing,
22
+ * and the run reports zero tokens and zero cost forever. Combined with the cost guardrail, that means
23
+ * a runaway harness would never be interrupted: the ceiling would have nothing to compare against.
24
+ *
25
+ * `AIPromptRun.PromptID`, `.ModelID` and `.VendorID` are all NOT NULL, and every one is resolved to a
26
+ * REAL catalog row rather than a placeholder — see {@link resolveAccountingIds}.
27
+ */
28
+ export declare class HarnessAgentBase extends BaseAgent {
29
+ private adapter;
30
+ private sandboxProvider;
31
+ private sandboxHandle;
32
+ private harnessRow;
33
+ private turnIndex;
34
+ /**
35
+ * Substitutes a harness turn for the prompt call a Loop agent would make.
36
+ *
37
+ * The returned {@link AIPromptRunResult} is shaped exactly as `AIPromptRunner` would shape one,
38
+ * because everything downstream — `DetermineNextStep`, the malformed-response retry machinery,
39
+ * step recording — reads it without knowing or caring that a harness produced it.
40
+ */
41
+ protected executePrompt(promptParams: AIPromptParams): Promise<AIPromptRunResult>;
42
+ /** Provisions the sandbox, resolves credentials and launches the harness session. */
43
+ private startHarnessSession;
44
+ /**
45
+ * Resolves what the agent may do inside its sandbox.
46
+ *
47
+ * Defaults to `strict` when unset — the safe direction. An agent that has never been given a
48
+ * posture should be unable to mutate anything, rather than inheriting whatever the harness does
49
+ * by default, which for a coding agent is a great deal.
50
+ */
51
+ private resolvePermissionPolicy;
52
+ /**
53
+ * Says out loud when a configured policy will not actually be enforced.
54
+ *
55
+ * Two distinct gaps, previously conflated behind one `PermissionHooks` check — which is why four
56
+ * adapters could ignore a policy entirely while the runtime warned about something else:
57
+ *
58
+ * 1. **`PermissionPolicy: false`** — the adapter never translated the policy into harness flags.
59
+ * The posture and allow/deny lists are inert; the harness runs on its own defaults. This is
60
+ * the serious one, because the agent's metadata reads as though something is gated.
61
+ * 2. **`PermissionHooks: false` under `strict`** — the policy applies, but there is no channel to
62
+ * route an approval through, so anything requiring one is denied rather than escalated.
63
+ *
64
+ * Warn, don't fail. Refusing the run would take every adapter without a verified flag vocabulary
65
+ * offline, and an unenforced policy on a properly-provisioned sandbox is still contained by the
66
+ * sandbox. What is not acceptable is the operator not knowing which situation they are in.
67
+ */
68
+ private warnOnUnenforceablePolicy;
69
+ /**
70
+ * Finds a prior harness session this run can continue, if the adapter can use one.
71
+ *
72
+ * ## Why this is worth doing
73
+ *
74
+ * Without it, every message in a conversation opens a COLD session and MJ replays the whole
75
+ * history into it. Measured on two consecutive messages in one conversation: the second cost
76
+ * $0.0448 against the first's $0.0155 — nearly 3x, spent entirely on re-reading context the
77
+ * harness had already been told once.
78
+ *
79
+ * ## Three gates, each guarding a different way this goes wrong
80
+ *
81
+ * 1. `SessionResume` capability — a harness that cannot resume must keep replaying. Offering a
82
+ * session id to an adapter that ignores it is harmless; ASSUMING it resumed is not, which is
83
+ * why the outcome is reported back rather than inferred.
84
+ * 2. Workspace scope must be durable. Harnesses key their session store by working directory, so
85
+ * a `run`-scoped workspace is a new directory every time and the session would never be
86
+ * found. Gating here keeps the failure at "no resume" rather than a silent miss.
87
+ * 3. Same conversation. That is the continuity boundary users already understand — a time-based
88
+ * cache would expire while someone is at lunch and, worse, leak stale context into an
89
+ * unrelated new conversation.
90
+ */
91
+ private findResumableSession;
92
+ /** Accumulates one turn's event stream into a single result. */
93
+ private runTurn;
94
+ /**
95
+ * Writes the `AIPromptRun` that carries this turn's usage.
96
+ *
97
+ * Returns undefined only when the row could not be created, which is logged loudly rather than
98
+ * swallowed: without it the run's cost and token totals stay at zero and its guardrails go blind.
99
+ */
100
+ private recordPromptRun;
101
+ /**
102
+ * Resolves the three NOT NULL foreign keys on `AIPromptRun` to REAL catalog rows.
103
+ *
104
+ * None of these is a placeholder, which is the point — inventing catalog rows to satisfy a
105
+ * constraint would pollute the model and vendor catalogs with fictions that then show up in
106
+ * every cost report:
107
+ *
108
+ * · PromptID — the agent type's system prompt. The harness turn really was produced by that
109
+ * template; it is the same one the Loop type renders.
110
+ * · VendorID — `AIAgentHarness.AIVendorID`. Claude Code really does call Anthropic.
111
+ * · ModelID — `AIAgentHarness.AIModelID`. Ideally this would be the model the harness
112
+ * REPORTED for the turn, resolved by name; that refinement belongs here once adapters
113
+ * surface it, and falls back to the declared model meanwhile.
114
+ */
115
+ private resolveAccountingIds;
116
+ /**
117
+ * Resolves the model the harness REPORTED using to an MJ catalog row.
118
+ *
119
+ * Recording the model we assumed rather than the one that ran is not a cosmetic problem: a
120
+ * harness picks its own model unless told otherwise, and Opus and Sonnet are not the same price,
121
+ * so the run's cost is attributed to the wrong model. Observed live — the harness ran
122
+ * `claude-opus-4-6` while the run recorded Claude Sonnet 5, purely because that was the harness
123
+ * row's declared anchor.
124
+ *
125
+ * Returns null when the reported name matches nothing, letting the caller fall back to the
126
+ * declared anchor. A miss is expected for a model newer than the catalog and must not fail the
127
+ * run — an approximate attribution still beats no AIPromptRun at all.
128
+ */
129
+ private resolveReportedModelId;
130
+ /**
131
+ * Records the harness session on the run.
132
+ *
133
+ * AIAgentRun.ExternalSessionID exists precisely so an MJ run can be correlated with the vendor's
134
+ * own session logs when diagnosing in-sandbox behaviour — the one place MJ's audit trail
135
+ * deliberately stops. It was added, documented, and then never populated, so answering "did this
136
+ * run resume its session?" meant reading the vendor's files off disk instead of the run record.
137
+ */
138
+ private persistExternalSessionId;
139
+ /**
140
+ * Builds the environment injected into the sandbox.
141
+ *
142
+ * ## Secrets travel as process environment, never as prompt text
143
+ *
144
+ * Everything resolved here is handed to the sandbox executor and becomes the harness PROCESS's
145
+ * environment. None of it is rendered into the turn prompt, so a credential never enters the
146
+ * model's context and cannot be echoed back, logged as conversation, or persisted to a run step.
147
+ * It lives exactly as long as the process does.
148
+ *
149
+ * ## Resolution order — credentials first, env as the documented fallback
150
+ *
151
+ * Mirrors how MJ's AI layer already resolves vendor keys, because operators should not have to
152
+ * learn a second scheme:
153
+ *
154
+ * 1. `MJ: AI Agent Credentials` grants for this agent, read from `MJ: Credentials`. The
155
+ * governed path — auditable, revocable, per-agent.
156
+ * 2. The server's own `process.env[EnvVariableName]`. If a harness needs ANTHROPIC_API_KEY and
157
+ * no credential row grants one, the MJAPI process's own value is used.
158
+ * 3. When the agent has no grants at all, the harness vendor's key under the existing
159
+ * `AI_VENDOR_API_KEY__<DRIVER>` convention — the zero-config path.
160
+ *
161
+ * Preferring credentials matters: env vars are process-wide, so falling back means an agent gets
162
+ * whatever the server holds rather than only what it was granted. That is the pragmatic path for
163
+ * dev and single-tenant installs, and the reason multi-tenant deployments should grant
164
+ * explicitly. The distinction is logged, not silent.
165
+ */
166
+ private resolveGrantedEnvironment;
167
+ /**
168
+ * Zero-config path: use the harness vendor's key from the environment when the agent has no
169
+ * explicit grants.
170
+ *
171
+ * Uses the same `AI_VENDOR_API_KEY__<DRIVER>` convention the AI layer already uses, so a
172
+ * developer who has MJ talking to Anthropic already has Claude Code working without seeding a
173
+ * credential row.
174
+ */
175
+ private applyVendorKeyFallback;
176
+ /**
177
+ * Reads a credential's value.
178
+ *
179
+ * Custody stays in `MJ: Credentials` — this only reads what the agent was granted, and does not
180
+ * cache it beyond the session.
181
+ */
182
+ private loadCredentialValue;
183
+ /** Loads the harness registry row this agent selected by name. */
184
+ private loadHarnessRow;
185
+ /** Resolves the adapter class named by the harness row. */
186
+ private resolveAdapter;
187
+ /** Chooses a sandbox provider from the agent's configuration. */
188
+ private createSandboxProvider;
189
+ /** Reads and parses the harness block from the agent's TypeConfiguration. */
190
+ private readHarnessConfig;
191
+ /**
192
+ * The text handed to the harness for this turn.
193
+ *
194
+ * ## Turn 1 carries the RENDERED system prompt — this is not optional
195
+ *
196
+ * The agent-type system prompt template holds the turn-end contract AND, critically, the
197
+ * `_OUTPUT_EXAMPLE` placeholder that shows the harness the exact JSON envelope shape. In the
198
+ * normal Loop path `AIPromptRunner` renders that template; a harness turn bypasses
199
+ * AIPromptRunner, so without rendering it here the harness never sees the schema at all.
200
+ *
201
+ * The failure that caused is worth recording, because it did not look like a missing prompt.
202
+ * The harness emitted well-formed JSON and simply GUESSED the vocabulary — `nextStep.type` came
203
+ * back as `complete`, then `respond`, then `undefined`, none of which are Loop step names. Five
204
+ * turns were burned while BaseAgent's retry feedback taught it the contract one rejection at a
205
+ * time, turning a one-turn question into a two-minute run. A model inventing plausible values
206
+ * for a schema it was never shown reads as a sloppy model; it is actually a missing prompt.
207
+ *
208
+ * Later turns send only the conversation: the harness has the contract from turn 1 and, where
209
+ * `SessionResume` is true, still has it in session context.
210
+ */
211
+ private buildTurnInput;
212
+ /**
213
+ * The turn-end contract, carrying the ACTUAL envelope schema.
214
+ *
215
+ * Deliberately does not depend on template rendering succeeding. The schema reaches the harness
216
+ * from `AIPrompt.OutputExample` directly, because the first attempt at this relied on the
217
+ * agent-type template rendering `_OUTPUT_EXAMPLE` — and when that silently fell back to raw
218
+ * template text, the harness received the literal string `{{ _OUTPUT_EXAMPLE }}` and was no
219
+ * better off than before. It then invented step names (`complete`, `result`, `undefined`) across
220
+ * five wasted turns.
221
+ *
222
+ * The step vocabulary is listed explicitly too. A harness that knows the SHAPE but guesses the
223
+ * VALUES still fails validation, and that is precisely the failure mode observed: well-formed
224
+ * JSON, invented `nextStep.type`.
225
+ */
226
+ private buildTurnEndContract;
227
+ /**
228
+ * Renders the agent type's system prompt through the same template engine AIPromptRunner uses,
229
+ * so the harness receives exactly what a Loop model would — including the output example.
230
+ *
231
+ * Falls back to the raw template text if rendering fails. A partially-substituted prompt still
232
+ * carries the envelope shape and lets the run proceed; throwing here would fail a run over a
233
+ * template warning, which is the worse trade.
234
+ */
235
+ private renderSystemPrompt;
236
+ /** Shapes a harness turn as the prompt result the rest of BaseAgent expects. */
237
+ private buildPromptResult;
238
+ /**
239
+ * Wires {@link EndHarnessSession} into `BaseAgent.Execute()`'s per-run teardown. `Execute()`
240
+ * calls this exactly once per call, from its top-level `finally` block, regardless of outcome —
241
+ * without it, `startHarnessSession()`'s provisioned sandbox (a live Docker container or a
242
+ * workspace directory) is never finalized and leaks for the life of the host process.
243
+ */
244
+ protected finalizeRun(outcome: 'success' | 'failure' | 'cancelled'): Promise<void>;
245
+ /** Tears the session and sandbox down on every exit path. */
246
+ EndHarnessSession(outcome: 'success' | 'failure' | 'cancelled'): Promise<void>;
247
+ private _agentRunId;
248
+ private _agentRunAgentId;
249
+ /**
250
+ * The agent's type-specific configuration.
251
+ *
252
+ * Read from BaseAgent's `_executeParams`, NOT from `_agentConfig`: the latter is an
253
+ * AgentConfiguration (agentType / systemPrompt / childPrompt) and carries no agent entity, so
254
+ * reaching for `.agent` there silently yields undefined and every run fails with "does not name
255
+ * a harness" no matter how it is configured.
256
+ */
257
+ private _agentRunConversationId;
258
+ private _agentTypeConfiguration;
259
+ private _executeAgentParams;
260
+ }
261
+ //# sourceMappingURL=HarnessAgentBase.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"HarnessAgentBase.d.ts","sourceRoot":"","sources":["../src/HarnessAgentBase.ts"],"names":[],"mappings":"AAEA,OAAO,EAAE,SAAS,EAAE,MAAM,2BAA2B,CAAC;AAEtD,OAAO,EAAE,cAAc,EAAE,iBAAiB,EAAE,MAAM,8BAA8B,CAAC;AA+DjF;;;;;;;;;;;;;;;;;;;;;;;;GAwBG;AACH,qBACa,gBAAiB,SAAQ,SAAS;IAC3C,OAAO,CAAC,OAAO,CAAmC;IAClD,OAAO,CAAC,eAAe,CAAiC;IACxD,OAAO,CAAC,aAAa,CAA8B;IACnD,OAAO,CAAC,UAAU,CAAuC;IACzD,OAAO,CAAC,SAAS,CAAK;IAEtB;;;;;;OAMG;cACsB,aAAa,CAAC,YAAY,EAAE,cAAc,GAAG,OAAO,CAAC,iBAAiB,CAAC;IA0BhG,qFAAqF;YACvE,mBAAmB;IA6CjC;;;;;;OAMG;IACH,OAAO,CAAC,uBAAuB;IAQ/B;;;;;;;;;;;;;;;OAeG;IACH,OAAO,CAAC,yBAAyB;IA2BjC;;;;;;;;;;;;;;;;;;;;;OAqBG;YACW,oBAAoB;IA6ClC,gEAAgE;YAClD,OAAO;IAgCrB;;;;;OAKG;YACW,eAAe;IAkE7B;;;;;;;;;;;;;OAaG;YACW,oBAAoB;IAYlC;;;;;;;;;;;;OAYG;YACW,sBAAsB;IA+CpC;;;;;;;OAOG;IACH,OAAO,CAAC,wBAAwB;IAOhC;;;;;;;;;;;;;;;;;;;;;;;;;;OA0BG;YACW,yBAAyB;IAgDvC;;;;;;;OAOG;IACH,OAAO,CAAC,sBAAsB;IA0B9B;;;;;OAKG;YACW,mBAAmB;IAcjC,kEAAkE;YACpD,cAAc;IA2B5B,2DAA2D;IAC3D,OAAO,CAAC,cAAc;IAiBtB,iEAAiE;IACjE,OAAO,CAAC,qBAAqB;IAM7B,6EAA6E;IAC7E,OAAO,CAAC,iBAAiB;IAazB;;;;;;;;;;;;;;;;;;;OAmBG;YACW,cAAc;IA4B5B;;;;;;;;;;;;;OAaG;IACH,OAAO,CAAC,oBAAoB;IAsB5B;;;;;;;OAOG;YACW,kBAAkB;IA8ChC,gFAAgF;IAChF,OAAO,CAAC,iBAAiB;IAkCzB;;;;;OAKG;cACsB,WAAW,CAAC,OAAO,EAAE,SAAS,GAAG,SAAS,GAAG,WAAW,GAAG,OAAO,CAAC,IAAI,CAAC;IAIjG,6DAA6D;IAChD,iBAAiB,CAAC,OAAO,EAAE,SAAS,GAAG,SAAS,GAAG,WAAW,GAAG,OAAO,CAAC,IAAI,CAAC;IAuB3F,OAAO,CAAC,WAAW;IAInB,OAAO,CAAC,gBAAgB;IAIxB;;;;;;;OAOG;IACH,OAAO,CAAC,uBAAuB;IAI/B,OAAO,CAAC,uBAAuB;IAI/B,OAAO,CAAC,mBAAmB;CAK9B"}