@memberjunction/ai-agent-harness 0.0.0 → 6.1.0-edge.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +7 -0
- package/README.md +192 -27
- package/dist/HarnessAgentBase.d.ts +254 -0
- package/dist/HarnessAgentBase.d.ts.map +1 -0
- package/dist/HarnessAgentBase.js +813 -0
- package/dist/HarnessAgentBase.js.map +1 -0
- package/dist/HarnessAgentType.d.ts +39 -0
- package/dist/HarnessAgentType.d.ts.map +1 -0
- package/dist/HarnessAgentType.js +50 -0
- package/dist/HarnessAgentType.js.map +1 -0
- package/dist/adapters/BaseCliHarnessAdapter.d.ts +93 -0
- package/dist/adapters/BaseCliHarnessAdapter.d.ts.map +1 -0
- package/dist/adapters/BaseCliHarnessAdapter.js +184 -0
- package/dist/adapters/BaseCliHarnessAdapter.js.map +1 -0
- package/dist/adapters/BaseHarnessAdapter.d.ts +114 -0
- package/dist/adapters/BaseHarnessAdapter.d.ts.map +1 -0
- package/dist/adapters/BaseHarnessAdapter.js +86 -0
- package/dist/adapters/BaseHarnessAdapter.js.map +1 -0
- package/dist/adapters/ClaudeCodeCliAdapter.d.ts +104 -0
- package/dist/adapters/ClaudeCodeCliAdapter.d.ts.map +1 -0
- package/dist/adapters/ClaudeCodeCliAdapter.js +268 -0
- package/dist/adapters/ClaudeCodeCliAdapter.js.map +1 -0
- package/dist/adapters/CodexAdapter.d.ts +27 -0
- package/dist/adapters/CodexAdapter.d.ts.map +1 -0
- package/dist/adapters/CodexAdapter.js +117 -0
- package/dist/adapters/CodexAdapter.js.map +1 -0
- package/dist/adapters/GeminiCliAdapter.d.ts +25 -0
- package/dist/adapters/GeminiCliAdapter.d.ts.map +1 -0
- package/dist/adapters/GeminiCliAdapter.js +98 -0
- package/dist/adapters/GeminiCliAdapter.js.map +1 -0
- package/dist/adapters/OpenCodeAdapter.d.ts +23 -0
- package/dist/adapters/OpenCodeAdapter.d.ts.map +1 -0
- package/dist/adapters/OpenCodeAdapter.js +104 -0
- package/dist/adapters/OpenCodeAdapter.js.map +1 -0
- package/dist/adapters/PiAdapter.d.ts +73 -0
- package/dist/adapters/PiAdapter.d.ts.map +1 -0
- package/dist/adapters/PiAdapter.js +237 -0
- package/dist/adapters/PiAdapter.js.map +1 -0
- package/dist/adapters/StdioJsonAdapter.d.ts +43 -0
- package/dist/adapters/StdioJsonAdapter.d.ts.map +1 -0
- package/dist/adapters/StdioJsonAdapter.js +109 -0
- package/dist/adapters/StdioJsonAdapter.js.map +1 -0
- package/dist/index.d.ts +26 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +28 -0
- package/dist/index.js.map +1 -0
- package/dist/sandbox/ChildProcessExecutor.d.ts +41 -0
- package/dist/sandbox/ChildProcessExecutor.d.ts.map +1 -0
- package/dist/sandbox/ChildProcessExecutor.js +86 -0
- package/dist/sandbox/ChildProcessExecutor.js.map +1 -0
- package/dist/sandbox/DockerSandboxProvider.d.ts +64 -0
- package/dist/sandbox/DockerSandboxProvider.d.ts.map +1 -0
- package/dist/sandbox/DockerSandboxProvider.js +176 -0
- package/dist/sandbox/DockerSandboxProvider.js.map +1 -0
- package/dist/sandbox/ISandboxProvider.d.ts +71 -0
- package/dist/sandbox/ISandboxProvider.d.ts.map +1 -0
- package/dist/sandbox/ISandboxProvider.js +2 -0
- package/dist/sandbox/ISandboxProvider.js.map +1 -0
- package/dist/sandbox/LocalDirectorySandboxProvider.d.ts +37 -0
- package/dist/sandbox/LocalDirectorySandboxProvider.d.ts.map +1 -0
- package/dist/sandbox/LocalDirectorySandboxProvider.js +75 -0
- package/dist/sandbox/LocalDirectorySandboxProvider.js.map +1 -0
- package/dist/sandbox/SandboxExecutor.d.ts +47 -0
- package/dist/sandbox/SandboxExecutor.d.ts.map +1 -0
- package/dist/sandbox/SandboxExecutor.js +2 -0
- package/dist/sandbox/SandboxExecutor.js.map +1 -0
- package/dist/types.d.ts +180 -0
- package/dist/types.d.ts.map +1 -0
- package/dist/types.js +2 -0
- package/dist/types.js.map +1 -0
- package/package.json +35 -8
package/LICENSE
ADDED
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
ISC License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2023 MemberJunction
|
|
4
|
+
|
|
5
|
+
Permission to use, copy, modify, and/or distribute this software for any purpose with or without fee is hereby granted, provided that the above copyright notice and this permission notice appear in all copies.
|
|
6
|
+
|
|
7
|
+
THE SOFTWARE IS PROVIDED "AS IS" AND THE AUTHOR DISCLAIMS ALL WARRANTIES WITH REGARD TO THIS SOFTWARE INCLUDING ALL IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS. IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR ANY SPECIAL, DIRECT, INDIRECT, OR CONSEQUENTIAL DAMAGES OR ANY DAMAGES WHATSOEVER RESULTING FROM LOSS OF USE, DATA OR PROFITS, WHETHER IN AN ACTION OF CONTRACT, NEGLIGENCE OR OTHER TORTIOUS ACTION, ARISING OUT OF OR IN CONNECTION WITH THE USE OR PERFORMANCE OF THIS SOFTWARE.
|
package/README.md
CHANGED
|
@@ -1,45 +1,210 @@
|
|
|
1
1
|
# @memberjunction/ai-agent-harness
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
Run an **external agent harness** — Claude Code, Codex CLI, OpenCode, Gemini CLI, Pi — as the
|
|
4
|
+
reasoning substrate for a MemberJunction agent, while MJ keeps identity, permissions, governed data
|
|
5
|
+
access, payload contracts, HITL, cost control and run-level audit.
|
|
4
6
|
|
|
5
|
-
**
|
|
7
|
+
> **Design principle: the harness is a substrate, not a peer.** MJ owns the run record, the
|
|
8
|
+
> credentials, the tool surface and the approval flow. The harness owns the reasoning inside a turn.
|
|
6
9
|
|
|
7
|
-
|
|
10
|
+
Design plan: [`plans/external-agent-harness.md`](../../../plans/external-agent-harness.md)
|
|
8
11
|
|
|
9
|
-
|
|
12
|
+
---
|
|
13
|
+
|
|
14
|
+
## The core idea: a harness turn *is* a Loop iteration
|
|
15
|
+
|
|
16
|
+
`BaseAgent` already runs an iterate → decide → execute-steps → iterate loop where the "decide" input
|
|
17
|
+
is a prompt execution. This package substitutes a **harness turn** for that prompt call and changes
|
|
18
|
+
nothing else.
|
|
19
|
+
|
|
20
|
+
```
|
|
21
|
+
┌─ BaseAgent loop (unchanged) ──────────────────────────────────┐
|
|
22
|
+
│ │
|
|
23
|
+
│ executePrompt() ──► [ HarnessAgentBase override ] │
|
|
24
|
+
│ │ │ │
|
|
25
|
+
│ │ ├─ adapter.RunTurn(input) │
|
|
26
|
+
│ │ ├─ accumulate usage │
|
|
27
|
+
│ │ └─ write AIPromptRun │
|
|
28
|
+
│ ▼ │
|
|
29
|
+
│ DetermineNextStep (inherited from LoopAgentType) │
|
|
30
|
+
│ ▼ │
|
|
31
|
+
│ validate ─► execute actions / sub-agents / skills │
|
|
32
|
+
│ ▼ │
|
|
33
|
+
│ checkExecutionGuardrails ─► next turn │
|
|
34
|
+
└───────────────────────────────────────────────────────────────┘
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
The harness ends each turn by emitting the **Loop next-step JSON envelope**. MJ then executes any
|
|
38
|
+
actions, sub-agents or skills through its own validated machinery and resumes the session with the
|
|
39
|
+
results.
|
|
40
|
+
|
|
41
|
+
**Why this matters:** every guardrail, payload ACL, HITL gate and accounting path already written for
|
|
42
|
+
Loop agents applies to harness agents with no new enforcement code — and there is exactly **one**
|
|
43
|
+
authority channel to audit, not two.
|
|
44
|
+
|
|
45
|
+
---
|
|
46
|
+
|
|
47
|
+
## Two registrations, one metadata column
|
|
48
|
+
|
|
49
|
+
ClassFactory registrations are namespaced per base class, so `'HarnessAgentType'` is registered
|
|
50
|
+
twice against different roots:
|
|
51
|
+
|
|
52
|
+
| Registered under | Class | Resolved by | Gives you |
|
|
53
|
+
|---|---|---|---|
|
|
54
|
+
| `BaseAgentType` | `HarnessAgentType` | `BaseAgentType.GetAgentTypeInstance` | the turn protocol (inherits Loop) |
|
|
55
|
+
| `BaseAgent` | `HarnessAgentBase` | `AgentRunner` (`AgentRunner.ts:101`) | the execution driver |
|
|
56
|
+
|
|
57
|
+
Both read `AIAgentType.DriverClass`, so **one metadata value selects both halves**. This is the
|
|
58
|
+
mechanism working as designed — `AgentRunner` already treats the type's `DriverClass` as a
|
|
59
|
+
`BaseAgent` key and falls back to plain `BaseAgent` when unregistered, which is why every Loop agent
|
|
60
|
+
gets the base execution class today.
|
|
61
|
+
|
|
62
|
+
---
|
|
63
|
+
|
|
64
|
+
## Adapters
|
|
65
|
+
|
|
66
|
+
| Harness | DriverClass | Mechanism | Notes |
|
|
67
|
+
|---|---|---|---|
|
|
68
|
+
| Claude Code | `ClaudeCodeCliAdapter` | CLI, `stream-json` | Session resume; permission hook pending |
|
|
69
|
+
| Codex | `CodexAdapter` | `codex exec --json` | Session resume |
|
|
70
|
+
| OpenCode | `OpenCodeAdapter` | `opencode run` JSON | Session resume |
|
|
71
|
+
| Gemini CLI | `GeminiCliAdapter` | `gemini --output-format json` | **No** resume — context replayed |
|
|
72
|
+
| Pi | `PiAdapter` | stdio-JSON contract | Requires `ExecutablePath` |
|
|
73
|
+
| *anything* | `StdioJsonAdapter` | documented JSON contract | Escape hatch — zero MJ code |
|
|
10
74
|
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
2. Enable secure, token-less publishing from CI/CD workflows
|
|
14
|
-
3. Establish provenance for packages published under this name
|
|
75
|
+
Register your own with `@RegisterClass(BaseHarnessAdapter, 'MyAdapter')` and point a harness row's
|
|
76
|
+
`DriverClass` at it. No core changes required.
|
|
15
77
|
|
|
16
|
-
|
|
78
|
+
### Capability honesty
|
|
17
79
|
|
|
18
|
-
|
|
80
|
+
`AIAgentHarness.CapabilitySettings` (typed as `IHarnessCapabilitySettings`) declares what an adapter
|
|
81
|
+
**actually implements**, because the runtime *emulates what is missing*. Gemini CLI reports
|
|
82
|
+
`SessionResume: false`, so context is replayed each turn and those extra tokens are budgeted against
|
|
83
|
+
the run's guardrails rather than quietly absorbed.
|
|
19
84
|
|
|
20
|
-
|
|
85
|
+
Claiming a capability that is not wired up produces a silent behavioural gap, not an error. Report
|
|
86
|
+
`false` and let the runtime compensate.
|
|
21
87
|
|
|
22
|
-
|
|
88
|
+
---
|
|
89
|
+
|
|
90
|
+
## Sandboxes: the provider owns process placement
|
|
91
|
+
|
|
92
|
+
Adapters never call `spawn`. They run everything through `SandboxExecutor`, obtained from the
|
|
93
|
+
handle the provider returns.
|
|
94
|
+
|
|
95
|
+
| Provider | Execution | Isolation |
|
|
96
|
+
|---|---|---|
|
|
97
|
+
| `LocalDirectorySandboxProvider` | direct spawn | **None** — dev only |
|
|
98
|
+
| `DockerSandboxProvider` | `docker exec`, container per run | Real FS boundary; `networkPolicy: 'none'` enforced |
|
|
99
|
+
|
|
100
|
+
> ⚠️ The local provider scopes a *directory*; it does **not** contain the *process*. `networkPolicy`
|
|
101
|
+
> is advisory there. Anyone who believes `'none'` is enforced locally has a false sense of
|
|
102
|
+
> containment, which is worse than knowing the boundary is soft.
|
|
103
|
+
|
|
104
|
+
`HarnessProcess` is deliberately **stream-based**, not `ChildProcess`-based — a Kubernetes exec is
|
|
105
|
+
streams over a websocket and a remote runner is HTTP, and neither could honestly implement a
|
|
106
|
+
`ChildProcess` contract. It also makes adapters unit-testable with a fake executor: no binary, no
|
|
107
|
+
container, no network.
|
|
23
108
|
|
|
24
|
-
|
|
25
|
-
2. Configure the trusted publisher (e.g., GitHub Actions)
|
|
26
|
-
3. Specify the repository and workflow that should be allowed to publish
|
|
27
|
-
4. Use the configured workflow to publish your actual package
|
|
109
|
+
### `WorkspacePath` means "as the harness sees it"
|
|
28
110
|
|
|
29
|
-
|
|
111
|
+
A host path under the local provider; a **container-internal** path under Docker. Pass it to harness
|
|
112
|
+
processes — do **not** open it with `fs` unless you know you are on the local provider.
|
|
30
113
|
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
-
|
|
114
|
+
---
|
|
115
|
+
|
|
116
|
+
## Configuration
|
|
117
|
+
|
|
118
|
+
Per-agent, in `AIAgent.TypeConfiguration`, validated against `AIAgentType.ConfigSchema`:
|
|
36
119
|
|
|
37
|
-
|
|
120
|
+
```jsonc
|
|
121
|
+
{
|
|
122
|
+
"harnessName": "Claude Code", // lookup into MJ: AI Agent Harnesses
|
|
123
|
+
"sandbox": {
|
|
124
|
+
"provider": "local", // local | docker
|
|
125
|
+
"image": "ghcr.io/memberjunction/harness-sandbox:latest",
|
|
126
|
+
"workspaceScope": "agent-user", // run | agent | agent-user
|
|
127
|
+
"networkPolicy": "mcp-only"
|
|
128
|
+
}
|
|
129
|
+
}
|
|
130
|
+
```
|
|
38
131
|
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
132
|
+
**Workspace scope** decides how long files live: `run` is discarded, `agent` is shared across every
|
|
133
|
+
run of that agent, `agent-user` (default) is per agent per user — continuity without one user's
|
|
134
|
+
working files leaking into another's session.
|
|
42
135
|
|
|
43
136
|
---
|
|
44
137
|
|
|
45
|
-
|
|
138
|
+
## Accounting — why every turn writes an `AIPromptRun`
|
|
139
|
+
|
|
140
|
+
Run totals are **derived**: `calculateTokenStats` sums `AIAgentRunStep.PromptRun` rollups. A turn
|
|
141
|
+
that records no prompt run contributes nothing, so the run reports zero tokens and zero cost
|
|
142
|
+
*forever* — and its cost ceiling has nothing to compare against.
|
|
143
|
+
|
|
144
|
+
`AIPromptRun.PromptID`, `.ModelID` and `.VendorID` are all **NOT NULL**, and each resolves to a
|
|
145
|
+
**real** catalog row rather than a placeholder:
|
|
146
|
+
|
|
147
|
+
| Column | Resolves to | Why it is not a fiction |
|
|
148
|
+
|---|---|---|
|
|
149
|
+
| `PromptID` | the agent type's system prompt | that template really did produce the turn |
|
|
150
|
+
| `VendorID` | `AIAgentHarness.AIVendorID` | Claude Code really does call Anthropic |
|
|
151
|
+
| `ModelID` | `AIAgentHarness.AIModelID` | the harness really does run that model |
|
|
152
|
+
|
|
153
|
+
If none resolves, the runtime **fails loudly** rather than skipping the row. A silent skip is exactly
|
|
154
|
+
how a cost ceiling stops protecting anything.
|
|
155
|
+
|
|
156
|
+
---
|
|
157
|
+
|
|
158
|
+
## Credentials
|
|
159
|
+
|
|
160
|
+
`MJ: AI Agent Credentials` records the **grant edge** — which credentials an agent carries into its
|
|
161
|
+
sandbox. Custody stays in `MJ: Credentials` / `CredentialEngine`.
|
|
162
|
+
|
|
163
|
+
Environment injection is the **only** channel by which a secret reaches the harness, and it carries
|
|
164
|
+
exactly what was granted — never the MJAPI process environment, never DB credentials, never a user
|
|
165
|
+
token.
|
|
166
|
+
|
|
167
|
+
Distinct from `MJ: AI Credential Bindings`, which is inference-selection plumbing for
|
|
168
|
+
`AIPromptRunner` failover when *MJ itself* executes a prompt.
|
|
169
|
+
|
|
170
|
+
---
|
|
171
|
+
|
|
172
|
+
## The audit boundary
|
|
173
|
+
|
|
174
|
+
MJ records what **crosses the boundary**: MCP loopback reads and the turn-end step. Activity *inside*
|
|
175
|
+
the sandbox — file edits, shell commands — streams to `onProgress` for live view but is **not**
|
|
176
|
+
persisted as run steps.
|
|
177
|
+
|
|
178
|
+
This "opaque super-step" granularity is **intentional**. In-sandbox behaviour is governed by posture
|
|
179
|
+
policy, not by run steps. Widening it is a design change, not a bug fix.
|
|
180
|
+
|
|
181
|
+
---
|
|
182
|
+
|
|
183
|
+
## Deployment
|
|
184
|
+
|
|
185
|
+
| Environment | Provider | Notes |
|
|
186
|
+
|---|---|---|
|
|
187
|
+
| Local dev | `local` | Fast; uses the dev's own installed CLI and auth |
|
|
188
|
+
| Local parity | `docker` | Same path as production |
|
|
189
|
+
| AWS / Azure | `docker` → ECS/Fargate or ACI | Sandbox image versioned **separately** from MJAPI |
|
|
190
|
+
|
|
191
|
+
Do **not** bake harness binaries into the MJAPI image. A harness running inside the API container
|
|
192
|
+
inherits that container's network reach and IAM role — the wrong blast radius for a process
|
|
193
|
+
executing an autonomous agent's shell commands.
|
|
194
|
+
|
|
195
|
+
---
|
|
196
|
+
|
|
197
|
+
## Known gaps
|
|
198
|
+
|
|
199
|
+
- **`PermissionHooks: false` on every adapter.** The `strict` posture needs an MCP permission-prompt
|
|
200
|
+
tool that does not exist yet. Reported honestly so the runtime cannot assume interception it lacks.
|
|
201
|
+
- **`networkPolicy` `mcp-only` / `allowlist` are not enforced at the packet level** under Docker —
|
|
202
|
+
documented as such rather than aliased to `open`.
|
|
203
|
+
- **MCP loopback is not yet wired.** `HarnessSessionConfig` carries the fields; the server and
|
|
204
|
+
per-run scoped credential are still to come.
|
|
205
|
+
- **`ModelID` uses the declared model**, not the model the harness reported for the turn. The
|
|
206
|
+
refinement belongs in `resolveAccountingIds` once adapters surface it.
|
|
207
|
+
|
|
208
|
+
## License
|
|
209
|
+
|
|
210
|
+
ISC
|
|
@@ -0,0 +1,254 @@
|
|
|
1
|
+
import { BaseAgent } from '@memberjunction/ai-agents';
|
|
2
|
+
import { AIPromptParams, AIPromptRunResult } from '@memberjunction/ai-core-plus';
|
|
3
|
+
/**
|
|
4
|
+
* Executes an MJ agent whose reasoning substrate is an external harness.
|
|
5
|
+
*
|
|
6
|
+
* ## The single seam
|
|
7
|
+
*
|
|
8
|
+
* `BaseAgent` already runs an iterate → decide → execute-steps → iterate loop where the "decide"
|
|
9
|
+
* input is a prompt execution. This class overrides exactly one method — {@link executePrompt} — and
|
|
10
|
+
* substitutes a harness turn for that prompt call. Everything else is untouched `BaseAgent`: the
|
|
11
|
+
* loop, next-step validation, action and sub-agent execution, payload merging under ACLs, guardrail
|
|
12
|
+
* checks between iterations, and run-step recording.
|
|
13
|
+
*
|
|
14
|
+
* That is worth stating precisely because it is the whole architectural bet. `executePrompt` is a
|
|
15
|
+
* five-line protected method with a single call site, so substituting it changes what produces a
|
|
16
|
+
* decision without changing anything about how decisions are validated or enforced.
|
|
17
|
+
*
|
|
18
|
+
* ## Accounting is not optional
|
|
19
|
+
*
|
|
20
|
+
* A harness turn must produce a real `AIPromptRun` row. Run totals are DERIVED — `calculateTokenStats`
|
|
21
|
+
* sums `AIAgentRunStep.PromptRun` rollups — so a turn that records no prompt run contributes nothing,
|
|
22
|
+
* and the run reports zero tokens and zero cost forever. Combined with the cost guardrail, that means
|
|
23
|
+
* a runaway harness would never be interrupted: the ceiling would have nothing to compare against.
|
|
24
|
+
*
|
|
25
|
+
* `AIPromptRun.PromptID`, `.ModelID` and `.VendorID` are all NOT NULL, and every one is resolved to a
|
|
26
|
+
* REAL catalog row rather than a placeholder — see {@link resolveAccountingIds}.
|
|
27
|
+
*/
|
|
28
|
+
export declare class HarnessAgentBase extends BaseAgent {
|
|
29
|
+
private adapter;
|
|
30
|
+
private sandboxProvider;
|
|
31
|
+
private sandboxHandle;
|
|
32
|
+
private harnessRow;
|
|
33
|
+
private turnIndex;
|
|
34
|
+
/**
|
|
35
|
+
* Substitutes a harness turn for the prompt call a Loop agent would make.
|
|
36
|
+
*
|
|
37
|
+
* The returned {@link AIPromptRunResult} is shaped exactly as `AIPromptRunner` would shape one,
|
|
38
|
+
* because everything downstream — `DetermineNextStep`, the malformed-response retry machinery,
|
|
39
|
+
* step recording — reads it without knowing or caring that a harness produced it.
|
|
40
|
+
*/
|
|
41
|
+
protected executePrompt(promptParams: AIPromptParams): Promise<AIPromptRunResult>;
|
|
42
|
+
/** Provisions the sandbox, resolves credentials and launches the harness session. */
|
|
43
|
+
private startHarnessSession;
|
|
44
|
+
/**
|
|
45
|
+
* Resolves what the agent may do inside its sandbox.
|
|
46
|
+
*
|
|
47
|
+
* Defaults to `strict` when unset — the safe direction. An agent that has never been given a
|
|
48
|
+
* posture should be unable to mutate anything, rather than inheriting whatever the harness does
|
|
49
|
+
* by default, which for a coding agent is a great deal.
|
|
50
|
+
*/
|
|
51
|
+
private resolvePermissionPolicy;
|
|
52
|
+
/**
|
|
53
|
+
* Says out loud when a configured policy will not actually be enforced.
|
|
54
|
+
*
|
|
55
|
+
* Two distinct gaps, previously conflated behind one `PermissionHooks` check — which is why four
|
|
56
|
+
* adapters could ignore a policy entirely while the runtime warned about something else:
|
|
57
|
+
*
|
|
58
|
+
* 1. **`PermissionPolicy: false`** — the adapter never translated the policy into harness flags.
|
|
59
|
+
* The posture and allow/deny lists are inert; the harness runs on its own defaults. This is
|
|
60
|
+
* the serious one, because the agent's metadata reads as though something is gated.
|
|
61
|
+
* 2. **`PermissionHooks: false` under `strict`** — the policy applies, but there is no channel to
|
|
62
|
+
* route an approval through, so anything requiring one is denied rather than escalated.
|
|
63
|
+
*
|
|
64
|
+
* Warn, don't fail. Refusing the run would take every adapter without a verified flag vocabulary
|
|
65
|
+
* offline, and an unenforced policy on a properly-provisioned sandbox is still contained by the
|
|
66
|
+
* sandbox. What is not acceptable is the operator not knowing which situation they are in.
|
|
67
|
+
*/
|
|
68
|
+
private warnOnUnenforceablePolicy;
|
|
69
|
+
/**
|
|
70
|
+
* Finds a prior harness session this run can continue, if the adapter can use one.
|
|
71
|
+
*
|
|
72
|
+
* ## Why this is worth doing
|
|
73
|
+
*
|
|
74
|
+
* Without it, every message in a conversation opens a COLD session and MJ replays the whole
|
|
75
|
+
* history into it. Measured on two consecutive messages in one conversation: the second cost
|
|
76
|
+
* $0.0448 against the first's $0.0155 — nearly 3x, spent entirely on re-reading context the
|
|
77
|
+
* harness had already been told once.
|
|
78
|
+
*
|
|
79
|
+
* ## Three gates, each guarding a different way this goes wrong
|
|
80
|
+
*
|
|
81
|
+
* 1. `SessionResume` capability — a harness that cannot resume must keep replaying. Offering a
|
|
82
|
+
* session id to an adapter that ignores it is harmless; ASSUMING it resumed is not, which is
|
|
83
|
+
* why the outcome is reported back rather than inferred.
|
|
84
|
+
* 2. Workspace scope must be durable. Harnesses key their session store by working directory, so
|
|
85
|
+
* a `run`-scoped workspace is a new directory every time and the session would never be
|
|
86
|
+
* found. Gating here keeps the failure at "no resume" rather than a silent miss.
|
|
87
|
+
* 3. Same conversation. That is the continuity boundary users already understand — a time-based
|
|
88
|
+
* cache would expire while someone is at lunch and, worse, leak stale context into an
|
|
89
|
+
* unrelated new conversation.
|
|
90
|
+
*/
|
|
91
|
+
private findResumableSession;
|
|
92
|
+
/** Accumulates one turn's event stream into a single result. */
|
|
93
|
+
private runTurn;
|
|
94
|
+
/**
|
|
95
|
+
* Writes the `AIPromptRun` that carries this turn's usage.
|
|
96
|
+
*
|
|
97
|
+
* Returns undefined only when the row could not be created, which is logged loudly rather than
|
|
98
|
+
* swallowed: without it the run's cost and token totals stay at zero and its guardrails go blind.
|
|
99
|
+
*/
|
|
100
|
+
private recordPromptRun;
|
|
101
|
+
/**
|
|
102
|
+
* Resolves the three NOT NULL foreign keys on `AIPromptRun` to REAL catalog rows.
|
|
103
|
+
*
|
|
104
|
+
* None of these is a placeholder, which is the point — inventing catalog rows to satisfy a
|
|
105
|
+
* constraint would pollute the model and vendor catalogs with fictions that then show up in
|
|
106
|
+
* every cost report:
|
|
107
|
+
*
|
|
108
|
+
* · PromptID — the agent type's system prompt. The harness turn really was produced by that
|
|
109
|
+
* template; it is the same one the Loop type renders.
|
|
110
|
+
* · VendorID — `AIAgentHarness.AIVendorID`. Claude Code really does call Anthropic.
|
|
111
|
+
* · ModelID — `AIAgentHarness.AIModelID`. Ideally this would be the model the harness
|
|
112
|
+
* REPORTED for the turn, resolved by name; that refinement belongs here once adapters
|
|
113
|
+
* surface it, and falls back to the declared model meanwhile.
|
|
114
|
+
*/
|
|
115
|
+
private resolveAccountingIds;
|
|
116
|
+
/**
|
|
117
|
+
* Resolves the model the harness REPORTED using to an MJ catalog row.
|
|
118
|
+
*
|
|
119
|
+
* Recording the model we assumed rather than the one that ran is not a cosmetic problem: a
|
|
120
|
+
* harness picks its own model unless told otherwise, and Opus and Sonnet are not the same price,
|
|
121
|
+
* so the run's cost is attributed to the wrong model. Observed live — the harness ran
|
|
122
|
+
* `claude-opus-4-6` while the run recorded Claude Sonnet 5, purely because that was the harness
|
|
123
|
+
* row's declared anchor.
|
|
124
|
+
*
|
|
125
|
+
* Returns null when the reported name matches nothing, letting the caller fall back to the
|
|
126
|
+
* declared anchor. A miss is expected for a model newer than the catalog and must not fail the
|
|
127
|
+
* run — an approximate attribution still beats no AIPromptRun at all.
|
|
128
|
+
*/
|
|
129
|
+
private resolveReportedModelId;
|
|
130
|
+
/**
|
|
131
|
+
* Records the harness session on the run.
|
|
132
|
+
*
|
|
133
|
+
* AIAgentRun.ExternalSessionID exists precisely so an MJ run can be correlated with the vendor's
|
|
134
|
+
* own session logs when diagnosing in-sandbox behaviour — the one place MJ's audit trail
|
|
135
|
+
* deliberately stops. It was added, documented, and then never populated, so answering "did this
|
|
136
|
+
* run resume its session?" meant reading the vendor's files off disk instead of the run record.
|
|
137
|
+
*/
|
|
138
|
+
private persistExternalSessionId;
|
|
139
|
+
/**
|
|
140
|
+
* Builds the environment injected into the sandbox.
|
|
141
|
+
*
|
|
142
|
+
* ## Secrets travel as process environment, never as prompt text
|
|
143
|
+
*
|
|
144
|
+
* Everything resolved here is handed to the sandbox executor and becomes the harness PROCESS's
|
|
145
|
+
* environment. None of it is rendered into the turn prompt, so a credential never enters the
|
|
146
|
+
* model's context and cannot be echoed back, logged as conversation, or persisted to a run step.
|
|
147
|
+
* It lives exactly as long as the process does.
|
|
148
|
+
*
|
|
149
|
+
* ## Resolution order — credentials first, env as the documented fallback
|
|
150
|
+
*
|
|
151
|
+
* Mirrors how MJ's AI layer already resolves vendor keys, because operators should not have to
|
|
152
|
+
* learn a second scheme:
|
|
153
|
+
*
|
|
154
|
+
* 1. `MJ: AI Agent Credentials` grants for this agent, read from `MJ: Credentials`. The
|
|
155
|
+
* governed path — auditable, revocable, per-agent.
|
|
156
|
+
* 2. The server's own `process.env[EnvVariableName]`. If a harness needs ANTHROPIC_API_KEY and
|
|
157
|
+
* no credential row grants one, the MJAPI process's own value is used.
|
|
158
|
+
* 3. When the agent has no grants at all, the harness vendor's key under the existing
|
|
159
|
+
* `AI_VENDOR_API_KEY__<DRIVER>` convention — the zero-config path.
|
|
160
|
+
*
|
|
161
|
+
* Preferring credentials matters: env vars are process-wide, so falling back means an agent gets
|
|
162
|
+
* whatever the server holds rather than only what it was granted. That is the pragmatic path for
|
|
163
|
+
* dev and single-tenant installs, and the reason multi-tenant deployments should grant
|
|
164
|
+
* explicitly. The distinction is logged, not silent.
|
|
165
|
+
*/
|
|
166
|
+
private resolveGrantedEnvironment;
|
|
167
|
+
/**
|
|
168
|
+
* Zero-config path: use the harness vendor's key from the environment when the agent has no
|
|
169
|
+
* explicit grants.
|
|
170
|
+
*
|
|
171
|
+
* Uses the same `AI_VENDOR_API_KEY__<DRIVER>` convention the AI layer already uses, so a
|
|
172
|
+
* developer who has MJ talking to Anthropic already has Claude Code working without seeding a
|
|
173
|
+
* credential row.
|
|
174
|
+
*/
|
|
175
|
+
private applyVendorKeyFallback;
|
|
176
|
+
/**
|
|
177
|
+
* Reads a credential's value.
|
|
178
|
+
*
|
|
179
|
+
* Custody stays in `MJ: Credentials` — this only reads what the agent was granted, and does not
|
|
180
|
+
* cache it beyond the session.
|
|
181
|
+
*/
|
|
182
|
+
private loadCredentialValue;
|
|
183
|
+
/** Loads the harness registry row this agent selected by name. */
|
|
184
|
+
private loadHarnessRow;
|
|
185
|
+
/** Resolves the adapter class named by the harness row. */
|
|
186
|
+
private resolveAdapter;
|
|
187
|
+
/** Chooses a sandbox provider from the agent's configuration. */
|
|
188
|
+
private createSandboxProvider;
|
|
189
|
+
/** Reads and parses the harness block from the agent's TypeConfiguration. */
|
|
190
|
+
private readHarnessConfig;
|
|
191
|
+
/**
|
|
192
|
+
* The text handed to the harness for this turn.
|
|
193
|
+
*
|
|
194
|
+
* ## Turn 1 carries the RENDERED system prompt — this is not optional
|
|
195
|
+
*
|
|
196
|
+
* The agent-type system prompt template holds the turn-end contract AND, critically, the
|
|
197
|
+
* `_OUTPUT_EXAMPLE` placeholder that shows the harness the exact JSON envelope shape. In the
|
|
198
|
+
* normal Loop path `AIPromptRunner` renders that template; a harness turn bypasses
|
|
199
|
+
* AIPromptRunner, so without rendering it here the harness never sees the schema at all.
|
|
200
|
+
*
|
|
201
|
+
* The failure that caused is worth recording, because it did not look like a missing prompt.
|
|
202
|
+
* The harness emitted well-formed JSON and simply GUESSED the vocabulary — `nextStep.type` came
|
|
203
|
+
* back as `complete`, then `respond`, then `undefined`, none of which are Loop step names. Five
|
|
204
|
+
* turns were burned while BaseAgent's retry feedback taught it the contract one rejection at a
|
|
205
|
+
* time, turning a one-turn question into a two-minute run. A model inventing plausible values
|
|
206
|
+
* for a schema it was never shown reads as a sloppy model; it is actually a missing prompt.
|
|
207
|
+
*
|
|
208
|
+
* Later turns send only the conversation: the harness has the contract from turn 1 and, where
|
|
209
|
+
* `SessionResume` is true, still has it in session context.
|
|
210
|
+
*/
|
|
211
|
+
private buildTurnInput;
|
|
212
|
+
/**
|
|
213
|
+
* The turn-end contract, carrying the ACTUAL envelope schema.
|
|
214
|
+
*
|
|
215
|
+
* Deliberately does not depend on template rendering succeeding. The schema reaches the harness
|
|
216
|
+
* from `AIPrompt.OutputExample` directly, because the first attempt at this relied on the
|
|
217
|
+
* agent-type template rendering `_OUTPUT_EXAMPLE` — and when that silently fell back to raw
|
|
218
|
+
* template text, the harness received the literal string `{{ _OUTPUT_EXAMPLE }}` and was no
|
|
219
|
+
* better off than before. It then invented step names (`complete`, `result`, `undefined`) across
|
|
220
|
+
* five wasted turns.
|
|
221
|
+
*
|
|
222
|
+
* The step vocabulary is listed explicitly too. A harness that knows the SHAPE but guesses the
|
|
223
|
+
* VALUES still fails validation, and that is precisely the failure mode observed: well-formed
|
|
224
|
+
* JSON, invented `nextStep.type`.
|
|
225
|
+
*/
|
|
226
|
+
private buildTurnEndContract;
|
|
227
|
+
/**
|
|
228
|
+
* Renders the agent type's system prompt through the same template engine AIPromptRunner uses,
|
|
229
|
+
* so the harness receives exactly what a Loop model would — including the output example.
|
|
230
|
+
*
|
|
231
|
+
* Falls back to the raw template text if rendering fails. A partially-substituted prompt still
|
|
232
|
+
* carries the envelope shape and lets the run proceed; throwing here would fail a run over a
|
|
233
|
+
* template warning, which is the worse trade.
|
|
234
|
+
*/
|
|
235
|
+
private renderSystemPrompt;
|
|
236
|
+
/** Shapes a harness turn as the prompt result the rest of BaseAgent expects. */
|
|
237
|
+
private buildPromptResult;
|
|
238
|
+
/** Tears the session and sandbox down on every exit path. */
|
|
239
|
+
EndHarnessSession(outcome: 'success' | 'failure' | 'cancelled'): Promise<void>;
|
|
240
|
+
private _agentRunId;
|
|
241
|
+
private _agentRunAgentId;
|
|
242
|
+
/**
|
|
243
|
+
* The agent's type-specific configuration.
|
|
244
|
+
*
|
|
245
|
+
* Read from BaseAgent's `_executeParams`, NOT from `_agentConfig`: the latter is an
|
|
246
|
+
* AgentConfiguration (agentType / systemPrompt / childPrompt) and carries no agent entity, so
|
|
247
|
+
* reaching for `.agent` there silently yields undefined and every run fails with "does not name
|
|
248
|
+
* a harness" no matter how it is configured.
|
|
249
|
+
*/
|
|
250
|
+
private _agentRunConversationId;
|
|
251
|
+
private _agentTypeConfiguration;
|
|
252
|
+
private _executeAgentParams;
|
|
253
|
+
}
|
|
254
|
+
//# sourceMappingURL=HarnessAgentBase.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"HarnessAgentBase.d.ts","sourceRoot":"","sources":["../src/HarnessAgentBase.ts"],"names":[],"mappings":"AAEA,OAAO,EAAE,SAAS,EAAE,MAAM,2BAA2B,CAAC;AAEtD,OAAO,EAAE,cAAc,EAAE,iBAAiB,EAAE,MAAM,8BAA8B,CAAC;AA+DjF;;;;;;;;;;;;;;;;;;;;;;;;GAwBG;AACH,qBACa,gBAAiB,SAAQ,SAAS;IAC3C,OAAO,CAAC,OAAO,CAAmC;IAClD,OAAO,CAAC,eAAe,CAAiC;IACxD,OAAO,CAAC,aAAa,CAA8B;IACnD,OAAO,CAAC,UAAU,CAAuC;IACzD,OAAO,CAAC,SAAS,CAAK;IAEtB;;;;;;OAMG;cACsB,aAAa,CAAC,YAAY,EAAE,cAAc,GAAG,OAAO,CAAC,iBAAiB,CAAC;IA0BhG,qFAAqF;YACvE,mBAAmB;IA6CjC;;;;;;OAMG;IACH,OAAO,CAAC,uBAAuB;IAQ/B;;;;;;;;;;;;;;;OAeG;IACH,OAAO,CAAC,yBAAyB;IA2BjC;;;;;;;;;;;;;;;;;;;;;OAqBG;YACW,oBAAoB;IA6ClC,gEAAgE;YAClD,OAAO;IAgCrB;;;;;OAKG;YACW,eAAe;IAkE7B;;;;;;;;;;;;;OAaG;YACW,oBAAoB;IAYlC;;;;;;;;;;;;OAYG;YACW,sBAAsB;IA+CpC;;;;;;;OAOG;IACH,OAAO,CAAC,wBAAwB;IAOhC;;;;;;;;;;;;;;;;;;;;;;;;;;OA0BG;YACW,yBAAyB;IAgDvC;;;;;;;OAOG;IACH,OAAO,CAAC,sBAAsB;IA0B9B;;;;;OAKG;YACW,mBAAmB;IAcjC,kEAAkE;YACpD,cAAc;IA2B5B,2DAA2D;IAC3D,OAAO,CAAC,cAAc;IAiBtB,iEAAiE;IACjE,OAAO,CAAC,qBAAqB;IAM7B,6EAA6E;IAC7E,OAAO,CAAC,iBAAiB;IAazB;;;;;;;;;;;;;;;;;;;OAmBG;YACW,cAAc;IA4B5B;;;;;;;;;;;;;OAaG;IACH,OAAO,CAAC,oBAAoB;IAsB5B;;;;;;;OAOG;YACW,kBAAkB;IA8ChC,gFAAgF;IAChF,OAAO,CAAC,iBAAiB;IAkCzB,6DAA6D;IAChD,iBAAiB,CAAC,OAAO,EAAE,SAAS,GAAG,SAAS,GAAG,WAAW,GAAG,OAAO,CAAC,IAAI,CAAC;IAuB3F,OAAO,CAAC,WAAW;IAInB,OAAO,CAAC,gBAAgB;IAIxB;;;;;;;OAOG;IACH,OAAO,CAAC,uBAAuB;IAI/B,OAAO,CAAC,uBAAuB;IAI/B,OAAO,CAAC,mBAAmB;CAK9B"}
|