@odla-ai/harness 0.7.0 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/{chunk-K76I2TCQ.js → chunk-CR6RE3A2.js} +3 -3
- package/dist/{chunk-QZXCQSPZ.js → chunk-ISR434K7.js} +353 -76
- package/dist/chunk-ISR434K7.js.map +1 -0
- package/dist/{chunk-XWXRNPGY.js → chunk-NWNBSX56.js} +4 -4
- package/dist/{chunk-FVOMJKWK.js → chunk-Q7CKOT7T.js} +2 -2
- package/dist/{chunk-LNQNFGQC.js → chunk-RXNHCGWE.js} +1 -1
- package/dist/chunk-RXNHCGWE.js.map +1 -0
- package/dist/cli.cjs.map +1 -1
- package/dist/cli.js +4 -4
- package/dist/code-runtime-cli.cjs +353 -76
- package/dist/code-runtime-cli.cjs.map +1 -1
- package/dist/code-runtime-cli.js +4 -4
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +2 -2
- package/dist/index.d.ts +2 -2
- package/dist/index.js +2 -2
- package/dist/node.cjs +362 -85
- package/dist/node.cjs.map +1 -1
- package/dist/node.d.cts +2 -2
- package/dist/node.d.ts +2 -2
- package/dist/node.js +5 -5
- package/dist/testing.cjs.map +1 -1
- package/dist/testing.d.cts +1 -1
- package/dist/testing.d.ts +1 -1
- package/dist/testing.js +1 -1
- package/dist/{types-CK5EKmKm.d.cts → types-BNJikP5h.d.cts} +53 -5
- package/dist/{types-CK5EKmKm.d.ts → types-BNJikP5h.d.ts} +53 -5
- package/package.json +1 -1
- package/dist/chunk-LNQNFGQC.js.map +0 -1
- package/dist/chunk-QZXCQSPZ.js.map +0 -1
- /package/dist/{chunk-K76I2TCQ.js.map → chunk-CR6RE3A2.js.map} +0 -0
- /package/dist/{chunk-XWXRNPGY.js.map → chunk-NWNBSX56.js.map} +0 -0
- /package/dist/{chunk-FVOMJKWK.js.map → chunk-Q7CKOT7T.js.map} +0 -0
package/dist/node.d.cts
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import {
|
|
1
|
+
import { t as HarnessTaskSpec, g as HarnessAgentOutput, f as HarnessAgentInput, H as HarnessControlPlane, w as HarnessToolBroker, a as HarnessLease, d as HarnessInferenceRequest, e as HarnessInferenceResponse, i as CodeSessionEventData, x as HarnessToolName, y as HarnessToolRequest } from './types-BNJikP5h.cjs';
|
|
2
2
|
import { CodePortableCheckpoint, CodeSourceFile, CodeVerificationReceipt, CodeCheckpointState } from '@odla-ai/camel/code';
|
|
3
3
|
import { Skill, Inference, AgentRunBudget, AgentRun, CompactionPolicy } from '@odla-ai/ai';
|
|
4
4
|
import { PolicyOutcome } from '@odla-ai/camel/policy';
|
|
@@ -353,7 +353,7 @@ declare const V1_SYSTEM_PROMPT = "You are Theseus, the coding agent inside an od
|
|
|
353
353
|
/** The v2 prompt. v1's said "Use only the odla_read, odla_apply_git_diff, and
|
|
354
354
|
* odla_run_recipe tools", so leaving it in place would have told the model not
|
|
355
355
|
* to touch the tools this milestone exists to test. */
|
|
356
|
-
declare const V2_SYSTEM_PROMPT = "You are the coding agent inside an odla Code harness.\nStart by orienting: odla_list shows the files in the workspace and odla_search\nfinds a literal string across them. Prefer those over guessing a path.\nThen odla_read a bounded range, and odla_apply_git_diff to mutate.\nFor mutations, call odla_apply_git_diff with raw git diff text. It must start\nwith \"diff --git a/<path> b/<path>\", include matching \"---\" and \"+++\" file\nheaders and numbered \"@@\" hunks, and never use \"*** Begin Patch\" wrappers.\nThe workspace, model, and tool effects are controlled by the host broker.\nNever claim a build or test passed unless odla_run_recipe returned that result.";
|
|
356
|
+
declare const V2_SYSTEM_PROMPT = "You are the coding agent inside an odla Code harness.\nStart by orienting: odla_list shows the files in the workspace and odla_search\nfinds a literal string across them. Prefer those over guessing a path.\nThen odla_read a bounded range, and odla_apply_git_diff to mutate. When you\nneed several independent searches or file ranges, issue those read-only calls\ntogether in one turn; their results stay ordered and the harness overlaps them.\nNever issue odla_apply_git_diff or odla_run_recipe alongside another tool call.\nFor mutations, call odla_apply_git_diff with raw git diff text. It must start\nwith \"diff --git a/<path> b/<path>\", include matching \"---\" and \"+++\" file\nheaders and numbered \"@@\" hunks, and never use \"*** Begin Patch\" wrappers.\nThe workspace, model, and tool effects are controlled by the host broker.\nNever claim a build or test passed unless odla_run_recipe returned that result.";
|
|
357
357
|
/** Which tool surface the agent sees. `v1` reproduces the Theseus container's exact
|
|
358
358
|
* three tools so the recorded baseline stays comparable. */
|
|
359
359
|
type CodeSurface = "v1" | "v2" | "v3";
|
package/dist/node.d.ts
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import {
|
|
1
|
+
import { t as HarnessTaskSpec, g as HarnessAgentOutput, f as HarnessAgentInput, H as HarnessControlPlane, w as HarnessToolBroker, a as HarnessLease, d as HarnessInferenceRequest, e as HarnessInferenceResponse, i as CodeSessionEventData, x as HarnessToolName, y as HarnessToolRequest } from './types-BNJikP5h.js';
|
|
2
2
|
import { CodePortableCheckpoint, CodeSourceFile, CodeVerificationReceipt, CodeCheckpointState } from '@odla-ai/camel/code';
|
|
3
3
|
import { Skill, Inference, AgentRunBudget, AgentRun, CompactionPolicy } from '@odla-ai/ai';
|
|
4
4
|
import { PolicyOutcome } from '@odla-ai/camel/policy';
|
|
@@ -353,7 +353,7 @@ declare const V1_SYSTEM_PROMPT = "You are Theseus, the coding agent inside an od
|
|
|
353
353
|
/** The v2 prompt. v1's said "Use only the odla_read, odla_apply_git_diff, and
|
|
354
354
|
* odla_run_recipe tools", so leaving it in place would have told the model not
|
|
355
355
|
* to touch the tools this milestone exists to test. */
|
|
356
|
-
declare const V2_SYSTEM_PROMPT = "You are the coding agent inside an odla Code harness.\nStart by orienting: odla_list shows the files in the workspace and odla_search\nfinds a literal string across them. Prefer those over guessing a path.\nThen odla_read a bounded range, and odla_apply_git_diff to mutate.\nFor mutations, call odla_apply_git_diff with raw git diff text. It must start\nwith \"diff --git a/<path> b/<path>\", include matching \"---\" and \"+++\" file\nheaders and numbered \"@@\" hunks, and never use \"*** Begin Patch\" wrappers.\nThe workspace, model, and tool effects are controlled by the host broker.\nNever claim a build or test passed unless odla_run_recipe returned that result.";
|
|
356
|
+
declare const V2_SYSTEM_PROMPT = "You are the coding agent inside an odla Code harness.\nStart by orienting: odla_list shows the files in the workspace and odla_search\nfinds a literal string across them. Prefer those over guessing a path.\nThen odla_read a bounded range, and odla_apply_git_diff to mutate. When you\nneed several independent searches or file ranges, issue those read-only calls\ntogether in one turn; their results stay ordered and the harness overlaps them.\nNever issue odla_apply_git_diff or odla_run_recipe alongside another tool call.\nFor mutations, call odla_apply_git_diff with raw git diff text. It must start\nwith \"diff --git a/<path> b/<path>\", include matching \"---\" and \"+++\" file\nheaders and numbered \"@@\" hunks, and never use \"*** Begin Patch\" wrappers.\nThe workspace, model, and tool effects are controlled by the host broker.\nNever claim a build or test passed unless odla_run_recipe returned that result.";
|
|
357
357
|
/** Which tool surface the agent sees. `v1` reproduces the Theseus container's exact
|
|
358
358
|
* three tools so the recorded baseline stays comparable. */
|
|
359
359
|
type CodeSurface = "v1" | "v2" | "v3";
|
package/dist/node.js
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import {
|
|
2
2
|
runHarnessRunner,
|
|
3
3
|
runLeasedAttempt
|
|
4
|
-
} from "./chunk-
|
|
4
|
+
} from "./chunk-NWNBSX56.js";
|
|
5
5
|
import {
|
|
6
6
|
CODE_RUNTIME_PROTOCOL_VERSION,
|
|
7
7
|
CodeRuntimeCheckpointManager,
|
|
@@ -44,7 +44,7 @@ import {
|
|
|
44
44
|
validateMemory,
|
|
45
45
|
validateRelativePath,
|
|
46
46
|
verifyCodeCandidate
|
|
47
|
-
} from "./chunk-
|
|
47
|
+
} from "./chunk-ISR434K7.js";
|
|
48
48
|
import {
|
|
49
49
|
assertPinnedImage,
|
|
50
50
|
buildContainerRunArgs,
|
|
@@ -55,9 +55,9 @@ import {
|
|
|
55
55
|
stageWorkspace,
|
|
56
56
|
stageWorkspacePair,
|
|
57
57
|
verifyContainerEngineBoundary
|
|
58
|
-
} from "./chunk-
|
|
59
|
-
import "./chunk-
|
|
60
|
-
import "./chunk-
|
|
58
|
+
} from "./chunk-CR6RE3A2.js";
|
|
59
|
+
import "./chunk-Q7CKOT7T.js";
|
|
60
|
+
import "./chunk-RXNHCGWE.js";
|
|
61
61
|
|
|
62
62
|
// src/code-runtime-memory.ts
|
|
63
63
|
async function mutationId(memory) {
|
package/dist/testing.cjs.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"sources":["../src/testing.ts","../src/types.ts"],"sourcesContent":["import type { OracleResponse } from \"@odla-ai/ai\";\nimport {\n HARNESS_PROTOCOL_VERSION,\n type HarnessCompletion,\n type HarnessControlPlane,\n type HarnessEventInput,\n type HarnessInferenceRequest,\n type HarnessInferenceResponse,\n type HarnessLease,\n} from \"./types\";\n\n/** Create a deterministic, valid lease for harness unit and integration tests. */\nexport function fixtureLease(overrides: Partial<HarnessLease> = {}): HarnessLease {\n return {\n protocolVersion: HARNESS_PROTOCOL_VERSION,\n leaseId: \"lease_fixture\",\n generation: 1,\n expiresAt: Date.now() + 60_000,\n task: {\n taskId: \"task_fixture\",\n attemptId: \"attempt_fixture\",\n title: \"Fixture task\",\n prompt: \"Create HARNESS_RESULT.md\",\n workspace: \"fixture\",\n aiRoute: \"coding\",\n policy: {\n network: \"none\",\n timeoutMs: 30_000,\n maxOutputBytes: 1_000_000,\n maxPatchBytes: 1_000_000,\n },\n },\n ...overrides,\n };\n}\n\n/** In-memory control plane that records runner interactions without network access. */\nexport class FakeHarnessControlPlane implements HarnessControlPlane {\n readonly leases: HarnessLease[] = [];\n readonly events: HarnessEventInput[] = [];\n readonly completions: HarnessCompletion[] = [];\n readonly inferenceRequests: HarnessInferenceRequest[] = [];\n cancelRequested = false;\n response: OracleResponse = {\n id: \"oracle_fixture\",\n model: \"fixture-model\",\n provider: \"openai\",\n role: \"assistant\",\n content: [{ type: \"text\", text: \"deterministic fixture response\" }],\n stopReason: \"end_turn\",\n usage: { inputTokens: 5, outputTokens: 4 },\n };\n\n constructor(...leases: HarnessLease[]) { this.leases.push(...leases); }\n\n async lease(workspaces: string[]): Promise<HarnessLease | null> {\n const index = this.leases.findIndex((candidate) => workspaces.includes(candidate.task.workspace));\n return index < 0 ? null : this.leases.splice(index, 1)[0]!;\n }\n async heartbeat(): Promise<{ cancelRequested: boolean; expiresAt: number }> {\n return { cancelRequested: this.cancelRequested, expiresAt: Date.now() + 60_000 };\n }\n async appendEvents(_attemptId: string, _leaseId: string, events: HarnessEventInput[]): Promise<void> {\n this.events.push(...events);\n }\n async infer(_attemptId: string, _leaseId: string, request: HarnessInferenceRequest): Promise<HarnessInferenceResponse> {\n this.inferenceRequests.push(request);\n return {\n requestId: request.requestId,\n response: this.response,\n receipt: {\n provider: this.response.provider,\n model: this.response.model,\n policyVersion: 1,\n inputTokens: this.response.usage.inputTokens ?? 0,\n outputTokens: this.response.usage.outputTokens ?? 0,\n },\n };\n }\n async complete(_attemptId: string, _leaseId: string, completion: HarnessCompletion): Promise<void> {\n this.completions.push(completion);\n }\n}\n","import type { ChatInput, OracleResponse } from \"@odla-ai/ai\";\n\n/** Current JSONL protocol version exchanged between a runner and an agent container. */\nexport const HARNESS_PROTOCOL_VERSION = 1 as const;\n\n/** Default control-plane route used for model inference requested by coding agents. */\nexport const DEFAULT_AI_ROUTE = \"coding\" as const;\n\n/** Lifecycle state reported for a harness task and its active attempt. */\nexport type HarnessTaskStatus =\n | \"queued\"\n | \"running\"\n | \"cancel_requested\"\n | \"completed\"\n | \"failed\"\n | \"cancelled\";\n\n/** Lifecycle state of one execution attempt for a task. */\nexport type HarnessAttemptStatus = HarnessTaskStatus;\n\n/** Trusted or untrusted participant that emitted a harness event. */\nexport type HarnessActor = \"operator\" | \"runner\" | \"agent\" | \"model\" | \"system\";\n\n/** Resource and isolation limits enforced while an untrusted task executes. */\nexport interface HarnessPolicy {\n network: \"none\";\n timeoutMs: number;\n maxOutputBytes: number;\n maxPatchBytes: number;\n}\n\n/** Immutable task instructions and execution policy delivered with a lease. */\nexport interface HarnessTaskSpec {\n taskId: string;\n attemptId: string;\n title: string;\n prompt: string;\n workspace: string;\n aiRoute: string;\n policy: HarnessPolicy;\n parentAttemptId?: string | null;\n checkpointSeq?: number | null;\n}\n\n/** Time-bound assignment authorizing a runner to execute one task attempt. */\nexport interface HarnessLease {\n protocolVersion: typeof HARNESS_PROTOCOL_VERSION;\n leaseId: string;\n generation: number;\n expiresAt: number;\n task: HarnessTaskSpec;\n}\n\n/** Runner-supplied event before the control plane assigns sequence and task metadata. */\nexport interface HarnessEventInput {\n eventId: string;\n kind: string;\n actor: HarnessActor;\n payload: unknown;\n createdAt: number;\n}\n\n/** Persisted, ordered event associated with a specific task attempt. */\nexport interface HarnessEvent extends HarnessEventInput {\n seq: number;\n taskId: string;\n attemptId: string;\n}\n\n/** List-view metadata for a task and its current attempt. */\nexport interface HarnessTaskSummary {\n taskId: string;\n attemptId: string;\n appId: string;\n env: string;\n title: string;\n prompt: string;\n workspace: string;\n aiRoute: string;\n status: HarnessTaskStatus;\n parentAttemptId: string | null;\n checkpointSeq: number | null;\n runnerId: string | null;\n generation: number;\n createdAt: number;\n updatedAt: number;\n}\n\n/** List-view metadata for one attempt, including retry ancestry and runner ownership. */\nexport interface HarnessAttemptSummary {\n attemptId: string;\n taskId: string;\n parentAttemptId: string | null;\n checkpointSeq: number | null;\n status: HarnessAttemptStatus;\n runnerId: string | null;\n generation: number;\n createdAt: number;\n updatedAt: number;\n}\n\n/** Complete task view including attempts, events, result, and generated patch. */\nexport interface HarnessTaskDetail extends HarnessTaskSummary {\n attempts: HarnessAttemptSummary[];\n events: HarnessEvent[];\n patch: string | null;\n result: unknown;\n}\n\n/** Public control-plane view of a registered harness runner. */\nexport interface HarnessRunnerView {\n runnerId: string;\n appId: string;\n env: string;\n name: string;\n createdAt: number;\n lastSeenAt: number | null;\n revokedAt: number | null;\n}\n\n/** Credential-free normalized model request sent from an agent through the runner. */\nexport interface HarnessInferenceRequest {\n requestId: string;\n /** Exact initial, follow-up, or resume command whose budget this call consumes. */\n interactionId?: string;\n call: ChatInput;\n}\n\n/** Normalized model response plus auditable provider, policy, and token metadata. */\nexport interface HarnessInferenceResponse {\n requestId: string;\n response: OracleResponse;\n receipt: {\n provider: string;\n model: string;\n policyVersion: number;\n inputTokens: number;\n outputTokens: number;\n /**\n * USD charged for this call, priced against the model the control plane\n * actually resolved.\n *\n * ABSENT when the live catalog has no price for that model — never zero.\n * An unpriced call is unknown spend, and reporting it as free is what let\n * a goal's `maxUsd` look enforced while nothing enforced it. The runtime\n * cannot compute this itself: it asks for `brokered` and only the control\n * plane knows which model answered.\n */\n costUsd?: number;\n };\n}\n\n/** Bounded, content-free session activity emitted by a Code runtime. Message\n * bodies are projected separately into the app's owner-private odla-db chat. */\nexport type CodeSessionEventData =\n | { type: \"message\"; actor: \"agent\" | \"system\"; body: string }\n | { type: \"diagnostic\"; level: \"error\"; message: string }\n | { type: \"thinking\"; available: true; durationMs: number }\n | { type: \"tool\"; phase: \"started\"; tool: HarnessToolName }\n | { type: \"tool\"; phase: \"completed\"; tool: HarnessToolName; ok: boolean; durationMs: number }\n | {\n type: \"usage\"; provider: string; model: string;\n inputTokens: number; outputTokens: number; durationMs: number;\n interactionId?: string; interactionTokens?: number; interactionMaxTokens?: number;\n /** USD for this call; absent when the model is unpriced, never zero. */\n costUsd?: number;\n /** Cumulative USD for this owner interaction, when every call in it was\n * priced. Absent the moment one was not, so a partial total can never be\n * mistaken for the whole. */\n interactionCostUsd?: number;\n }\n | {\n type: \"status\"; status: \"running\" | \"idle\" | \"failed\" | \"checkpointed\";\n durationMs?: number;\n };\n\n/** Registry-assigned cursor and timestamp for an owner-visible Code event. */\nexport type CodeSessionEvent = CodeSessionEventData & {\n eventId: string; sequence: number; createdAt: number;\n};\n\n/** Closed set of effects an agent container may request from its trusted broker. */\nexport type HarnessToolName =\n | \"sandbox.read\"\n | \"sandbox.list\"\n | \"sandbox.search\"\n | \"sandbox.overview\"\n | \"sandbox.where_is\"\n | \"sandbox.who_imports\"\n | \"sandbox.who_touches\"\n | \"sandbox.apply_patch\"\n | \"sandbox.run_recipe\";\n/** Correlated, structured tool request emitted by an untrusted agent container. */\nexport interface HarnessToolRequest {\n requestId: string;\n tool: HarnessToolName;\n input: Record<string, unknown>;\n}\n/** Bounded tool result returned to an agent after trusted policy evaluation. */\nexport interface HarnessToolResponse {\n requestId: string;\n ok: boolean;\n content: string;\n details?: Record<string, unknown>;\n}\n\n/** Validated JSONL message emitted by an untrusted agent container. */\nexport type HarnessAgentOutput =\n | {\n protocolVersion: typeof HARNESS_PROTOCOL_VERSION;\n type: \"event\";\n kind: string;\n payload?: unknown;\n }\n | {\n protocolVersion: typeof HARNESS_PROTOCOL_VERSION;\n type: \"inference.request\";\n requestId: string;\n call: ChatInput;\n }\n | ({ protocolVersion: typeof HARNESS_PROTOCOL_VERSION; type: \"tool.request\" } & HarnessToolRequest)\n | {\n protocolVersion: typeof HARNESS_PROTOCOL_VERSION;\n type: \"attempt.complete\";\n status: \"completed\" | \"failed\" | \"cancelled\";\n result?: unknown;\n };\n\n/** JSONL command or inference result written by the trusted runner to an agent. */\nexport type HarnessAgentInput =\n | {\n protocolVersion: typeof HARNESS_PROTOCOL_VERSION;\n type: \"task.start\";\n task: HarnessTaskSpec;\n }\n | {\n protocolVersion: typeof HARNESS_PROTOCOL_VERSION;\n type: \"inference.response\";\n requestId: string;\n response: OracleResponse;\n }\n | ({ protocolVersion: typeof HARNESS_PROTOCOL_VERSION; type: \"tool.response\" } & HarnessToolResponse)\n | {\n protocolVersion: typeof HARNESS_PROTOCOL_VERSION;\n type: \"attempt.cancel\";\n reason: string;\n };\n\n/** Terminal attempt report submitted by a runner to the control plane. */\nexport interface HarnessCompletion {\n status: \"completed\" | \"failed\" | \"cancelled\";\n result?: unknown;\n patch?: string;\n error?: string;\n}\n\n/** Operations a credentialed runner may perform against the harness control plane. */\nexport interface HarnessControlPlane {\n lease(workspaces: string[]): Promise<HarnessLease | null>;\n heartbeat(attemptId: string, leaseId: string): Promise<{ cancelRequested: boolean; expiresAt: number }>;\n appendEvents(attemptId: string, leaseId: string, events: HarnessEventInput[]): Promise<void>;\n infer(attemptId: string, leaseId: string, request: HarnessInferenceRequest): Promise<HarnessInferenceResponse>;\n complete(attemptId: string, leaseId: string, completion: HarnessCompletion): Promise<void>;\n}\n\n/** Trusted inference bridge used to keep model credentials outside agent containers. */\nexport interface HarnessAiConnection {\n infer(request: HarnessInferenceRequest): Promise<HarnessInferenceResponse>;\n}\n\n/** Trusted tool boundary. Implementations must evaluate CaMeL policy before effects. */\nexport interface HarnessToolBroker {\n execute(\n context: { lease: HarnessLease; workspaceDir: string; signal?: AbortSignal },\n request: HarnessToolRequest,\n ): Promise<HarnessToolResponse>;\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;;;ACGO,IAAM,2BAA2B;;;ADSjC,SAAS,aAAa,YAAmC,CAAC,GAAiB;AAChF,SAAO;AAAA,IACL,iBAAiB;AAAA,IACjB,SAAS;AAAA,IACT,YAAY;AAAA,IACZ,WAAW,KAAK,IAAI,IAAI;AAAA,IACxB,MAAM;AAAA,MACJ,QAAQ;AAAA,MACR,WAAW;AAAA,MACX,OAAO;AAAA,MACP,QAAQ;AAAA,MACR,WAAW;AAAA,MACX,SAAS;AAAA,MACT,QAAQ;AAAA,QACN,SAAS;AAAA,QACT,WAAW;AAAA,QACX,gBAAgB;AAAA,QAChB,eAAe;AAAA,MACjB;AAAA,IACF;AAAA,IACA,GAAG;AAAA,EACL;AACF;AAGO,IAAM,0BAAN,MAA6D;AAAA,EACzD,SAAyB,CAAC;AAAA,EAC1B,SAA8B,CAAC;AAAA,EAC/B,cAAmC,CAAC;AAAA,EACpC,oBAA+C,CAAC;AAAA,EACzD,kBAAkB;AAAA,EAClB,WAA2B;AAAA,IACzB,IAAI;AAAA,IACJ,OAAO;AAAA,IACP,UAAU;AAAA,IACV,MAAM;AAAA,IACN,SAAS,CAAC,EAAE,MAAM,QAAQ,MAAM,iCAAiC,CAAC;AAAA,IAClE,YAAY;AAAA,IACZ,OAAO,EAAE,aAAa,GAAG,cAAc,EAAE;AAAA,EAC3C;AAAA,EAEA,eAAe,QAAwB;AAAE,SAAK,OAAO,KAAK,GAAG,MAAM;AAAA,EAAG;AAAA,EAEtE,MAAM,MAAM,YAAoD;AAC9D,UAAM,QAAQ,KAAK,OAAO,UAAU,CAAC,cAAc,WAAW,SAAS,UAAU,KAAK,SAAS,CAAC;AAChG,WAAO,QAAQ,IAAI,OAAO,KAAK,OAAO,OAAO,OAAO,CAAC,EAAE,CAAC;AAAA,EAC1D;AAAA,EACA,MAAM,YAAsE;AAC1E,WAAO,EAAE,iBAAiB,KAAK,iBAAiB,WAAW,KAAK,IAAI,IAAI,IAAO;AAAA,EACjF;AAAA,EACA,MAAM,aAAa,YAAoB,UAAkB,QAA4C;AACnG,SAAK,OAAO,KAAK,GAAG,MAAM;AAAA,EAC5B;AAAA,EACA,MAAM,MAAM,YAAoB,UAAkB,SAAqE;AACrH,SAAK,kBAAkB,KAAK,OAAO;AACnC,WAAO;AAAA,MACL,WAAW,QAAQ;AAAA,MACnB,UAAU,KAAK;AAAA,MACf,SAAS;AAAA,QACP,UAAU,KAAK,SAAS;AAAA,QACxB,OAAO,KAAK,SAAS;AAAA,QACrB,eAAe;AAAA,QACf,aAAa,KAAK,SAAS,MAAM,eAAe;AAAA,QAChD,cAAc,KAAK,SAAS,MAAM,gBAAgB;AAAA,MACpD;AAAA,IACF;AAAA,EACF;AAAA,EACA,MAAM,SAAS,YAAoB,UAAkB,YAA8C;AACjG,SAAK,YAAY,KAAK,UAAU;AAAA,EAClC;AACF;","names":[]}
|
|
1
|
+
{"version":3,"sources":["../src/testing.ts","../src/types.ts"],"sourcesContent":["import type { OracleResponse } from \"@odla-ai/ai\";\nimport {\n HARNESS_PROTOCOL_VERSION,\n type HarnessCompletion,\n type HarnessControlPlane,\n type HarnessEventInput,\n type HarnessInferenceRequest,\n type HarnessInferenceResponse,\n type HarnessLease,\n} from \"./types\";\n\n/** Create a deterministic, valid lease for harness unit and integration tests. */\nexport function fixtureLease(overrides: Partial<HarnessLease> = {}): HarnessLease {\n return {\n protocolVersion: HARNESS_PROTOCOL_VERSION,\n leaseId: \"lease_fixture\",\n generation: 1,\n expiresAt: Date.now() + 60_000,\n task: {\n taskId: \"task_fixture\",\n attemptId: \"attempt_fixture\",\n title: \"Fixture task\",\n prompt: \"Create HARNESS_RESULT.md\",\n workspace: \"fixture\",\n aiRoute: \"coding\",\n policy: {\n network: \"none\",\n timeoutMs: 30_000,\n maxOutputBytes: 1_000_000,\n maxPatchBytes: 1_000_000,\n },\n },\n ...overrides,\n };\n}\n\n/** In-memory control plane that records runner interactions without network access. */\nexport class FakeHarnessControlPlane implements HarnessControlPlane {\n readonly leases: HarnessLease[] = [];\n readonly events: HarnessEventInput[] = [];\n readonly completions: HarnessCompletion[] = [];\n readonly inferenceRequests: HarnessInferenceRequest[] = [];\n cancelRequested = false;\n response: OracleResponse = {\n id: \"oracle_fixture\",\n model: \"fixture-model\",\n provider: \"openai\",\n role: \"assistant\",\n content: [{ type: \"text\", text: \"deterministic fixture response\" }],\n stopReason: \"end_turn\",\n usage: { inputTokens: 5, outputTokens: 4 },\n };\n\n constructor(...leases: HarnessLease[]) { this.leases.push(...leases); }\n\n async lease(workspaces: string[]): Promise<HarnessLease | null> {\n const index = this.leases.findIndex((candidate) => workspaces.includes(candidate.task.workspace));\n return index < 0 ? null : this.leases.splice(index, 1)[0]!;\n }\n async heartbeat(): Promise<{ cancelRequested: boolean; expiresAt: number }> {\n return { cancelRequested: this.cancelRequested, expiresAt: Date.now() + 60_000 };\n }\n async appendEvents(_attemptId: string, _leaseId: string, events: HarnessEventInput[]): Promise<void> {\n this.events.push(...events);\n }\n async infer(_attemptId: string, _leaseId: string, request: HarnessInferenceRequest): Promise<HarnessInferenceResponse> {\n this.inferenceRequests.push(request);\n return {\n requestId: request.requestId,\n response: this.response,\n receipt: {\n provider: this.response.provider,\n model: this.response.model,\n policyVersion: 1,\n inputTokens: this.response.usage.inputTokens ?? 0,\n outputTokens: this.response.usage.outputTokens ?? 0,\n },\n };\n }\n async complete(_attemptId: string, _leaseId: string, completion: HarnessCompletion): Promise<void> {\n this.completions.push(completion);\n }\n}\n","import type { ChatInput, OracleResponse } from \"@odla-ai/ai\";\nimport type { CodeToolPresentation } from \"./code-session-event-types\";\nexport type { CodeToolLocationPreview, CodeToolPresentation } from \"./code-session-event-types\";\n\n/** Current JSONL protocol version exchanged between a runner and an agent container. */\nexport const HARNESS_PROTOCOL_VERSION = 1 as const;\n\n/** Default control-plane route used for model inference requested by coding agents. */\nexport const DEFAULT_AI_ROUTE = \"coding\" as const;\n\n/** Lifecycle state reported for a harness task and its active attempt. */\nexport type HarnessTaskStatus =\n | \"queued\"\n | \"running\"\n | \"cancel_requested\"\n | \"completed\"\n | \"failed\"\n | \"cancelled\";\n\n/** Lifecycle state of one execution attempt for a task. */\nexport type HarnessAttemptStatus = HarnessTaskStatus;\n\n/** Trusted or untrusted participant that emitted a harness event. */\nexport type HarnessActor = \"operator\" | \"runner\" | \"agent\" | \"model\" | \"system\";\n\n/** Resource and isolation limits enforced while an untrusted task executes. */\nexport interface HarnessPolicy {\n network: \"none\";\n timeoutMs: number;\n maxOutputBytes: number;\n maxPatchBytes: number;\n}\n\n/** Immutable task instructions and execution policy delivered with a lease. */\nexport interface HarnessTaskSpec {\n taskId: string;\n attemptId: string;\n title: string;\n prompt: string;\n workspace: string;\n aiRoute: string;\n policy: HarnessPolicy;\n parentAttemptId?: string | null;\n checkpointSeq?: number | null;\n}\n\n/** Time-bound assignment authorizing a runner to execute one task attempt. */\nexport interface HarnessLease {\n protocolVersion: typeof HARNESS_PROTOCOL_VERSION;\n leaseId: string;\n generation: number;\n expiresAt: number;\n task: HarnessTaskSpec;\n}\n\n/** Runner-supplied event before the control plane assigns sequence and task metadata. */\nexport interface HarnessEventInput {\n eventId: string;\n kind: string;\n actor: HarnessActor;\n payload: unknown;\n createdAt: number;\n}\n\n/** Persisted, ordered event associated with a specific task attempt. */\nexport interface HarnessEvent extends HarnessEventInput {\n seq: number;\n taskId: string;\n attemptId: string;\n}\n\n/** List-view metadata for a task and its current attempt. */\nexport interface HarnessTaskSummary {\n taskId: string;\n attemptId: string;\n appId: string;\n env: string;\n title: string;\n prompt: string;\n workspace: string;\n aiRoute: string;\n status: HarnessTaskStatus;\n parentAttemptId: string | null;\n checkpointSeq: number | null;\n runnerId: string | null;\n generation: number;\n createdAt: number;\n updatedAt: number;\n}\n\n/** List-view metadata for one attempt, including retry ancestry and runner ownership. */\nexport interface HarnessAttemptSummary {\n attemptId: string;\n taskId: string;\n parentAttemptId: string | null;\n checkpointSeq: number | null;\n status: HarnessAttemptStatus;\n runnerId: string | null;\n generation: number;\n createdAt: number;\n updatedAt: number;\n}\n\n/** Complete task view including attempts, events, result, and generated patch. */\nexport interface HarnessTaskDetail extends HarnessTaskSummary {\n attempts: HarnessAttemptSummary[];\n events: HarnessEvent[];\n patch: string | null;\n result: unknown;\n}\n\n/** Public control-plane view of a registered harness runner. */\nexport interface HarnessRunnerView {\n runnerId: string;\n appId: string;\n env: string;\n name: string;\n createdAt: number;\n lastSeenAt: number | null;\n revokedAt: number | null;\n}\n\n/** Credential-free normalized model request sent from an agent through the runner. */\nexport interface HarnessInferenceRequest {\n requestId: string;\n /** Exact initial, follow-up, or resume command whose budget this call consumes. */\n interactionId?: string;\n call: ChatInput;\n}\n\n/** Normalized model response plus auditable provider, policy, and token metadata. */\nexport interface HarnessInferenceResponse {\n requestId: string;\n response: OracleResponse;\n receipt: {\n provider: string;\n model: string;\n policyVersion: number;\n inputTokens: number;\n outputTokens: number;\n /**\n * USD charged for this call, priced against the model the control plane\n * actually resolved.\n *\n * ABSENT when the live catalog has no price for that model — never zero.\n * An unpriced call is unknown spend, and reporting it as free is what let\n * a goal's `maxUsd` look enforced while nothing enforced it. The runtime\n * cannot compute this itself: it asks for `brokered` and only the control\n * plane knows which model answered.\n */\n costUsd?: number;\n };\n}\n\n/** Bounded, content-minimized session activity emitted by a Code runtime.\n * Message bodies are projected separately into the app's owner-private\n * odla-db chat. `interactionId` is optional so stored v1 events remain valid. */\nexport type CodeSessionEventData = (\n | { type: \"message\"; actor: \"agent\" | \"system\"; body: string }\n | { type: \"diagnostic\"; level: \"error\"; message: string }\n | { type: \"thinking\"; available: true; durationMs: number }\n | {\n type: \"tool\";\n phase: \"started\";\n tool: HarnessToolName;\n operationId?: string;\n presentation?: CodeToolPresentation;\n }\n | {\n type: \"tool\";\n phase: \"completed\";\n tool: HarnessToolName;\n ok: boolean;\n durationMs: number;\n operationId?: string;\n presentation?: CodeToolPresentation;\n }\n | {\n type: \"usage\"; provider: string; model: string;\n inputTokens: number; outputTokens: number; durationMs: number;\n interactionTokens?: number; interactionMaxTokens?: number;\n /** USD for this call; absent when the model is unpriced, never zero. */\n costUsd?: number;\n /** Cumulative USD for this owner interaction, when every call in it was\n * priced. Absent the moment one was not, so a partial total can never be\n * mistaken for the whole. */\n interactionCostUsd?: number;\n }\n | {\n type: \"status\"; status: \"running\" | \"idle\" | \"failed\" | \"checkpointed\";\n durationMs?: number;\n }\n) & { interactionId?: string };\n\n/** Registry-assigned cursor and timestamp for an owner-visible Code event. */\nexport type CodeSessionEvent = CodeSessionEventData & {\n eventId: string; sequence: number; createdAt: number;\n};\n\n/** Closed set of effects an agent container may request from its trusted broker. */\nexport type HarnessToolName =\n | \"sandbox.read\"\n | \"sandbox.list\"\n | \"sandbox.search\"\n | \"sandbox.overview\"\n | \"sandbox.where_is\"\n | \"sandbox.who_imports\"\n | \"sandbox.who_touches\"\n | \"sandbox.apply_patch\"\n | \"sandbox.run_recipe\";\n/** Correlated, structured tool request emitted by an untrusted agent container. */\nexport interface HarnessToolRequest {\n requestId: string;\n tool: HarnessToolName;\n input: Record<string, unknown>;\n}\n/** Bounded tool result returned to an agent after trusted policy evaluation. */\nexport interface HarnessToolResponse {\n requestId: string;\n ok: boolean;\n content: string;\n details?: Record<string, unknown>;\n}\n\n/** Validated JSONL message emitted by an untrusted agent container. */\nexport type HarnessAgentOutput =\n | {\n protocolVersion: typeof HARNESS_PROTOCOL_VERSION;\n type: \"event\";\n kind: string;\n payload?: unknown;\n }\n | {\n protocolVersion: typeof HARNESS_PROTOCOL_VERSION;\n type: \"inference.request\";\n requestId: string;\n call: ChatInput;\n }\n | ({ protocolVersion: typeof HARNESS_PROTOCOL_VERSION; type: \"tool.request\" } & HarnessToolRequest)\n | {\n protocolVersion: typeof HARNESS_PROTOCOL_VERSION;\n type: \"attempt.complete\";\n status: \"completed\" | \"failed\" | \"cancelled\";\n result?: unknown;\n };\n\n/** JSONL command or inference result written by the trusted runner to an agent. */\nexport type HarnessAgentInput =\n | {\n protocolVersion: typeof HARNESS_PROTOCOL_VERSION;\n type: \"task.start\";\n task: HarnessTaskSpec;\n }\n | {\n protocolVersion: typeof HARNESS_PROTOCOL_VERSION;\n type: \"inference.response\";\n requestId: string;\n response: OracleResponse;\n }\n | ({ protocolVersion: typeof HARNESS_PROTOCOL_VERSION; type: \"tool.response\" } & HarnessToolResponse)\n | {\n protocolVersion: typeof HARNESS_PROTOCOL_VERSION;\n type: \"attempt.cancel\";\n reason: string;\n };\n\n/** Terminal attempt report submitted by a runner to the control plane. */\nexport interface HarnessCompletion {\n status: \"completed\" | \"failed\" | \"cancelled\";\n result?: unknown;\n patch?: string;\n error?: string;\n}\n\n/** Operations a credentialed runner may perform against the harness control plane. */\nexport interface HarnessControlPlane {\n lease(workspaces: string[]): Promise<HarnessLease | null>;\n heartbeat(attemptId: string, leaseId: string): Promise<{ cancelRequested: boolean; expiresAt: number }>;\n appendEvents(attemptId: string, leaseId: string, events: HarnessEventInput[]): Promise<void>;\n infer(attemptId: string, leaseId: string, request: HarnessInferenceRequest): Promise<HarnessInferenceResponse>;\n complete(attemptId: string, leaseId: string, completion: HarnessCompletion): Promise<void>;\n}\n\n/** Trusted inference bridge used to keep model credentials outside agent containers. */\nexport interface HarnessAiConnection {\n infer(request: HarnessInferenceRequest): Promise<HarnessInferenceResponse>;\n}\n\n/** Trusted tool boundary. Implementations must evaluate CaMeL policy before effects. */\nexport interface HarnessToolBroker {\n execute(\n context: { lease: HarnessLease; workspaceDir: string; signal?: AbortSignal },\n request: HarnessToolRequest,\n ): Promise<HarnessToolResponse>;\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;;;ACKO,IAAM,2BAA2B;;;ADOjC,SAAS,aAAa,YAAmC,CAAC,GAAiB;AAChF,SAAO;AAAA,IACL,iBAAiB;AAAA,IACjB,SAAS;AAAA,IACT,YAAY;AAAA,IACZ,WAAW,KAAK,IAAI,IAAI;AAAA,IACxB,MAAM;AAAA,MACJ,QAAQ;AAAA,MACR,WAAW;AAAA,MACX,OAAO;AAAA,MACP,QAAQ;AAAA,MACR,WAAW;AAAA,MACX,SAAS;AAAA,MACT,QAAQ;AAAA,QACN,SAAS;AAAA,QACT,WAAW;AAAA,QACX,gBAAgB;AAAA,QAChB,eAAe;AAAA,MACjB;AAAA,IACF;AAAA,IACA,GAAG;AAAA,EACL;AACF;AAGO,IAAM,0BAAN,MAA6D;AAAA,EACzD,SAAyB,CAAC;AAAA,EAC1B,SAA8B,CAAC;AAAA,EAC/B,cAAmC,CAAC;AAAA,EACpC,oBAA+C,CAAC;AAAA,EACzD,kBAAkB;AAAA,EAClB,WAA2B;AAAA,IACzB,IAAI;AAAA,IACJ,OAAO;AAAA,IACP,UAAU;AAAA,IACV,MAAM;AAAA,IACN,SAAS,CAAC,EAAE,MAAM,QAAQ,MAAM,iCAAiC,CAAC;AAAA,IAClE,YAAY;AAAA,IACZ,OAAO,EAAE,aAAa,GAAG,cAAc,EAAE;AAAA,EAC3C;AAAA,EAEA,eAAe,QAAwB;AAAE,SAAK,OAAO,KAAK,GAAG,MAAM;AAAA,EAAG;AAAA,EAEtE,MAAM,MAAM,YAAoD;AAC9D,UAAM,QAAQ,KAAK,OAAO,UAAU,CAAC,cAAc,WAAW,SAAS,UAAU,KAAK,SAAS,CAAC;AAChG,WAAO,QAAQ,IAAI,OAAO,KAAK,OAAO,OAAO,OAAO,CAAC,EAAE,CAAC;AAAA,EAC1D;AAAA,EACA,MAAM,YAAsE;AAC1E,WAAO,EAAE,iBAAiB,KAAK,iBAAiB,WAAW,KAAK,IAAI,IAAI,IAAO;AAAA,EACjF;AAAA,EACA,MAAM,aAAa,YAAoB,UAAkB,QAA4C;AACnG,SAAK,OAAO,KAAK,GAAG,MAAM;AAAA,EAC5B;AAAA,EACA,MAAM,MAAM,YAAoB,UAAkB,SAAqE;AACrH,SAAK,kBAAkB,KAAK,OAAO;AACnC,WAAO;AAAA,MACL,WAAW,QAAQ;AAAA,MACnB,UAAU,KAAK;AAAA,MACf,SAAS;AAAA,QACP,UAAU,KAAK,SAAS;AAAA,QACxB,OAAO,KAAK,SAAS;AAAA,QACrB,eAAe;AAAA,QACf,aAAa,KAAK,SAAS,MAAM,eAAe;AAAA,QAChD,cAAc,KAAK,SAAS,MAAM,gBAAgB;AAAA,MACpD;AAAA,IACF;AAAA,EACF;AAAA,EACA,MAAM,SAAS,YAAoB,UAAkB,YAA8C;AACjG,SAAK,YAAY,KAAK,UAAU;AAAA,EAClC;AACF;","names":[]}
|
package/dist/testing.d.cts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { OracleResponse } from '@odla-ai/ai';
|
|
2
|
-
import { H as HarnessControlPlane, a as HarnessLease, b as HarnessEventInput, c as HarnessCompletion, d as HarnessInferenceRequest, e as HarnessInferenceResponse } from './types-
|
|
2
|
+
import { H as HarnessControlPlane, a as HarnessLease, b as HarnessEventInput, c as HarnessCompletion, d as HarnessInferenceRequest, e as HarnessInferenceResponse } from './types-BNJikP5h.cjs';
|
|
3
3
|
|
|
4
4
|
/** Create a deterministic, valid lease for harness unit and integration tests. */
|
|
5
5
|
declare function fixtureLease(overrides?: Partial<HarnessLease>): HarnessLease;
|
package/dist/testing.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { OracleResponse } from '@odla-ai/ai';
|
|
2
|
-
import { H as HarnessControlPlane, a as HarnessLease, b as HarnessEventInput, c as HarnessCompletion, d as HarnessInferenceRequest, e as HarnessInferenceResponse } from './types-
|
|
2
|
+
import { H as HarnessControlPlane, a as HarnessLease, b as HarnessEventInput, c as HarnessCompletion, d as HarnessInferenceRequest, e as HarnessInferenceResponse } from './types-BNJikP5h.js';
|
|
3
3
|
|
|
4
4
|
/** Create a deterministic, valid lease for harness unit and integration tests. */
|
|
5
5
|
declare function fixtureLease(overrides?: Partial<HarnessLease>): HarnessLease;
|
package/dist/testing.js
CHANGED
|
@@ -1,5 +1,47 @@
|
|
|
1
1
|
import { OracleResponse, ChatInput } from '@odla-ai/ai';
|
|
2
2
|
|
|
3
|
+
/** One bounded result location that an owner may inspect without loading the
|
|
4
|
+
* complete broker result into the activity stream. */
|
|
5
|
+
interface CodeToolLocationPreview {
|
|
6
|
+
path: string;
|
|
7
|
+
line?: number;
|
|
8
|
+
text?: string;
|
|
9
|
+
}
|
|
10
|
+
/** Deliberately small, tool-specific owner presentation. These projections are
|
|
11
|
+
* built after the trusted broker has bounded the corresponding request/result;
|
|
12
|
+
* they are not arbitrary model narration or complete tool output. */
|
|
13
|
+
type CodeToolPresentation = {
|
|
14
|
+
kind: "query";
|
|
15
|
+
query?: string;
|
|
16
|
+
scope?: string;
|
|
17
|
+
count?: number;
|
|
18
|
+
results?: CodeToolLocationPreview[];
|
|
19
|
+
excerpt?: string;
|
|
20
|
+
} | {
|
|
21
|
+
kind: "read";
|
|
22
|
+
path: string;
|
|
23
|
+
startLine?: number;
|
|
24
|
+
endLine?: number;
|
|
25
|
+
excerpt?: string;
|
|
26
|
+
} | {
|
|
27
|
+
kind: "list";
|
|
28
|
+
scope?: string;
|
|
29
|
+
count?: number;
|
|
30
|
+
paths?: string[];
|
|
31
|
+
} | {
|
|
32
|
+
kind: "patch";
|
|
33
|
+
paths?: string[];
|
|
34
|
+
additions?: number;
|
|
35
|
+
deletions?: number;
|
|
36
|
+
} | {
|
|
37
|
+
kind: "recipe";
|
|
38
|
+
recipeId: string;
|
|
39
|
+
exitCode?: number;
|
|
40
|
+
timedOut?: boolean;
|
|
41
|
+
outputLimitExceeded?: boolean;
|
|
42
|
+
excerpt?: string;
|
|
43
|
+
};
|
|
44
|
+
|
|
3
45
|
/** Current JSONL protocol version exchanged between a runner and an agent container. */
|
|
4
46
|
declare const HARNESS_PROTOCOL_VERSION: 1;
|
|
5
47
|
/** Default control-plane route used for model inference requested by coding agents. */
|
|
@@ -128,9 +170,10 @@ interface HarnessInferenceResponse {
|
|
|
128
170
|
costUsd?: number;
|
|
129
171
|
};
|
|
130
172
|
}
|
|
131
|
-
/** Bounded, content-
|
|
132
|
-
* bodies are projected separately into the app's owner-private
|
|
133
|
-
|
|
173
|
+
/** Bounded, content-minimized session activity emitted by a Code runtime.
|
|
174
|
+
* Message bodies are projected separately into the app's owner-private
|
|
175
|
+
* odla-db chat. `interactionId` is optional so stored v1 events remain valid. */
|
|
176
|
+
type CodeSessionEventData = ({
|
|
134
177
|
type: "message";
|
|
135
178
|
actor: "agent" | "system";
|
|
136
179
|
body: string;
|
|
@@ -146,12 +189,16 @@ type CodeSessionEventData = {
|
|
|
146
189
|
type: "tool";
|
|
147
190
|
phase: "started";
|
|
148
191
|
tool: HarnessToolName;
|
|
192
|
+
operationId?: string;
|
|
193
|
+
presentation?: CodeToolPresentation;
|
|
149
194
|
} | {
|
|
150
195
|
type: "tool";
|
|
151
196
|
phase: "completed";
|
|
152
197
|
tool: HarnessToolName;
|
|
153
198
|
ok: boolean;
|
|
154
199
|
durationMs: number;
|
|
200
|
+
operationId?: string;
|
|
201
|
+
presentation?: CodeToolPresentation;
|
|
155
202
|
} | {
|
|
156
203
|
type: "usage";
|
|
157
204
|
provider: string;
|
|
@@ -159,7 +206,6 @@ type CodeSessionEventData = {
|
|
|
159
206
|
inputTokens: number;
|
|
160
207
|
outputTokens: number;
|
|
161
208
|
durationMs: number;
|
|
162
|
-
interactionId?: string;
|
|
163
209
|
interactionTokens?: number;
|
|
164
210
|
interactionMaxTokens?: number;
|
|
165
211
|
/** USD for this call; absent when the model is unpriced, never zero. */
|
|
@@ -172,6 +218,8 @@ type CodeSessionEventData = {
|
|
|
172
218
|
type: "status";
|
|
173
219
|
status: "running" | "idle" | "failed" | "checkpointed";
|
|
174
220
|
durationMs?: number;
|
|
221
|
+
}) & {
|
|
222
|
+
interactionId?: string;
|
|
175
223
|
};
|
|
176
224
|
/** Registry-assigned cursor and timestamp for an owner-visible Code event. */
|
|
177
225
|
type CodeSessionEvent = CodeSessionEventData & {
|
|
@@ -263,4 +311,4 @@ interface HarnessToolBroker {
|
|
|
263
311
|
}, request: HarnessToolRequest): Promise<HarnessToolResponse>;
|
|
264
312
|
}
|
|
265
313
|
|
|
266
|
-
export { type CodeSessionEvent as C, DEFAULT_AI_ROUTE as D, type HarnessControlPlane as H, type HarnessLease as a, type HarnessEventInput as b, type HarnessCompletion as c, type HarnessInferenceRequest as d, type HarnessInferenceResponse as e, type HarnessAgentInput as f, type HarnessAgentOutput as g, type HarnessAiConnection as h, type CodeSessionEventData as i,
|
|
314
|
+
export { type CodeSessionEvent as C, DEFAULT_AI_ROUTE as D, type HarnessControlPlane as H, type HarnessLease as a, type HarnessEventInput as b, type HarnessCompletion as c, type HarnessInferenceRequest as d, type HarnessInferenceResponse as e, type HarnessAgentInput as f, type HarnessAgentOutput as g, type HarnessAiConnection as h, type CodeSessionEventData as i, type CodeToolLocationPreview as j, type CodeToolPresentation as k, HARNESS_PROTOCOL_VERSION as l, type HarnessActor as m, type HarnessAttemptStatus as n, type HarnessAttemptSummary as o, type HarnessEvent as p, type HarnessPolicy as q, type HarnessRunnerView as r, type HarnessTaskDetail as s, type HarnessTaskSpec as t, type HarnessTaskStatus as u, type HarnessTaskSummary as v, type HarnessToolBroker as w, type HarnessToolName as x, type HarnessToolRequest as y, type HarnessToolResponse as z };
|
|
@@ -1,5 +1,47 @@
|
|
|
1
1
|
import { OracleResponse, ChatInput } from '@odla-ai/ai';
|
|
2
2
|
|
|
3
|
+
/** One bounded result location that an owner may inspect without loading the
|
|
4
|
+
* complete broker result into the activity stream. */
|
|
5
|
+
interface CodeToolLocationPreview {
|
|
6
|
+
path: string;
|
|
7
|
+
line?: number;
|
|
8
|
+
text?: string;
|
|
9
|
+
}
|
|
10
|
+
/** Deliberately small, tool-specific owner presentation. These projections are
|
|
11
|
+
* built after the trusted broker has bounded the corresponding request/result;
|
|
12
|
+
* they are not arbitrary model narration or complete tool output. */
|
|
13
|
+
type CodeToolPresentation = {
|
|
14
|
+
kind: "query";
|
|
15
|
+
query?: string;
|
|
16
|
+
scope?: string;
|
|
17
|
+
count?: number;
|
|
18
|
+
results?: CodeToolLocationPreview[];
|
|
19
|
+
excerpt?: string;
|
|
20
|
+
} | {
|
|
21
|
+
kind: "read";
|
|
22
|
+
path: string;
|
|
23
|
+
startLine?: number;
|
|
24
|
+
endLine?: number;
|
|
25
|
+
excerpt?: string;
|
|
26
|
+
} | {
|
|
27
|
+
kind: "list";
|
|
28
|
+
scope?: string;
|
|
29
|
+
count?: number;
|
|
30
|
+
paths?: string[];
|
|
31
|
+
} | {
|
|
32
|
+
kind: "patch";
|
|
33
|
+
paths?: string[];
|
|
34
|
+
additions?: number;
|
|
35
|
+
deletions?: number;
|
|
36
|
+
} | {
|
|
37
|
+
kind: "recipe";
|
|
38
|
+
recipeId: string;
|
|
39
|
+
exitCode?: number;
|
|
40
|
+
timedOut?: boolean;
|
|
41
|
+
outputLimitExceeded?: boolean;
|
|
42
|
+
excerpt?: string;
|
|
43
|
+
};
|
|
44
|
+
|
|
3
45
|
/** Current JSONL protocol version exchanged between a runner and an agent container. */
|
|
4
46
|
declare const HARNESS_PROTOCOL_VERSION: 1;
|
|
5
47
|
/** Default control-plane route used for model inference requested by coding agents. */
|
|
@@ -128,9 +170,10 @@ interface HarnessInferenceResponse {
|
|
|
128
170
|
costUsd?: number;
|
|
129
171
|
};
|
|
130
172
|
}
|
|
131
|
-
/** Bounded, content-
|
|
132
|
-
* bodies are projected separately into the app's owner-private
|
|
133
|
-
|
|
173
|
+
/** Bounded, content-minimized session activity emitted by a Code runtime.
|
|
174
|
+
* Message bodies are projected separately into the app's owner-private
|
|
175
|
+
* odla-db chat. `interactionId` is optional so stored v1 events remain valid. */
|
|
176
|
+
type CodeSessionEventData = ({
|
|
134
177
|
type: "message";
|
|
135
178
|
actor: "agent" | "system";
|
|
136
179
|
body: string;
|
|
@@ -146,12 +189,16 @@ type CodeSessionEventData = {
|
|
|
146
189
|
type: "tool";
|
|
147
190
|
phase: "started";
|
|
148
191
|
tool: HarnessToolName;
|
|
192
|
+
operationId?: string;
|
|
193
|
+
presentation?: CodeToolPresentation;
|
|
149
194
|
} | {
|
|
150
195
|
type: "tool";
|
|
151
196
|
phase: "completed";
|
|
152
197
|
tool: HarnessToolName;
|
|
153
198
|
ok: boolean;
|
|
154
199
|
durationMs: number;
|
|
200
|
+
operationId?: string;
|
|
201
|
+
presentation?: CodeToolPresentation;
|
|
155
202
|
} | {
|
|
156
203
|
type: "usage";
|
|
157
204
|
provider: string;
|
|
@@ -159,7 +206,6 @@ type CodeSessionEventData = {
|
|
|
159
206
|
inputTokens: number;
|
|
160
207
|
outputTokens: number;
|
|
161
208
|
durationMs: number;
|
|
162
|
-
interactionId?: string;
|
|
163
209
|
interactionTokens?: number;
|
|
164
210
|
interactionMaxTokens?: number;
|
|
165
211
|
/** USD for this call; absent when the model is unpriced, never zero. */
|
|
@@ -172,6 +218,8 @@ type CodeSessionEventData = {
|
|
|
172
218
|
type: "status";
|
|
173
219
|
status: "running" | "idle" | "failed" | "checkpointed";
|
|
174
220
|
durationMs?: number;
|
|
221
|
+
}) & {
|
|
222
|
+
interactionId?: string;
|
|
175
223
|
};
|
|
176
224
|
/** Registry-assigned cursor and timestamp for an owner-visible Code event. */
|
|
177
225
|
type CodeSessionEvent = CodeSessionEventData & {
|
|
@@ -263,4 +311,4 @@ interface HarnessToolBroker {
|
|
|
263
311
|
}, request: HarnessToolRequest): Promise<HarnessToolResponse>;
|
|
264
312
|
}
|
|
265
313
|
|
|
266
|
-
export { type CodeSessionEvent as C, DEFAULT_AI_ROUTE as D, type HarnessControlPlane as H, type HarnessLease as a, type HarnessEventInput as b, type HarnessCompletion as c, type HarnessInferenceRequest as d, type HarnessInferenceResponse as e, type HarnessAgentInput as f, type HarnessAgentOutput as g, type HarnessAiConnection as h, type CodeSessionEventData as i,
|
|
314
|
+
export { type CodeSessionEvent as C, DEFAULT_AI_ROUTE as D, type HarnessControlPlane as H, type HarnessLease as a, type HarnessEventInput as b, type HarnessCompletion as c, type HarnessInferenceRequest as d, type HarnessInferenceResponse as e, type HarnessAgentInput as f, type HarnessAgentOutput as g, type HarnessAiConnection as h, type CodeSessionEventData as i, type CodeToolLocationPreview as j, type CodeToolPresentation as k, HARNESS_PROTOCOL_VERSION as l, type HarnessActor as m, type HarnessAttemptStatus as n, type HarnessAttemptSummary as o, type HarnessEvent as p, type HarnessPolicy as q, type HarnessRunnerView as r, type HarnessTaskDetail as s, type HarnessTaskSpec as t, type HarnessTaskStatus as u, type HarnessTaskSummary as v, type HarnessToolBroker as w, type HarnessToolName as x, type HarnessToolRequest as y, type HarnessToolResponse as z };
|
package/package.json
CHANGED
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"sources":["../src/types.ts"],"sourcesContent":["import type { ChatInput, OracleResponse } from \"@odla-ai/ai\";\n\n/** Current JSONL protocol version exchanged between a runner and an agent container. */\nexport const HARNESS_PROTOCOL_VERSION = 1 as const;\n\n/** Default control-plane route used for model inference requested by coding agents. */\nexport const DEFAULT_AI_ROUTE = \"coding\" as const;\n\n/** Lifecycle state reported for a harness task and its active attempt. */\nexport type HarnessTaskStatus =\n | \"queued\"\n | \"running\"\n | \"cancel_requested\"\n | \"completed\"\n | \"failed\"\n | \"cancelled\";\n\n/** Lifecycle state of one execution attempt for a task. */\nexport type HarnessAttemptStatus = HarnessTaskStatus;\n\n/** Trusted or untrusted participant that emitted a harness event. */\nexport type HarnessActor = \"operator\" | \"runner\" | \"agent\" | \"model\" | \"system\";\n\n/** Resource and isolation limits enforced while an untrusted task executes. */\nexport interface HarnessPolicy {\n network: \"none\";\n timeoutMs: number;\n maxOutputBytes: number;\n maxPatchBytes: number;\n}\n\n/** Immutable task instructions and execution policy delivered with a lease. */\nexport interface HarnessTaskSpec {\n taskId: string;\n attemptId: string;\n title: string;\n prompt: string;\n workspace: string;\n aiRoute: string;\n policy: HarnessPolicy;\n parentAttemptId?: string | null;\n checkpointSeq?: number | null;\n}\n\n/** Time-bound assignment authorizing a runner to execute one task attempt. */\nexport interface HarnessLease {\n protocolVersion: typeof HARNESS_PROTOCOL_VERSION;\n leaseId: string;\n generation: number;\n expiresAt: number;\n task: HarnessTaskSpec;\n}\n\n/** Runner-supplied event before the control plane assigns sequence and task metadata. */\nexport interface HarnessEventInput {\n eventId: string;\n kind: string;\n actor: HarnessActor;\n payload: unknown;\n createdAt: number;\n}\n\n/** Persisted, ordered event associated with a specific task attempt. */\nexport interface HarnessEvent extends HarnessEventInput {\n seq: number;\n taskId: string;\n attemptId: string;\n}\n\n/** List-view metadata for a task and its current attempt. */\nexport interface HarnessTaskSummary {\n taskId: string;\n attemptId: string;\n appId: string;\n env: string;\n title: string;\n prompt: string;\n workspace: string;\n aiRoute: string;\n status: HarnessTaskStatus;\n parentAttemptId: string | null;\n checkpointSeq: number | null;\n runnerId: string | null;\n generation: number;\n createdAt: number;\n updatedAt: number;\n}\n\n/** List-view metadata for one attempt, including retry ancestry and runner ownership. */\nexport interface HarnessAttemptSummary {\n attemptId: string;\n taskId: string;\n parentAttemptId: string | null;\n checkpointSeq: number | null;\n status: HarnessAttemptStatus;\n runnerId: string | null;\n generation: number;\n createdAt: number;\n updatedAt: number;\n}\n\n/** Complete task view including attempts, events, result, and generated patch. */\nexport interface HarnessTaskDetail extends HarnessTaskSummary {\n attempts: HarnessAttemptSummary[];\n events: HarnessEvent[];\n patch: string | null;\n result: unknown;\n}\n\n/** Public control-plane view of a registered harness runner. */\nexport interface HarnessRunnerView {\n runnerId: string;\n appId: string;\n env: string;\n name: string;\n createdAt: number;\n lastSeenAt: number | null;\n revokedAt: number | null;\n}\n\n/** Credential-free normalized model request sent from an agent through the runner. */\nexport interface HarnessInferenceRequest {\n requestId: string;\n /** Exact initial, follow-up, or resume command whose budget this call consumes. */\n interactionId?: string;\n call: ChatInput;\n}\n\n/** Normalized model response plus auditable provider, policy, and token metadata. */\nexport interface HarnessInferenceResponse {\n requestId: string;\n response: OracleResponse;\n receipt: {\n provider: string;\n model: string;\n policyVersion: number;\n inputTokens: number;\n outputTokens: number;\n /**\n * USD charged for this call, priced against the model the control plane\n * actually resolved.\n *\n * ABSENT when the live catalog has no price for that model — never zero.\n * An unpriced call is unknown spend, and reporting it as free is what let\n * a goal's `maxUsd` look enforced while nothing enforced it. The runtime\n * cannot compute this itself: it asks for `brokered` and only the control\n * plane knows which model answered.\n */\n costUsd?: number;\n };\n}\n\n/** Bounded, content-free session activity emitted by a Code runtime. Message\n * bodies are projected separately into the app's owner-private odla-db chat. */\nexport type CodeSessionEventData =\n | { type: \"message\"; actor: \"agent\" | \"system\"; body: string }\n | { type: \"diagnostic\"; level: \"error\"; message: string }\n | { type: \"thinking\"; available: true; durationMs: number }\n | { type: \"tool\"; phase: \"started\"; tool: HarnessToolName }\n | { type: \"tool\"; phase: \"completed\"; tool: HarnessToolName; ok: boolean; durationMs: number }\n | {\n type: \"usage\"; provider: string; model: string;\n inputTokens: number; outputTokens: number; durationMs: number;\n interactionId?: string; interactionTokens?: number; interactionMaxTokens?: number;\n /** USD for this call; absent when the model is unpriced, never zero. */\n costUsd?: number;\n /** Cumulative USD for this owner interaction, when every call in it was\n * priced. Absent the moment one was not, so a partial total can never be\n * mistaken for the whole. */\n interactionCostUsd?: number;\n }\n | {\n type: \"status\"; status: \"running\" | \"idle\" | \"failed\" | \"checkpointed\";\n durationMs?: number;\n };\n\n/** Registry-assigned cursor and timestamp for an owner-visible Code event. */\nexport type CodeSessionEvent = CodeSessionEventData & {\n eventId: string; sequence: number; createdAt: number;\n};\n\n/** Closed set of effects an agent container may request from its trusted broker. */\nexport type HarnessToolName =\n | \"sandbox.read\"\n | \"sandbox.list\"\n | \"sandbox.search\"\n | \"sandbox.overview\"\n | \"sandbox.where_is\"\n | \"sandbox.who_imports\"\n | \"sandbox.who_touches\"\n | \"sandbox.apply_patch\"\n | \"sandbox.run_recipe\";\n/** Correlated, structured tool request emitted by an untrusted agent container. */\nexport interface HarnessToolRequest {\n requestId: string;\n tool: HarnessToolName;\n input: Record<string, unknown>;\n}\n/** Bounded tool result returned to an agent after trusted policy evaluation. */\nexport interface HarnessToolResponse {\n requestId: string;\n ok: boolean;\n content: string;\n details?: Record<string, unknown>;\n}\n\n/** Validated JSONL message emitted by an untrusted agent container. */\nexport type HarnessAgentOutput =\n | {\n protocolVersion: typeof HARNESS_PROTOCOL_VERSION;\n type: \"event\";\n kind: string;\n payload?: unknown;\n }\n | {\n protocolVersion: typeof HARNESS_PROTOCOL_VERSION;\n type: \"inference.request\";\n requestId: string;\n call: ChatInput;\n }\n | ({ protocolVersion: typeof HARNESS_PROTOCOL_VERSION; type: \"tool.request\" } & HarnessToolRequest)\n | {\n protocolVersion: typeof HARNESS_PROTOCOL_VERSION;\n type: \"attempt.complete\";\n status: \"completed\" | \"failed\" | \"cancelled\";\n result?: unknown;\n };\n\n/** JSONL command or inference result written by the trusted runner to an agent. */\nexport type HarnessAgentInput =\n | {\n protocolVersion: typeof HARNESS_PROTOCOL_VERSION;\n type: \"task.start\";\n task: HarnessTaskSpec;\n }\n | {\n protocolVersion: typeof HARNESS_PROTOCOL_VERSION;\n type: \"inference.response\";\n requestId: string;\n response: OracleResponse;\n }\n | ({ protocolVersion: typeof HARNESS_PROTOCOL_VERSION; type: \"tool.response\" } & HarnessToolResponse)\n | {\n protocolVersion: typeof HARNESS_PROTOCOL_VERSION;\n type: \"attempt.cancel\";\n reason: string;\n };\n\n/** Terminal attempt report submitted by a runner to the control plane. */\nexport interface HarnessCompletion {\n status: \"completed\" | \"failed\" | \"cancelled\";\n result?: unknown;\n patch?: string;\n error?: string;\n}\n\n/** Operations a credentialed runner may perform against the harness control plane. */\nexport interface HarnessControlPlane {\n lease(workspaces: string[]): Promise<HarnessLease | null>;\n heartbeat(attemptId: string, leaseId: string): Promise<{ cancelRequested: boolean; expiresAt: number }>;\n appendEvents(attemptId: string, leaseId: string, events: HarnessEventInput[]): Promise<void>;\n infer(attemptId: string, leaseId: string, request: HarnessInferenceRequest): Promise<HarnessInferenceResponse>;\n complete(attemptId: string, leaseId: string, completion: HarnessCompletion): Promise<void>;\n}\n\n/** Trusted inference bridge used to keep model credentials outside agent containers. */\nexport interface HarnessAiConnection {\n infer(request: HarnessInferenceRequest): Promise<HarnessInferenceResponse>;\n}\n\n/** Trusted tool boundary. Implementations must evaluate CaMeL policy before effects. */\nexport interface HarnessToolBroker {\n execute(\n context: { lease: HarnessLease; workspaceDir: string; signal?: AbortSignal },\n request: HarnessToolRequest,\n ): Promise<HarnessToolResponse>;\n}\n"],"mappings":";AAGO,IAAM,2BAA2B;AAGjC,IAAM,mBAAmB;","names":[]}
|