@odla-ai/harness 0.4.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  import {
2
2
  HARNESS_PROTOCOL_VERSION
3
- } from "./chunk-3QP4VDQS.js";
3
+ } from "./chunk-LNQNFGQC.js";
4
4
 
5
5
  // src/protocol.ts
6
6
  var CONTROL = /[\u0000-\u0008\u000b\u000c\u000e-\u001f\u007f]/;
@@ -90,4 +90,4 @@ export {
90
90
  encodeAgentInput,
91
91
  makeHarnessEvent
92
92
  };
93
- //# sourceMappingURL=chunk-C5VQI2IF.js.map
93
+ //# sourceMappingURL=chunk-FVOMJKWK.js.map
@@ -1,10 +1,10 @@
1
1
  import {
2
2
  encodeAgentInput,
3
3
  parseAgentOutput
4
- } from "./chunk-C5VQI2IF.js";
4
+ } from "./chunk-FVOMJKWK.js";
5
5
  import {
6
6
  HARNESS_PROTOCOL_VERSION
7
- } from "./chunk-3QP4VDQS.js";
7
+ } from "./chunk-LNQNFGQC.js";
8
8
 
9
9
  // src/container.ts
10
10
  import { execFile, spawn } from "child_process";
@@ -548,4 +548,4 @@ export {
548
548
  stageWorkspacePair,
549
549
  safeWorkspaceLabel
550
550
  };
551
- //# sourceMappingURL=chunk-GKDKIU4P.js.map
551
+ //# sourceMappingURL=chunk-K76I2TCQ.js.map
@@ -6,4 +6,4 @@ export {
6
6
  HARNESS_PROTOCOL_VERSION,
7
7
  DEFAULT_AI_ROUTE
8
8
  };
9
- //# sourceMappingURL=chunk-3QP4VDQS.js.map
9
+ //# sourceMappingURL=chunk-LNQNFGQC.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"sources":["../src/types.ts"],"sourcesContent":["import type { ChatInput, OracleResponse } from \"@odla-ai/ai\";\n\n/** Current JSONL protocol version exchanged between a runner and an agent container. */\nexport const HARNESS_PROTOCOL_VERSION = 1 as const;\n\n/** Default control-plane route used for model inference requested by coding agents. */\nexport const DEFAULT_AI_ROUTE = \"coding\" as const;\n\n/** Lifecycle state reported for a harness task and its active attempt. */\nexport type HarnessTaskStatus =\n | \"queued\"\n | \"running\"\n | \"cancel_requested\"\n | \"completed\"\n | \"failed\"\n | \"cancelled\";\n\n/** Lifecycle state of one execution attempt for a task. */\nexport type HarnessAttemptStatus = HarnessTaskStatus;\n\n/** Trusted or untrusted participant that emitted a harness event. */\nexport type HarnessActor = \"operator\" | \"runner\" | \"agent\" | \"model\" | \"system\";\n\n/** Resource and isolation limits enforced while an untrusted task executes. */\nexport interface HarnessPolicy {\n network: \"none\";\n timeoutMs: number;\n maxOutputBytes: number;\n maxPatchBytes: number;\n}\n\n/** Immutable task instructions and execution policy delivered with a lease. */\nexport interface HarnessTaskSpec {\n taskId: string;\n attemptId: string;\n title: string;\n prompt: string;\n workspace: string;\n aiRoute: string;\n policy: HarnessPolicy;\n parentAttemptId?: string | null;\n checkpointSeq?: number | null;\n}\n\n/** Time-bound assignment authorizing a runner to execute one task attempt. */\nexport interface HarnessLease {\n protocolVersion: typeof HARNESS_PROTOCOL_VERSION;\n leaseId: string;\n generation: number;\n expiresAt: number;\n task: HarnessTaskSpec;\n}\n\n/** Runner-supplied event before the control plane assigns sequence and task metadata. */\nexport interface HarnessEventInput {\n eventId: string;\n kind: string;\n actor: HarnessActor;\n payload: unknown;\n createdAt: number;\n}\n\n/** Persisted, ordered event associated with a specific task attempt. */\nexport interface HarnessEvent extends HarnessEventInput {\n seq: number;\n taskId: string;\n attemptId: string;\n}\n\n/** List-view metadata for a task and its current attempt. */\nexport interface HarnessTaskSummary {\n taskId: string;\n attemptId: string;\n appId: string;\n env: string;\n title: string;\n prompt: string;\n workspace: string;\n aiRoute: string;\n status: HarnessTaskStatus;\n parentAttemptId: string | null;\n checkpointSeq: number | null;\n runnerId: string | null;\n generation: number;\n createdAt: number;\n updatedAt: number;\n}\n\n/** List-view metadata for one attempt, including retry ancestry and runner ownership. */\nexport interface HarnessAttemptSummary {\n attemptId: string;\n taskId: string;\n parentAttemptId: string | null;\n checkpointSeq: number | null;\n status: HarnessAttemptStatus;\n runnerId: string | null;\n generation: number;\n createdAt: number;\n updatedAt: number;\n}\n\n/** Complete task view including attempts, events, result, and generated patch. */\nexport interface HarnessTaskDetail extends HarnessTaskSummary {\n attempts: HarnessAttemptSummary[];\n events: HarnessEvent[];\n patch: string | null;\n result: unknown;\n}\n\n/** Public control-plane view of a registered harness runner. */\nexport interface HarnessRunnerView {\n runnerId: string;\n appId: string;\n env: string;\n name: string;\n createdAt: number;\n lastSeenAt: number | null;\n revokedAt: number | null;\n}\n\n/** Credential-free normalized model request sent from an agent through the runner. */\nexport interface HarnessInferenceRequest {\n requestId: string;\n /** Exact initial, follow-up, or resume command whose budget this call consumes. */\n interactionId?: string;\n call: ChatInput;\n}\n\n/** Normalized model response plus auditable provider, policy, and token metadata. */\nexport interface HarnessInferenceResponse {\n requestId: string;\n response: OracleResponse;\n receipt: {\n provider: string;\n model: string;\n policyVersion: number;\n inputTokens: number;\n outputTokens: number;\n /**\n * USD charged for this call, priced against the model the control plane\n * actually resolved.\n *\n * ABSENT when the live catalog has no price for that model — never zero.\n * An unpriced call is unknown spend, and reporting it as free is what let\n * a goal's `maxUsd` look enforced while nothing enforced it. The runtime\n * cannot compute this itself: it asks for `brokered` and only the control\n * plane knows which model answered.\n */\n costUsd?: number;\n };\n}\n\n/** Bounded, content-free session activity emitted by a Code runtime. Message\n * bodies are projected separately into the app's owner-private odla-db chat. */\nexport type CodeSessionEventData =\n | { type: \"message\"; actor: \"agent\" | \"system\"; body: string }\n | { type: \"diagnostic\"; level: \"error\"; message: string }\n | { type: \"thinking\"; available: true; durationMs: number }\n | { type: \"tool\"; phase: \"started\"; tool: HarnessToolName }\n | { type: \"tool\"; phase: \"completed\"; tool: HarnessToolName; ok: boolean; durationMs: number }\n | {\n type: \"usage\"; provider: string; model: string;\n inputTokens: number; outputTokens: number; durationMs: number;\n interactionId?: string; interactionTokens?: number; interactionMaxTokens?: number;\n /** USD for this call; absent when the model is unpriced, never zero. */\n costUsd?: number;\n /** Cumulative USD for this owner interaction, when every call in it was\n * priced. Absent the moment one was not, so a partial total can never be\n * mistaken for the whole. */\n interactionCostUsd?: number;\n }\n | {\n type: \"status\"; status: \"running\" | \"idle\" | \"failed\" | \"checkpointed\";\n durationMs?: number;\n };\n\n/** Registry-assigned cursor and timestamp for an owner-visible Code event. */\nexport type CodeSessionEvent = CodeSessionEventData & {\n eventId: string; sequence: number; createdAt: number;\n};\n\n/** Closed set of effects an agent container may request from its trusted broker. */\nexport type HarnessToolName =\n | \"sandbox.read\"\n | \"sandbox.list\"\n | \"sandbox.search\"\n | \"sandbox.overview\"\n | \"sandbox.where_is\"\n | \"sandbox.who_imports\"\n | \"sandbox.who_touches\"\n | \"sandbox.apply_patch\"\n | \"sandbox.run_recipe\";\n/** Correlated, structured tool request emitted by an untrusted agent container. */\nexport interface HarnessToolRequest {\n requestId: string;\n tool: HarnessToolName;\n input: Record<string, unknown>;\n}\n/** Bounded tool result returned to an agent after trusted policy evaluation. */\nexport interface HarnessToolResponse {\n requestId: string;\n ok: boolean;\n content: string;\n details?: Record<string, unknown>;\n}\n\n/** Validated JSONL message emitted by an untrusted agent container. */\nexport type HarnessAgentOutput =\n | {\n protocolVersion: typeof HARNESS_PROTOCOL_VERSION;\n type: \"event\";\n kind: string;\n payload?: unknown;\n }\n | {\n protocolVersion: typeof HARNESS_PROTOCOL_VERSION;\n type: \"inference.request\";\n requestId: string;\n call: ChatInput;\n }\n | ({ protocolVersion: typeof HARNESS_PROTOCOL_VERSION; type: \"tool.request\" } & HarnessToolRequest)\n | {\n protocolVersion: typeof HARNESS_PROTOCOL_VERSION;\n type: \"attempt.complete\";\n status: \"completed\" | \"failed\" | \"cancelled\";\n result?: unknown;\n };\n\n/** JSONL command or inference result written by the trusted runner to an agent. */\nexport type HarnessAgentInput =\n | {\n protocolVersion: typeof HARNESS_PROTOCOL_VERSION;\n type: \"task.start\";\n task: HarnessTaskSpec;\n }\n | {\n protocolVersion: typeof HARNESS_PROTOCOL_VERSION;\n type: \"inference.response\";\n requestId: string;\n response: OracleResponse;\n }\n | ({ protocolVersion: typeof HARNESS_PROTOCOL_VERSION; type: \"tool.response\" } & HarnessToolResponse)\n | {\n protocolVersion: typeof HARNESS_PROTOCOL_VERSION;\n type: \"attempt.cancel\";\n reason: string;\n };\n\n/** Terminal attempt report submitted by a runner to the control plane. */\nexport interface HarnessCompletion {\n status: \"completed\" | \"failed\" | \"cancelled\";\n result?: unknown;\n patch?: string;\n error?: string;\n}\n\n/** Operations a credentialed runner may perform against the harness control plane. */\nexport interface HarnessControlPlane {\n lease(workspaces: string[]): Promise<HarnessLease | null>;\n heartbeat(attemptId: string, leaseId: string): Promise<{ cancelRequested: boolean; expiresAt: number }>;\n appendEvents(attemptId: string, leaseId: string, events: HarnessEventInput[]): Promise<void>;\n infer(attemptId: string, leaseId: string, request: HarnessInferenceRequest): Promise<HarnessInferenceResponse>;\n complete(attemptId: string, leaseId: string, completion: HarnessCompletion): Promise<void>;\n}\n\n/** Trusted inference bridge used to keep model credentials outside agent containers. */\nexport interface HarnessAiConnection {\n infer(request: HarnessInferenceRequest): Promise<HarnessInferenceResponse>;\n}\n\n/** Trusted tool boundary. Implementations must evaluate CaMeL policy before effects. */\nexport interface HarnessToolBroker {\n execute(\n context: { lease: HarnessLease; workspaceDir: string; signal?: AbortSignal },\n request: HarnessToolRequest,\n ): Promise<HarnessToolResponse>;\n}\n"],"mappings":";AAGO,IAAM,2BAA2B;AAGjC,IAAM,mBAAmB;","names":[]}
@@ -4,10 +4,10 @@ import {
4
4
  stageWorkspace,
5
5
  stageWorkspacePair,
6
6
  verifyContainerEngineBoundary
7
- } from "./chunk-GKDKIU4P.js";
7
+ } from "./chunk-K76I2TCQ.js";
8
8
  import {
9
9
  HARNESS_PROTOCOL_VERSION
10
- } from "./chunk-3QP4VDQS.js";
10
+ } from "./chunk-LNQNFGQC.js";
11
11
 
12
12
  // src/workspace-digest.ts
13
13
  import { createHash } from "crypto";
@@ -1372,6 +1372,9 @@ async function handleCodeRuntimeInference(input) {
1372
1372
  call: request.call
1373
1373
  });
1374
1374
  state.tokens += response2.receipt.inputTokens + response2.receipt.outputTokens;
1375
+ const { costUsd } = response2.receipt;
1376
+ if (costUsd === void 0) state.costKnown = false;
1377
+ else state.costUsd += costUsd;
1375
1378
  await input.event({
1376
1379
  type: "usage",
1377
1380
  provider: response2.receipt.provider,
@@ -1381,7 +1384,9 @@ async function handleCodeRuntimeInference(input) {
1381
1384
  durationMs: Date.now() - startedAt,
1382
1385
  interactionId: command.commandId,
1383
1386
  interactionTokens: state.tokens,
1384
- interactionMaxTokens: metadata.maxTokensPerInteraction
1387
+ interactionMaxTokens: metadata.maxTokensPerInteraction,
1388
+ ...costUsd === void 0 ? {} : { costUsd },
1389
+ ...state.costKnown ? { interactionCostUsd: state.costUsd } : {}
1385
1390
  }).catch(() => void 0);
1386
1391
  return {
1387
1392
  protocolVersion: HARNESS_PROTOCOL_VERSION,
@@ -2297,10 +2302,16 @@ async function startGoalPursuit(input) {
2297
2302
  attempt: async ({ prompt }) => {
2298
2303
  const result = await input.attempt(prompt);
2299
2304
  return {
2300
- // The runtime charges tokens through the control plane's own
2301
- // per-interaction reservation, so the goal budget bounds ATTEMPTS here
2302
- // and the token ceiling is enforced where the credential lives.
2303
- tokens: 0,
2305
+ // What the attempt actually spent, so the runner's token_budget and
2306
+ // cost_budget checks can be reached. This used to be a hardcoded 0 with
2307
+ // no cost at all, which made maxTokens and maxUsd unreachable while
2308
+ // callers reasonably read them as hard ceilings.
2309
+ //
2310
+ // costUsd is omitted rather than zeroed when any call in the attempt
2311
+ // was unpriced: the runner only enforces a cost budget while the cost
2312
+ // is known, and a zero would make it enforce against a lie.
2313
+ tokens: result.tokens ?? 0,
2314
+ ...result.costUsd === void 0 ? {} : { costUsd: result.costUsd },
2304
2315
  ...result.status === "failed" ? { error: result.error ?? "attempt failed" } : {}
2305
2316
  };
2306
2317
  },
@@ -2519,7 +2530,7 @@ var CodePiRuntimeEngine = class {
2519
2530
  recipeAuthorization: this.options.recipeAuthorization
2520
2531
  }, lease, metadata.role));
2521
2532
  const startedAt = Date.now();
2522
- const interaction = { tokens: 0, noticeEmitted: false };
2533
+ const interaction = { tokens: 0, noticeEmitted: false, costUsd: 0, costKnown: true };
2523
2534
  const inference = createCodeRuntimeInference({
2524
2535
  command,
2525
2536
  metadata,
@@ -2557,7 +2568,11 @@ var CodePiRuntimeEngine = class {
2557
2568
  await this.#diagnostic(command, active, detail);
2558
2569
  await this.#failure(command, active, detail);
2559
2570
  }
2560
- return result;
2571
+ return {
2572
+ ...result,
2573
+ tokens: interaction.tokens,
2574
+ ...interaction.costKnown ? { costUsd: interaction.costUsd } : {}
2575
+ };
2561
2576
  }
2562
2577
  /** Report every brokered effect as it starts and finishes. */
2563
2578
  #observed(command, active, broker) {
@@ -2651,4 +2666,4 @@ export {
2651
2666
  runGoal,
2652
2667
  CodePiRuntimeEngine
2653
2668
  };
2654
- //# sourceMappingURL=chunk-ANNX7VGK.js.map
2669
+ //# sourceMappingURL=chunk-UVGZHNLW.js.map