@odla-ai/harness 0.8.1 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/node.d.cts CHANGED
@@ -1,6 +1,6 @@
1
1
  import { t as HarnessTaskSpec, g as HarnessAgentOutput, f as HarnessAgentInput, H as HarnessControlPlane, w as HarnessToolBroker, a as HarnessLease, d as HarnessInferenceRequest, e as HarnessInferenceResponse, i as CodeSessionEventData, x as HarnessToolName, y as HarnessToolRequest } from './types-BNJikP5h.cjs';
2
2
  import { CodePortableCheckpoint, CodeSourceFile, CodeVerificationReceipt, CodeCheckpointState } from '@odla-ai/camel/code';
3
- import { Skill, Inference, AgentRunBudget, AgentRun, CompactionPolicy } from '@odla-ai/ai';
3
+ import { TaintLabel, ToolOutput, Skill, Inference, AgentRunBudget, AgentRun, CompactionPolicy } from '@odla-ai/ai';
4
4
  import { PolicyOutcome } from '@odla-ai/camel/policy';
5
5
  import { Graph, PartitionVerdict } from '@odla-ai/graph';
6
6
 
@@ -142,6 +142,7 @@ declare class CodeRuntimeControlError extends Error {
142
142
  readonly name = "CodeRuntimeControlError";
143
143
  constructor(message: string, status: number, code?: string);
144
144
  }
145
+
145
146
  /** Endpoint, host credential, timeout, and cancellation settings for a Code runtime client. */
146
147
  interface CodeRuntimeClientOptions {
147
148
  endpoint: string;
@@ -237,6 +238,30 @@ interface CodeRuntimeControlPlane {
237
238
  heartbeat(runtimeVersion: string, capabilities: CodeRuntimeCapabilities): Promise<CodeRuntimeSnapshot>;
238
239
  acknowledge(commandId: string, result: CodeRuntimeCommandResult): Promise<void>;
239
240
  }
241
+ /** Serializable tool metadata furnished by Registry for one live Code session. */
242
+ interface CodeRuntimeCollaborationToolManifest {
243
+ name: string;
244
+ description: string;
245
+ inputSchema: Record<string, unknown>;
246
+ concurrency?: "parallel";
247
+ outputTaint?: TaintLabel[];
248
+ acceptsTaint?: TaintLabel[];
249
+ }
250
+ /** Serializable skill metadata furnished by Registry for one live Code session. */
251
+ interface CodeRuntimeCollaborationSkillManifest {
252
+ name: string;
253
+ instructions?: string;
254
+ tools: CodeRuntimeCollaborationToolManifest[];
255
+ }
256
+ /** One model tool invocation sent to Registry under the session's authority. */
257
+ interface CodeRuntimeCollaborationToolRequest {
258
+ commandId: string;
259
+ /** Stable provider-issued tool-use id, used by Registry for replay safety. */
260
+ toolCallId: string;
261
+ skill: string;
262
+ tool: string;
263
+ input: Record<string, unknown>;
264
+ }
240
265
  /** Credentialless agent operations brokered over the same outbound host identity. */
241
266
  interface CodeRuntimeAgentControlPlane extends CodeRuntimeControlPlane {
242
267
  source(sessionId: string): Promise<CodeRuntimeSourceSnapshot>;
@@ -252,6 +277,14 @@ interface CodeRuntimeAgentControlPlane extends CodeRuntimeControlPlane {
252
277
  */
253
278
  recallMemories?(sessionId: string, subjects: readonly string[], limit: number): Promise<CodeRuntimeMemory[]>;
254
279
  rememberMemory?(sessionId: string, memory: CodeRuntimeNewMemory): Promise<void>;
280
+ /**
281
+ * Registry-scoped collaboration surface for this exact live session.
282
+ *
283
+ * Optional for rolling compatibility with a Registry that predates the
284
+ * collaboration broker. Absence removes the skills, not the Code session.
285
+ */
286
+ collaborationSkills?(sessionId: string, commandId: string): Promise<CodeRuntimeCollaborationSkillManifest[]>;
287
+ executeCollaborationTool?(sessionId: string, request: CodeRuntimeCollaborationToolRequest, signal?: AbortSignal): Promise<ToolOutput>;
255
288
  appendSessionEvent(sessionId: string, eventId: string, event: CodeSessionEventData): Promise<void>;
256
289
  reportSessionFailure(sessionId: string, message: string): Promise<void>;
257
290
  }
@@ -406,10 +439,10 @@ interface CodeAgentAttemptOptions {
406
439
  /** Run-wide ceiling enforced by runAgent against incremental usage. */
407
440
  budget?: AgentRunBudget;
408
441
  signal?: AbortSignal;
409
- /** Skills the host adds beyond the sandbox surface — PM, chatbuilt by the
410
- * caller because they need a tenant-scoped database credential the harness
411
- * deliberately does not hold. Without them a coordinator can read and patch
412
- * code and then has no way to record a task, a decision or a bug. */
442
+ /** Skills added beyond the sandbox surface — PM, Discussionwhose handlers
443
+ * proxy to Registry over the fenced control plane. The host and harness hold
444
+ * no tenant credential. Without them a coordinator can read and patch code
445
+ * and then has no way to record or discuss durable project state. */
413
446
  extraSkills?: Skill[];
414
447
  /** Emitted per brokered call so the engine can report tool activity. */
415
448
  onToolCall?(call: {
@@ -514,11 +547,11 @@ interface TheseusRuntimeEngineOptions {
514
547
  onDiagnostic?: (message: string) => void;
515
548
  /** Skills the host adds for a session, beyond the sandbox surface.
516
549
  *
517
- * A function of the command rather than a fixed list, because PM is scoped to
518
- * one app and one project: the credential and the project are properties of
519
- * the session being served, not of the engine. Absent means the coordinator
520
- * gets code tools only, which is what it had able to patch a file and
521
- * unable to record that it did.
550
+ * A function of the command rather than a fixed list because Registry fences
551
+ * collaboration to one app, project, and live interaction. The returned
552
+ * handlers proxy through the control plane; neither the engine nor the model
553
+ * receives a tenant credential. Absent means the coordinator gets code tools
554
+ * only — able to patch a file and unable to record that it did.
522
555
  */
523
556
  sessionSkills?: (command: CodeRuntimeCommand) => Skill[] | Promise<Skill[]>;
524
557
  /** Frozen checkout snapshot owned by the foreground terminal process. */
@@ -535,6 +568,22 @@ declare class TheseusRuntimeEngine implements CodeRuntimeCommandEngine {
535
568
  close(): Promise<void>;
536
569
  }
537
570
 
571
+ /**
572
+ * Build the per-command collaboration skill loader used by production Code
573
+ * runtimes. Registry supplies data-only manifests and executes every handler;
574
+ * the model and its credentialless container never receive the host token or a
575
+ * network client.
576
+ */
577
+ declare function createCodeRuntimeSessionSkillLoader(control: Pick<CodeRuntimeAgentControlPlane, "collaborationSkills" | "executeCollaborationTool">): (command: CodeRuntimeCommand) => Promise<Skill[]>;
578
+ /** Registry-brokered skills added for one session beyond the sandbox surface.
579
+ *
580
+ * Resolved per interaction because PM and Discussion are fenced to one app,
581
+ * project, and current command. A failure costs the coordinator those tools
582
+ * and must not cost it the
583
+ * run — trading "cannot record the work" for "cannot do the work" is a worse
584
+ * outcome than the gap it replaces. */
585
+ declare function sessionSkillsFor(options: TheseusRuntimeEngineOptions, command: CodeRuntimeCommand): Promise<Skill[]>;
586
+
538
587
  /** Inputs for capturing one running staged workspace as a portable checkpoint. */
539
588
  interface CreateCodeWorkspaceCheckpointInput {
540
589
  workspace: StagedWorkspace;
@@ -704,10 +753,11 @@ interface RunCodeAgentOptions {
704
753
  * Extra skills composed alongside the code tools — PM, chat, anything else
705
754
  * the host can authorize.
706
755
  *
707
- * They arrive as an argument rather than being built here on purpose: those
708
- * skills need a database credential scoped to a tenant, which the host has
709
- * and the harness must not. It also keeps @odla-ai/pm and @odla-ai/chat out
710
- * of this package's install graph for a capability not every consumer wants.
756
+ * They arrive as an argument rather than being built here on purpose. In
757
+ * production Registry sends metadata-only manifests, while every handler
758
+ * proxies execution back through the fenced Code control plane. Neither the
759
+ * Code host nor the harness receives a tenant credential. This also keeps
760
+ * @odla-ai/pm and @odla-ai/chat out of this package's install graph.
711
761
  */
712
762
  extraSkills?: Skill[];
713
763
  /** Exact registered build/test recipe IDs exposed in the tool schema. */
@@ -1354,4 +1404,4 @@ interface CodeVerificationEvidence {
1354
1404
  /** Rebuild a candidate from an exact trusted base and emit a prose-free clean-verifier receipt. */
1355
1405
  declare function verifyCodeCandidate(input: VerifyCodeCandidateInput): Promise<CodeVerificationEvidence>;
1356
1406
 
1357
- export { CODE_RUNTIME_PROTOCOL_VERSION, type CodeAgentAttemptOptions, type CodeAgentAttemptResult, type CodeAgentRun, type CodeBuildRecipe, type CodeExpectedArtifact, type CodeLocalSourceDescriptor, type CodeMemory, type CodeMemoryEvidence, type CodeMemoryKind, type CodeMemoryStore, type CodeRecipeExecutor, type CodeRecipeResult, type CodeRuntimeAgentControlPlane, type CodeRuntimeBinding, type CodeRuntimeCandidateResponse, type CodeRuntimeCapabilities, CodeRuntimeCheckpointManager, type CodeRuntimeClientOptions, type CodeRuntimeCommand, type CodeRuntimeCommandEngine, type CodeRuntimeCommandKind, type CodeRuntimeCommandResult, CodeRuntimeControlError, type CodeRuntimeControlPlane, type CodeRuntimeInferenceOptions, type CodeRuntimeLoopOptions, type CodeRuntimeMemory, type CodeRuntimeNewMemory, CodeRuntimeReconciler, type CodeRuntimeReviewRequest, type CodeRuntimeReviewResponse, type CodeRuntimeSnapshot, type CodeRuntimeSourceSnapshot, type CodeSkillOpts, type CodeSurface, type CodeToolBrokerOptions, type CodeToolCallRecord, type CodeToolDecision, type CodeVerificationEvidence, type CodeVerificationLog, type CodeVerificationPolicy, type ContainerEngine, type ContainerEngineSelectionOptions, type ContainerEngineVerificationOptions, type ContainerLimits, type ContainerRunOptions, type ContainerRunResult, type CreateCodeWorkspaceCheckpointInput, type DecomposedRun, DecompositionError, type GoalAttempt, type GoalAttemptInput, type GoalAttemptOutcome, type GoalAttemptRecord, type GoalBudget, type GoalEvent, type GoalOutcome, type GoalRun, type GoalRunSpec, type GoalStoppedReason, type GoalStrategy, type HarnessRunnerOptions, type IntegrateOptions, type IntegrationStep, MAX_MEMORY_BODY, MEASURED_PREMIUM, type MaterializedCodeSource, type MaterializedGitTree, type NewCodeMemory, type PrepareRuntimeCheckpointInput, type PreparedRuntimeCheckpoint, type RacedAttemptOptions, type RacedOutcome, type RecallOptions, type RecipeDependencies, type RestoreCodeWorkspaceCheckpointInput, type RestoredCodeWorkspaceCheckpoint, type RunCodeAgentOptions, type RuntimeCheckpointSession, SYSTEM_PROMPT_FOR, type StageWorkspaceOptions, type StagedWorkspace, type StrategyChoice, type StrategySignals, type SubGoal, type SubGoalResult, TheseusRuntimeEngine, type TheseusRuntimeEngineOptions, V1_SYSTEM_PROMPT, V2_SYSTEM_PROMPT, V3_SYSTEM_PROMPT, type VerifyCodeCandidateInput, applyCodePatch, assertCodeBuildRecipe, assertDisjointPlan, assertPinnedImage, attachCodeRuntimeReferences, buildContainerRunArgs, buildRecipeContainerArgs, chooseStrategy, codeSkill, createCodeRuntimeControlClient, createCodeRuntimeInference, createCodeToolBroker, createCodeWorkspaceCheckpoint, createContainerRecipeExecutor, describePatchFailure, digestStagedWorkspace, feedbackIsActionable, hazardFromAttempt, installedDependencies, integrateSubGoals, isCheckpointEffectCompleted, materializeCodeRuntimeSource, materializeCommandWorkspace, materializeGitTree, outcomeCloses, outcomeMemory, patchPaths, planReachCollisions, prepareRuntimeCheckpoint, racedAttempt, recallAbout, registeredFiles, renderMemories, renderOutcome, resolveCodePath, restoreCodeWorkspaceCheckpoint, runCodeAgent, runCodeAgentAttempt, runCodeRuntimeHeartbeatLoop, runContainerAttempt, runGoal, runHarnessRunner, runLeasedAttempt, runtimeMemoryStore, safeWorkspaceLabel, selectContainerEngine, selectWinner, stageWorkspace, stageWorkspacePair, straySubGoalFiles, stripPatchEnvelope, validateCodePatch, validateMemory, validateRelativePath, verifyCodeCandidate, verifyContainerEngineBoundary, withRecipeDependencies };
1407
+ export { CODE_RUNTIME_PROTOCOL_VERSION, type CodeAgentAttemptOptions, type CodeAgentAttemptResult, type CodeAgentRun, type CodeBuildRecipe, type CodeExpectedArtifact, type CodeLocalSourceDescriptor, type CodeMemory, type CodeMemoryEvidence, type CodeMemoryKind, type CodeMemoryStore, type CodeRecipeExecutor, type CodeRecipeResult, type CodeRuntimeAgentControlPlane, type CodeRuntimeBinding, type CodeRuntimeCandidateResponse, type CodeRuntimeCapabilities, CodeRuntimeCheckpointManager, type CodeRuntimeClientOptions, type CodeRuntimeCollaborationSkillManifest, type CodeRuntimeCollaborationToolManifest, type CodeRuntimeCollaborationToolRequest, type CodeRuntimeCommand, type CodeRuntimeCommandEngine, type CodeRuntimeCommandKind, type CodeRuntimeCommandResult, CodeRuntimeControlError, type CodeRuntimeControlPlane, type CodeRuntimeInferenceOptions, type CodeRuntimeLoopOptions, type CodeRuntimeMemory, type CodeRuntimeNewMemory, CodeRuntimeReconciler, type CodeRuntimeReviewRequest, type CodeRuntimeReviewResponse, type CodeRuntimeSnapshot, type CodeRuntimeSourceSnapshot, type CodeSkillOpts, type CodeSurface, type CodeToolBrokerOptions, type CodeToolCallRecord, type CodeToolDecision, type CodeVerificationEvidence, type CodeVerificationLog, type CodeVerificationPolicy, type ContainerEngine, type ContainerEngineSelectionOptions, type ContainerEngineVerificationOptions, type ContainerLimits, type ContainerRunOptions, type ContainerRunResult, type CreateCodeWorkspaceCheckpointInput, type DecomposedRun, DecompositionError, type GoalAttempt, type GoalAttemptInput, type GoalAttemptOutcome, type GoalAttemptRecord, type GoalBudget, type GoalEvent, type GoalOutcome, type GoalRun, type GoalRunSpec, type GoalStoppedReason, type GoalStrategy, type HarnessRunnerOptions, type IntegrateOptions, type IntegrationStep, MAX_MEMORY_BODY, MEASURED_PREMIUM, type MaterializedCodeSource, type MaterializedGitTree, type NewCodeMemory, type PrepareRuntimeCheckpointInput, type PreparedRuntimeCheckpoint, type RacedAttemptOptions, type RacedOutcome, type RecallOptions, type RecipeDependencies, type RestoreCodeWorkspaceCheckpointInput, type RestoredCodeWorkspaceCheckpoint, type RunCodeAgentOptions, type RuntimeCheckpointSession, SYSTEM_PROMPT_FOR, type StageWorkspaceOptions, type StagedWorkspace, type StrategyChoice, type StrategySignals, type SubGoal, type SubGoalResult, TheseusRuntimeEngine, type TheseusRuntimeEngineOptions, V1_SYSTEM_PROMPT, V2_SYSTEM_PROMPT, V3_SYSTEM_PROMPT, type VerifyCodeCandidateInput, applyCodePatch, assertCodeBuildRecipe, assertDisjointPlan, assertPinnedImage, attachCodeRuntimeReferences, buildContainerRunArgs, buildRecipeContainerArgs, chooseStrategy, codeSkill, createCodeRuntimeControlClient, createCodeRuntimeInference, createCodeRuntimeSessionSkillLoader, createCodeToolBroker, createCodeWorkspaceCheckpoint, createContainerRecipeExecutor, describePatchFailure, digestStagedWorkspace, feedbackIsActionable, hazardFromAttempt, installedDependencies, integrateSubGoals, isCheckpointEffectCompleted, materializeCodeRuntimeSource, materializeCommandWorkspace, materializeGitTree, outcomeCloses, outcomeMemory, patchPaths, planReachCollisions, prepareRuntimeCheckpoint, racedAttempt, recallAbout, registeredFiles, renderMemories, renderOutcome, resolveCodePath, restoreCodeWorkspaceCheckpoint, runCodeAgent, runCodeAgentAttempt, runCodeRuntimeHeartbeatLoop, runContainerAttempt, runGoal, runHarnessRunner, runLeasedAttempt, runtimeMemoryStore, safeWorkspaceLabel, selectContainerEngine, selectWinner, sessionSkillsFor, stageWorkspace, stageWorkspacePair, straySubGoalFiles, stripPatchEnvelope, validateCodePatch, validateMemory, validateRelativePath, verifyCodeCandidate, verifyContainerEngineBoundary, withRecipeDependencies };
package/dist/node.d.ts CHANGED
@@ -1,6 +1,6 @@
1
1
  import { t as HarnessTaskSpec, g as HarnessAgentOutput, f as HarnessAgentInput, H as HarnessControlPlane, w as HarnessToolBroker, a as HarnessLease, d as HarnessInferenceRequest, e as HarnessInferenceResponse, i as CodeSessionEventData, x as HarnessToolName, y as HarnessToolRequest } from './types-BNJikP5h.js';
2
2
  import { CodePortableCheckpoint, CodeSourceFile, CodeVerificationReceipt, CodeCheckpointState } from '@odla-ai/camel/code';
3
- import { Skill, Inference, AgentRunBudget, AgentRun, CompactionPolicy } from '@odla-ai/ai';
3
+ import { TaintLabel, ToolOutput, Skill, Inference, AgentRunBudget, AgentRun, CompactionPolicy } from '@odla-ai/ai';
4
4
  import { PolicyOutcome } from '@odla-ai/camel/policy';
5
5
  import { Graph, PartitionVerdict } from '@odla-ai/graph';
6
6
 
@@ -142,6 +142,7 @@ declare class CodeRuntimeControlError extends Error {
142
142
  readonly name = "CodeRuntimeControlError";
143
143
  constructor(message: string, status: number, code?: string);
144
144
  }
145
+
145
146
  /** Endpoint, host credential, timeout, and cancellation settings for a Code runtime client. */
146
147
  interface CodeRuntimeClientOptions {
147
148
  endpoint: string;
@@ -237,6 +238,30 @@ interface CodeRuntimeControlPlane {
237
238
  heartbeat(runtimeVersion: string, capabilities: CodeRuntimeCapabilities): Promise<CodeRuntimeSnapshot>;
238
239
  acknowledge(commandId: string, result: CodeRuntimeCommandResult): Promise<void>;
239
240
  }
241
+ /** Serializable tool metadata furnished by Registry for one live Code session. */
242
+ interface CodeRuntimeCollaborationToolManifest {
243
+ name: string;
244
+ description: string;
245
+ inputSchema: Record<string, unknown>;
246
+ concurrency?: "parallel";
247
+ outputTaint?: TaintLabel[];
248
+ acceptsTaint?: TaintLabel[];
249
+ }
250
+ /** Serializable skill metadata furnished by Registry for one live Code session. */
251
+ interface CodeRuntimeCollaborationSkillManifest {
252
+ name: string;
253
+ instructions?: string;
254
+ tools: CodeRuntimeCollaborationToolManifest[];
255
+ }
256
+ /** One model tool invocation sent to Registry under the session's authority. */
257
+ interface CodeRuntimeCollaborationToolRequest {
258
+ commandId: string;
259
+ /** Stable provider-issued tool-use id, used by Registry for replay safety. */
260
+ toolCallId: string;
261
+ skill: string;
262
+ tool: string;
263
+ input: Record<string, unknown>;
264
+ }
240
265
  /** Credentialless agent operations brokered over the same outbound host identity. */
241
266
  interface CodeRuntimeAgentControlPlane extends CodeRuntimeControlPlane {
242
267
  source(sessionId: string): Promise<CodeRuntimeSourceSnapshot>;
@@ -252,6 +277,14 @@ interface CodeRuntimeAgentControlPlane extends CodeRuntimeControlPlane {
252
277
  */
253
278
  recallMemories?(sessionId: string, subjects: readonly string[], limit: number): Promise<CodeRuntimeMemory[]>;
254
279
  rememberMemory?(sessionId: string, memory: CodeRuntimeNewMemory): Promise<void>;
280
+ /**
281
+ * Registry-scoped collaboration surface for this exact live session.
282
+ *
283
+ * Optional for rolling compatibility with a Registry that predates the
284
+ * collaboration broker. Absence removes the skills, not the Code session.
285
+ */
286
+ collaborationSkills?(sessionId: string, commandId: string): Promise<CodeRuntimeCollaborationSkillManifest[]>;
287
+ executeCollaborationTool?(sessionId: string, request: CodeRuntimeCollaborationToolRequest, signal?: AbortSignal): Promise<ToolOutput>;
255
288
  appendSessionEvent(sessionId: string, eventId: string, event: CodeSessionEventData): Promise<void>;
256
289
  reportSessionFailure(sessionId: string, message: string): Promise<void>;
257
290
  }
@@ -406,10 +439,10 @@ interface CodeAgentAttemptOptions {
406
439
  /** Run-wide ceiling enforced by runAgent against incremental usage. */
407
440
  budget?: AgentRunBudget;
408
441
  signal?: AbortSignal;
409
- /** Skills the host adds beyond the sandbox surface — PM, chatbuilt by the
410
- * caller because they need a tenant-scoped database credential the harness
411
- * deliberately does not hold. Without them a coordinator can read and patch
412
- * code and then has no way to record a task, a decision or a bug. */
442
+ /** Skills added beyond the sandbox surface — PM, Discussionwhose handlers
443
+ * proxy to Registry over the fenced control plane. The host and harness hold
444
+ * no tenant credential. Without them a coordinator can read and patch code
445
+ * and then has no way to record or discuss durable project state. */
413
446
  extraSkills?: Skill[];
414
447
  /** Emitted per brokered call so the engine can report tool activity. */
415
448
  onToolCall?(call: {
@@ -514,11 +547,11 @@ interface TheseusRuntimeEngineOptions {
514
547
  onDiagnostic?: (message: string) => void;
515
548
  /** Skills the host adds for a session, beyond the sandbox surface.
516
549
  *
517
- * A function of the command rather than a fixed list, because PM is scoped to
518
- * one app and one project: the credential and the project are properties of
519
- * the session being served, not of the engine. Absent means the coordinator
520
- * gets code tools only, which is what it had able to patch a file and
521
- * unable to record that it did.
550
+ * A function of the command rather than a fixed list because Registry fences
551
+ * collaboration to one app, project, and live interaction. The returned
552
+ * handlers proxy through the control plane; neither the engine nor the model
553
+ * receives a tenant credential. Absent means the coordinator gets code tools
554
+ * only — able to patch a file and unable to record that it did.
522
555
  */
523
556
  sessionSkills?: (command: CodeRuntimeCommand) => Skill[] | Promise<Skill[]>;
524
557
  /** Frozen checkout snapshot owned by the foreground terminal process. */
@@ -535,6 +568,22 @@ declare class TheseusRuntimeEngine implements CodeRuntimeCommandEngine {
535
568
  close(): Promise<void>;
536
569
  }
537
570
 
571
+ /**
572
+ * Build the per-command collaboration skill loader used by production Code
573
+ * runtimes. Registry supplies data-only manifests and executes every handler;
574
+ * the model and its credentialless container never receive the host token or a
575
+ * network client.
576
+ */
577
+ declare function createCodeRuntimeSessionSkillLoader(control: Pick<CodeRuntimeAgentControlPlane, "collaborationSkills" | "executeCollaborationTool">): (command: CodeRuntimeCommand) => Promise<Skill[]>;
578
+ /** Registry-brokered skills added for one session beyond the sandbox surface.
579
+ *
580
+ * Resolved per interaction because PM and Discussion are fenced to one app,
581
+ * project, and current command. A failure costs the coordinator those tools
582
+ * and must not cost it the
583
+ * run — trading "cannot record the work" for "cannot do the work" is a worse
584
+ * outcome than the gap it replaces. */
585
+ declare function sessionSkillsFor(options: TheseusRuntimeEngineOptions, command: CodeRuntimeCommand): Promise<Skill[]>;
586
+
538
587
  /** Inputs for capturing one running staged workspace as a portable checkpoint. */
539
588
  interface CreateCodeWorkspaceCheckpointInput {
540
589
  workspace: StagedWorkspace;
@@ -704,10 +753,11 @@ interface RunCodeAgentOptions {
704
753
  * Extra skills composed alongside the code tools — PM, chat, anything else
705
754
  * the host can authorize.
706
755
  *
707
- * They arrive as an argument rather than being built here on purpose: those
708
- * skills need a database credential scoped to a tenant, which the host has
709
- * and the harness must not. It also keeps @odla-ai/pm and @odla-ai/chat out
710
- * of this package's install graph for a capability not every consumer wants.
756
+ * They arrive as an argument rather than being built here on purpose. In
757
+ * production Registry sends metadata-only manifests, while every handler
758
+ * proxies execution back through the fenced Code control plane. Neither the
759
+ * Code host nor the harness receives a tenant credential. This also keeps
760
+ * @odla-ai/pm and @odla-ai/chat out of this package's install graph.
711
761
  */
712
762
  extraSkills?: Skill[];
713
763
  /** Exact registered build/test recipe IDs exposed in the tool schema. */
@@ -1354,4 +1404,4 @@ interface CodeVerificationEvidence {
1354
1404
  /** Rebuild a candidate from an exact trusted base and emit a prose-free clean-verifier receipt. */
1355
1405
  declare function verifyCodeCandidate(input: VerifyCodeCandidateInput): Promise<CodeVerificationEvidence>;
1356
1406
 
1357
- export { CODE_RUNTIME_PROTOCOL_VERSION, type CodeAgentAttemptOptions, type CodeAgentAttemptResult, type CodeAgentRun, type CodeBuildRecipe, type CodeExpectedArtifact, type CodeLocalSourceDescriptor, type CodeMemory, type CodeMemoryEvidence, type CodeMemoryKind, type CodeMemoryStore, type CodeRecipeExecutor, type CodeRecipeResult, type CodeRuntimeAgentControlPlane, type CodeRuntimeBinding, type CodeRuntimeCandidateResponse, type CodeRuntimeCapabilities, CodeRuntimeCheckpointManager, type CodeRuntimeClientOptions, type CodeRuntimeCommand, type CodeRuntimeCommandEngine, type CodeRuntimeCommandKind, type CodeRuntimeCommandResult, CodeRuntimeControlError, type CodeRuntimeControlPlane, type CodeRuntimeInferenceOptions, type CodeRuntimeLoopOptions, type CodeRuntimeMemory, type CodeRuntimeNewMemory, CodeRuntimeReconciler, type CodeRuntimeReviewRequest, type CodeRuntimeReviewResponse, type CodeRuntimeSnapshot, type CodeRuntimeSourceSnapshot, type CodeSkillOpts, type CodeSurface, type CodeToolBrokerOptions, type CodeToolCallRecord, type CodeToolDecision, type CodeVerificationEvidence, type CodeVerificationLog, type CodeVerificationPolicy, type ContainerEngine, type ContainerEngineSelectionOptions, type ContainerEngineVerificationOptions, type ContainerLimits, type ContainerRunOptions, type ContainerRunResult, type CreateCodeWorkspaceCheckpointInput, type DecomposedRun, DecompositionError, type GoalAttempt, type GoalAttemptInput, type GoalAttemptOutcome, type GoalAttemptRecord, type GoalBudget, type GoalEvent, type GoalOutcome, type GoalRun, type GoalRunSpec, type GoalStoppedReason, type GoalStrategy, type HarnessRunnerOptions, type IntegrateOptions, type IntegrationStep, MAX_MEMORY_BODY, MEASURED_PREMIUM, type MaterializedCodeSource, type MaterializedGitTree, type NewCodeMemory, type PrepareRuntimeCheckpointInput, type PreparedRuntimeCheckpoint, type RacedAttemptOptions, type RacedOutcome, type RecallOptions, type RecipeDependencies, type RestoreCodeWorkspaceCheckpointInput, type RestoredCodeWorkspaceCheckpoint, type RunCodeAgentOptions, type RuntimeCheckpointSession, SYSTEM_PROMPT_FOR, type StageWorkspaceOptions, type StagedWorkspace, type StrategyChoice, type StrategySignals, type SubGoal, type SubGoalResult, TheseusRuntimeEngine, type TheseusRuntimeEngineOptions, V1_SYSTEM_PROMPT, V2_SYSTEM_PROMPT, V3_SYSTEM_PROMPT, type VerifyCodeCandidateInput, applyCodePatch, assertCodeBuildRecipe, assertDisjointPlan, assertPinnedImage, attachCodeRuntimeReferences, buildContainerRunArgs, buildRecipeContainerArgs, chooseStrategy, codeSkill, createCodeRuntimeControlClient, createCodeRuntimeInference, createCodeToolBroker, createCodeWorkspaceCheckpoint, createContainerRecipeExecutor, describePatchFailure, digestStagedWorkspace, feedbackIsActionable, hazardFromAttempt, installedDependencies, integrateSubGoals, isCheckpointEffectCompleted, materializeCodeRuntimeSource, materializeCommandWorkspace, materializeGitTree, outcomeCloses, outcomeMemory, patchPaths, planReachCollisions, prepareRuntimeCheckpoint, racedAttempt, recallAbout, registeredFiles, renderMemories, renderOutcome, resolveCodePath, restoreCodeWorkspaceCheckpoint, runCodeAgent, runCodeAgentAttempt, runCodeRuntimeHeartbeatLoop, runContainerAttempt, runGoal, runHarnessRunner, runLeasedAttempt, runtimeMemoryStore, safeWorkspaceLabel, selectContainerEngine, selectWinner, stageWorkspace, stageWorkspacePair, straySubGoalFiles, stripPatchEnvelope, validateCodePatch, validateMemory, validateRelativePath, verifyCodeCandidate, verifyContainerEngineBoundary, withRecipeDependencies };
1407
+ export { CODE_RUNTIME_PROTOCOL_VERSION, type CodeAgentAttemptOptions, type CodeAgentAttemptResult, type CodeAgentRun, type CodeBuildRecipe, type CodeExpectedArtifact, type CodeLocalSourceDescriptor, type CodeMemory, type CodeMemoryEvidence, type CodeMemoryKind, type CodeMemoryStore, type CodeRecipeExecutor, type CodeRecipeResult, type CodeRuntimeAgentControlPlane, type CodeRuntimeBinding, type CodeRuntimeCandidateResponse, type CodeRuntimeCapabilities, CodeRuntimeCheckpointManager, type CodeRuntimeClientOptions, type CodeRuntimeCollaborationSkillManifest, type CodeRuntimeCollaborationToolManifest, type CodeRuntimeCollaborationToolRequest, type CodeRuntimeCommand, type CodeRuntimeCommandEngine, type CodeRuntimeCommandKind, type CodeRuntimeCommandResult, CodeRuntimeControlError, type CodeRuntimeControlPlane, type CodeRuntimeInferenceOptions, type CodeRuntimeLoopOptions, type CodeRuntimeMemory, type CodeRuntimeNewMemory, CodeRuntimeReconciler, type CodeRuntimeReviewRequest, type CodeRuntimeReviewResponse, type CodeRuntimeSnapshot, type CodeRuntimeSourceSnapshot, type CodeSkillOpts, type CodeSurface, type CodeToolBrokerOptions, type CodeToolCallRecord, type CodeToolDecision, type CodeVerificationEvidence, type CodeVerificationLog, type CodeVerificationPolicy, type ContainerEngine, type ContainerEngineSelectionOptions, type ContainerEngineVerificationOptions, type ContainerLimits, type ContainerRunOptions, type ContainerRunResult, type CreateCodeWorkspaceCheckpointInput, type DecomposedRun, DecompositionError, type GoalAttempt, type GoalAttemptInput, type GoalAttemptOutcome, type GoalAttemptRecord, type GoalBudget, type GoalEvent, type GoalOutcome, type GoalRun, type GoalRunSpec, type GoalStoppedReason, type GoalStrategy, type HarnessRunnerOptions, type IntegrateOptions, type IntegrationStep, MAX_MEMORY_BODY, MEASURED_PREMIUM, type MaterializedCodeSource, type MaterializedGitTree, type NewCodeMemory, type PrepareRuntimeCheckpointInput, type PreparedRuntimeCheckpoint, type RacedAttemptOptions, type RacedOutcome, type RecallOptions, type RecipeDependencies, type RestoreCodeWorkspaceCheckpointInput, type RestoredCodeWorkspaceCheckpoint, type RunCodeAgentOptions, type RuntimeCheckpointSession, SYSTEM_PROMPT_FOR, type StageWorkspaceOptions, type StagedWorkspace, type StrategyChoice, type StrategySignals, type SubGoal, type SubGoalResult, TheseusRuntimeEngine, type TheseusRuntimeEngineOptions, V1_SYSTEM_PROMPT, V2_SYSTEM_PROMPT, V3_SYSTEM_PROMPT, type VerifyCodeCandidateInput, applyCodePatch, assertCodeBuildRecipe, assertDisjointPlan, assertPinnedImage, attachCodeRuntimeReferences, buildContainerRunArgs, buildRecipeContainerArgs, chooseStrategy, codeSkill, createCodeRuntimeControlClient, createCodeRuntimeInference, createCodeRuntimeSessionSkillLoader, createCodeToolBroker, createCodeWorkspaceCheckpoint, createContainerRecipeExecutor, describePatchFailure, digestStagedWorkspace, feedbackIsActionable, hazardFromAttempt, installedDependencies, integrateSubGoals, isCheckpointEffectCompleted, materializeCodeRuntimeSource, materializeCommandWorkspace, materializeGitTree, outcomeCloses, outcomeMemory, patchPaths, planReachCollisions, prepareRuntimeCheckpoint, racedAttempt, recallAbout, registeredFiles, renderMemories, renderOutcome, resolveCodePath, restoreCodeWorkspaceCheckpoint, runCodeAgent, runCodeAgentAttempt, runCodeRuntimeHeartbeatLoop, runContainerAttempt, runGoal, runHarnessRunner, runLeasedAttempt, runtimeMemoryStore, safeWorkspaceLabel, selectContainerEngine, selectWinner, sessionSkillsFor, stageWorkspace, stageWorkspacePair, straySubGoalFiles, stripPatchEnvelope, validateCodePatch, validateMemory, validateRelativePath, verifyCodeCandidate, verifyContainerEngineBoundary, withRecipeDependencies };
package/dist/node.js CHANGED
@@ -20,6 +20,7 @@ import {
20
20
  codeSkill,
21
21
  createCodeRuntimeControlClient,
22
22
  createCodeRuntimeInference,
23
+ createCodeRuntimeSessionSkillLoader,
23
24
  createCodeToolBroker,
24
25
  createCodeWorkspaceCheckpoint,
25
26
  createContainerRecipeExecutor,
@@ -39,12 +40,13 @@ import {
39
40
  runCodeAgentAttempt,
40
41
  runCodeRuntimeHeartbeatLoop,
41
42
  runGoal,
43
+ sessionSkillsFor,
42
44
  stripPatchEnvelope,
43
45
  validateCodePatch,
44
46
  validateMemory,
45
47
  validateRelativePath,
46
48
  verifyCodeCandidate
47
- } from "./chunk-IWVGSWY6.js";
49
+ } from "./chunk-NG7AYYH3.js";
48
50
  import {
49
51
  assertPinnedImage,
50
52
  buildContainerRunArgs,
@@ -383,6 +385,7 @@ export {
383
385
  codeSkill,
384
386
  createCodeRuntimeControlClient,
385
387
  createCodeRuntimeInference,
388
+ createCodeRuntimeSessionSkillLoader,
386
389
  createCodeToolBroker,
387
390
  createCodeWorkspaceCheckpoint,
388
391
  createContainerRecipeExecutor,
@@ -419,6 +422,7 @@ export {
419
422
  safeWorkspaceLabel,
420
423
  selectContainerEngine,
421
424
  selectWinner,
425
+ sessionSkillsFor,
422
426
  stageWorkspace,
423
427
  stageWorkspacePair,
424
428
  straySubGoalFiles,
package/dist/node.js.map CHANGED
@@ -1 +1 @@
1
- {"version":3,"sources":["../src/code-runtime-memory.ts","../src/code-goal-outcome.ts","../src/code-goal-race.ts","../src/code-goal-decompose.ts","../src/code-goal-strategy.ts","../src/code-recipe-dependencies.ts"],"sourcesContent":["// The brokered control plane, presented as the memory store the runner wants.\n//\n// The Goal Runner takes a CodeMemoryStore; the runtime has a control plane that\n// speaks HTTP over a host credential. This is the few lines between them, and it\n// exists so neither side learns about the other: the runner never knows memory\n// crosses a network, and the control plane never learns what a hazard is.\n\nimport type { CodeMemory, CodeMemoryStore, NewCodeMemory } from \"./code-memory\";\nimport type { CodeRuntimeAgentControlPlane } from \"./code-runtime\";\n\n/**\n * A key derived from the memory itself, so a retry writes once.\n *\n * Content-derived rather than random: the retry that matters is the one after a\n * timeout, where the caller cannot know whether the first write landed — and a\n * fresh random key would guarantee a duplicate exactly then.\n */\nasync function mutationId(memory: NewCodeMemory): Promise<string> {\n const source = `${memory.subject} ${memory.kind} ${memory.body}`;\n const bytes = await crypto.subtle.digest(\"SHA-256\", new TextEncoder().encode(source));\n return Array.from(new Uint8Array(bytes), (byte) => byte.toString(16).padStart(2, \"0\"))\n .join(\"\").slice(0, 40);\n}\n\n/**\n * Present a session's brokered memory as a store, or null when the control\n * plane does not offer one.\n *\n * Null rather than a throwing stub: a Code host talking to a registry that\n * predates memory should quietly lose the feature, not fail every goal that\n * would have recorded something.\n */\nexport function runtimeMemoryStore(\n control: CodeRuntimeAgentControlPlane, sessionId: string,\n): CodeMemoryStore | null {\n const recall = control.recallMemories;\n const remember = control.rememberMemory;\n if (!recall || !remember) return null;\n return {\n recall: async (subjects, limit) =>\n (await recall.call(control, sessionId, subjects, limit)) as CodeMemory[],\n remember: async (memory: NewCodeMemory) => {\n await remember.call(control, sessionId, {\n subject: memory.subject,\n kind: memory.kind,\n body: memory.body,\n ...(memory.evidence ? { evidence: memory.evidence } : {}),\n mutationId: await mutationId(memory),\n });\n // The control plane owns the id and the author — attribution a writer\n // chooses is not attribution — so what comes back here is only enough to\n // satisfy the interface the runner holds.\n return { ...memory, id: \"\", createdAt: Date.now() };\n },\n };\n}\n","// Comparing what a goal achieved to what it was for.\n//\n// A PM goal already carries a `proof` — the sentence that says how you would\n// know it was met. The Goal Runner already produces a signed verification\n// receipt for every attempt. Nothing connected them, so a goal moved to met\n// because an agent said it was.\n//\n// That is precisely the edit the PM audit trail now flags as self-serving: the\n// principal doing the work moving its own goalposts. I did it myself half a\n// dozen times this session, by hand, and the board has no way to tell those\n// apart from a real completion.\n//\n// So an outcome is a THIRD thing, produced by neither the goal's author nor its\n// executor: the verifier's verdict, recorded against the goal, with the receipt\n// that backs it. A goal reaches met because evidence exists, or it does not\n// reach met.\n\nimport type { CodeMemoryEvidence, NewCodeMemory } from \"./code-memory\";\n\n/** What a run actually did, judged rather than claimed. */\nexport interface GoalOutcome {\n /** The PM entity this goal came from, when it came from one. */\n pmEntityId?: string;\n goal: string;\n /** How the goal said it would be judged. */\n proof: string;\n /** Whether the clean verifier agreed. */\n met: boolean;\n /** Why it stopped — `proof_passed`, or the budget that ran out. */\n stoppedReason: string;\n attempts: number;\n tokens: number;\n costUsd?: number;\n /** The receipt that backs a met verdict. Absent means nothing backs it. */\n evidence?: CodeMemoryEvidence;\n}\n\n/**\n * Whether an outcome may move a PM item to done.\n *\n * Two conditions, and the second is the one that matters: the verifier agreed,\n * AND it left a receipt. A `met` with no evidence is an agent's assertion\n * wearing a verdict's clothes, and admitting it would make the whole audit\n * trail decorative.\n */\nexport function outcomeCloses(outcome: GoalOutcome): boolean {\n return outcome.met && outcome.evidence !== undefined;\n}\n\n/**\n * Say plainly what happened, for the PM comment that records it.\n *\n * Written for a human deciding whether to trust the result, so the numbers that\n * bound the claim — attempts, tokens, cost — are in the sentence rather than in\n * a linked artifact nobody opens. Unknown cost is said as unknown; reporting\n * $0.00 for a run whose pricing was unavailable would be a lie in the direction\n * that flatters the agent.\n */\nexport function renderOutcome(outcome: GoalOutcome): string {\n const spend = outcome.costUsd === undefined\n ? \"cost unknown\"\n : `$${outcome.costUsd.toFixed(4)}`;\n const scale = `${outcome.attempts} attempt(s), ${outcome.tokens.toLocaleString()} tokens, ${spend}`;\n if (!outcome.met) {\n return `Goal not met (${outcome.stoppedReason}) after ${scale}. Proof: ${outcome.proof}`;\n }\n if (!outcome.evidence) {\n // Deliberately not phrased as success. A verdict with nothing behind it is\n // the exact shape of a self-certifying agent, and it should read as one.\n return `Goal reported met after ${scale}, but no verification receipt was produced. ` +\n `Treat as unverified. Proof: ${outcome.proof}`;\n }\n return `Goal met after ${scale}. Verified by ${outcome.evidence.kind}:${outcome.evidence.ref}. ` +\n `Proof: ${outcome.proof}`;\n}\n\n/**\n * Record the outcome as a memory, so the next run knows what this one settled.\n *\n * Filed against the goal's subject rather than a file: an outcome is about an\n * intention, and the useful question later is \"has anyone tried this before\",\n * not \"what happened to line 40\".\n */\nexport function outcomeMemory(\n outcome: GoalOutcome,\n subject: string,\n authorId: string,\n): NewCodeMemory {\n return {\n subject,\n kind: \"outcome\",\n body: renderOutcome(outcome),\n ...(outcome.evidence ? { evidence: outcome.evidence } : {}),\n authorId,\n };\n}\n","// Racing: N attempts at the same goal, in parallel, winner picked by receipt.\n//\n// This is an ATTEMPT STRATEGY, not a second runner. `racedAttempt` returns a\n// GoalAttempt, so runGoal is unchanged and racing composes with everything it\n// already does — budgets, deadlines, re-prompting on the winner's feedback.\n//\n// It answers a different failure than re-prompting does. Re-prompting helps when\n// the gate said something actionable; racing helps when attempts fail\n// DIFFERENTLY each time and there is nothing to act on, only variance to sample\n// out. Measured on this corpus: gpt-5.6-terra failed two-stage-gate once in a\n// single-attempt run, then passed it twice.\n\nimport type { GoalAttempt, GoalAttemptInput, GoalAttemptOutcome } from \"./code-goal-runner\";\n\n/** One racer's result, tagged with which racer produced it. */\nexport interface RacedOutcome extends GoalAttemptOutcome {\n racer: number;\n}\n\n/**\n * One race: N independent attempts at the same goal, cheapest winner taken.\n *\n * The run is charged for every racer, not just the winner. Racing answers\n * variance, not size; at equal budget, depth beat it on every axis measured.\n */\nexport interface RacedAttemptOptions {\n /** How many attempts run per round. */\n racers: number;\n /** Runs one racer. Each MUST get an isolated workspace — racers edit in\n * parallel, and a shared tree would have them overwrite each other. */\n attempt(input: GoalAttemptInput & { racer: number }): Promise<GoalAttemptOutcome>;\n /** Override the ranking. Default: {@link selectWinner}. */\n select?(outcomes: RacedOutcome[]): RacedOutcome;\n /** Observe the whole field, for reporting what racing cost and bought. */\n onRound?(outcomes: RacedOutcome[], winner: RacedOutcome): void;\n}\n\n/**\n * Rank a field of racers.\n *\n * Passing candidates first — that is the only thing that decides the goal. Then\n * fewest steps, then cheapest, then smallest patch: three tie-breaks that all\n * prefer the attempt which did LESS to get there, on the view that a smaller\n * change reaching the same verified state is the better one to keep.\n *\n * Unknown cost ranks LAST among passing candidates rather than cheapest. An\n * unpriced racer must not win a cost comparison it never entered.\n */\nexport function selectWinner(outcomes: RacedOutcome[]): RacedOutcome {\n const ranked = [...outcomes].sort((left, right) => {\n if (left.gatePassed !== right.gatePassed) return left.gatePassed ? -1 : 1;\n // An attempt that could not run is worse than one that ran and failed: it\n // produced no evidence at all.\n const leftBroken = left.error ? 1 : 0;\n const rightBroken = right.error ? 1 : 0;\n if (leftBroken !== rightBroken) return leftBroken - rightBroken;\n const steps = (left.steps ?? Number.MAX_SAFE_INTEGER) - (right.steps ?? Number.MAX_SAFE_INTEGER);\n if (steps !== 0) return steps;\n // Unknown cost is not a number to compare, so handle it as its own case:\n // both unknown is a tie, one unknown loses to any known value. Subtracting\n // sentinels here quietly produced Infinity for the one-unknown case, which a\n // finite-check then skipped — leaving the unpriced racer to win by position.\n const leftPriced = left.costUsd !== undefined;\n const rightPriced = right.costUsd !== undefined;\n if (leftPriced !== rightPriced) return leftPriced ? -1 : 1;\n if (leftPriced && rightPriced && left.costUsd !== right.costUsd) {\n return left.costUsd! - right.costUsd!;\n }\n return (left.patchBytes ?? Number.MAX_SAFE_INTEGER) - (right.patchBytes ?? Number.MAX_SAFE_INTEGER);\n });\n return ranked[0]!;\n}\n\n/**\n * Build a GoalAttempt that races `racers` attempts and returns the winner.\n *\n * The returned outcome reports the round's TOTAL tokens and cost, not the\n * winner's. The runner charges its budget from what an attempt reports, and\n * charging only the winner would make racing look free — three racers would\n * cost what one did, and every budget in the system would be wrong by a factor\n * of N. Racing trades depth for breadth at honest spend, or it is not a trade.\n */\nexport function racedAttempt(options: RacedAttemptOptions): GoalAttempt {\n if (!Number.isSafeInteger(options.racers) || options.racers < 1) {\n throw new TypeError(\"racers must be a positive integer\");\n }\n const select = options.select ?? selectWinner;\n return async (input) => {\n const outcomes = await Promise.all(\n Array.from({ length: options.racers }, async (_unused, index): Promise<RacedOutcome> => {\n const racer = index + 1;\n try {\n return { ...(await options.attempt({ ...input, racer })), racer };\n } catch (cause) {\n return {\n racer, gatePassed: false, feedback: \"\", tokens: 0,\n error: (cause instanceof Error ? cause.message : String(cause)).slice(0, 500),\n };\n }\n }),\n );\n const winner = select(outcomes);\n options.onRound?.(outcomes, winner);\n\n const tokens = outcomes.reduce((total, outcome) => total + outcome.tokens, 0);\n const priced = outcomes.filter((outcome) => outcome.costUsd !== undefined);\n // Only report a cost when EVERY racer priced. A partial sum would understate\n // the round and let an unpriced racer hide spend inside a budget check.\n const costUsd = priced.length === outcomes.length\n ? priced.reduce((total, outcome) => total + (outcome.costUsd ?? 0), 0)\n : undefined;\n\n return {\n gatePassed: winner.gatePassed,\n feedback: winner.feedback,\n tokens,\n ...(costUsd === undefined ? {} : { costUsd }),\n ...(winner.steps === undefined ? {} : { steps: winner.steps }),\n ...(winner.patchBytes === undefined ? {} : { patchBytes: winner.patchBytes }),\n // The round only failed to RUN if every racer did. One survivor is a round.\n ...(outcomes.every((outcome) => outcome.error) ? { error: winner.error } : {}),\n };\n };\n}\n","// Decomposition: split a goal into sub-goals that cannot collide, work them in\n// parallel, integrate them one at a time.\n//\n// The obstacle is specific. The workspace model is single-candidate: every\n// attempt owns one staged tree, and the verifier's contract is \"one patch, from\n// one trusted base, re-applied clean\". It cannot tell you that three partial\n// patches compose. So decomposition here does NOT merge.\n//\n// Instead the partition is made CHECKABLE. A plan declares, per sub-goal, the\n// files it will touch; overlapping declarations are rejected before any agent\n// starts; sub-agents work in isolated trees from the same base; and integration\n// applies each sub-patch to an accumulating tree, gating after each. Parallel\n// exploration, sequential application — no merge algorithm anywhere, and a\n// sub-agent that strays outside its declared files fails its own integration.\n\nimport { validateCodePatch } from \"./code-patch\";\nimport { partition, type PartitionVerdict } from \"@odla-ai/graph\";\nimport { buildCodeGraph, FILE, IMPORTS, READS, WRITES } from \"@odla-ai/graph/code\";\n\n/** One independent slice of a goal, scoped to the files it may touch. */\nexport interface SubGoal {\n id: string;\n /** What this sub-agent is asked to do. */\n goal: string;\n /** The files it declares it will touch. Its patch is rejected if it strays. */\n files: string[];\n}\n\n/** A plan that cannot be run in parallel — overlapping files, or a stray edit. */\nexport class DecompositionError extends Error {\n constructor(message: string) {\n super(message);\n this.name = \"DecompositionError\";\n }\n}\n\n/**\n * Reject a plan that cannot be worked in parallel, BEFORE any agent starts.\n *\n * Two sub-goals declaring the same file is the whole failure mode: they would\n * produce patches against the same lines and one of them could not be applied.\n * Catching it here costs nothing; catching it at integration costs every token\n * both sub-agents spent.\n */\nexport function assertDisjointPlan(plan: readonly SubGoal[]): void {\n if (plan.length === 0) throw new DecompositionError(\"a plan needs at least one sub-goal\");\n const ids = new Set<string>();\n const owner = new Map<string, string>();\n for (const sub of plan) {\n if (!sub.id.trim()) throw new DecompositionError(\"every sub-goal needs an id\");\n if (ids.has(sub.id)) throw new DecompositionError(`duplicate sub-goal id \"${sub.id}\"`);\n ids.add(sub.id);\n if (!sub.goal.trim()) throw new DecompositionError(`sub-goal \"${sub.id}\" has no instruction`);\n if (sub.files.length === 0) {\n throw new DecompositionError(`sub-goal \"${sub.id}\" declares no files; it cannot be checked for collisions`);\n }\n for (const file of sub.files) {\n const claimed = owner.get(file);\n if (claimed !== undefined) {\n throw new DecompositionError(\n `sub-goals \"${claimed}\" and \"${sub.id}\" both declare \"${file}\"; a plan must partition the files it touches`,\n );\n }\n owner.set(file, sub.id);\n }\n }\n}\n\n/** Files a patch actually touches, or null when it is not a valid patch. */\nexport function patchPaths(patch: string, maxBytes = 256 * 1024): string[] | null {\n try { return validateCodePatch(patch, maxBytes); }\n catch { return null; }\n}\n\n/**\n * Check a sub-agent stayed inside what it declared.\n *\n * This is what makes the partition a guarantee rather than a hope: the plan is\n * checked up front, and each patch is checked against the plan. A sub-agent that\n * wandered into a neighbour's file is refused even if its patch would have\n * applied cleanly, because the NEXT sub-patch was written against a tree where\n * that edit does not exist.\n */\nexport function straySubGoalFiles(sub: SubGoal, patch: string): string[] {\n const declared = new Set(sub.files);\n return (patchPaths(patch) ?? []).filter((path) => !declared.has(path));\n}\n\n/** What one sub-goal produced: its patch, its cost, and whether it worked. */\nexport interface SubGoalResult {\n sub: SubGoal;\n /** The candidate patch, or empty when the sub-agent changed nothing. */\n patch: string;\n tokens: number;\n costUsd?: number;\n error?: string;\n}\n\n/** Applying one sub-goal's patch to the accumulating tree, and the verdict. */\nexport interface IntegrationStep {\n subGoalId: string;\n /** Did the accumulated tree still satisfy the proof after applying this one? */\n gatePassed: boolean;\n applied: boolean;\n reason?: \"stray_files\" | \"did_not_apply\" | \"gate_failed\" | \"sub_goal_failed\" | \"no_changes\";\n detail?: string;\n}\n\n/**\n * A decomposed pursuit end to end.\n *\n * Measured at 1.68x the tokens of straight depth for 0.88x the wall clock, so\n * this buys latency, not efficiency — see chooseStrategy for when that trade\n * is worth making.\n */\nexport interface DecomposedRun {\n met: boolean;\n steps: IntegrationStep[];\n tokens: number;\n costUsd?: number;\n}\n\n/** The plan, its results, and the two callbacks that apply and judge them. */\nexport interface IntegrateOptions {\n plan: readonly SubGoal[];\n results: readonly SubGoalResult[];\n /** Apply one patch to the accumulating tree; reject means it did not apply. */\n apply(patch: string): Promise<boolean>;\n /** Run the proof against the accumulating tree. */\n gate(): Promise<boolean>;\n}\n\n/**\n * Integrate sub-results one at a time, gating after each.\n *\n * Order follows the plan, and the first failure stops integration. Continuing\n * past one would mean gating a tree whose earlier layer is already known bad,\n * so every later verdict would describe a state nobody intends to ship.\n */\nexport async function integrateSubGoals(options: IntegrateOptions): Promise<DecomposedRun> {\n const steps: IntegrationStep[] = [];\n let tokens = 0;\n let costUsd = 0;\n let priced = true;\n let met = false;\n\n for (const sub of options.plan) {\n const result = options.results.find((entry) => entry.sub.id === sub.id);\n if (!result) {\n steps.push({ subGoalId: sub.id, gatePassed: false, applied: false, reason: \"sub_goal_failed\", detail: \"no result\" });\n break;\n }\n tokens += result.tokens;\n if (result.costUsd === undefined) priced = false;\n else costUsd += result.costUsd;\n\n if (result.error) {\n steps.push({ subGoalId: sub.id, gatePassed: false, applied: false, reason: \"sub_goal_failed\", detail: result.error });\n break;\n }\n if (!result.patch.trim()) {\n steps.push({ subGoalId: sub.id, gatePassed: false, applied: false, reason: \"no_changes\" });\n break;\n }\n const stray = straySubGoalFiles(sub, result.patch);\n if (stray.length > 0) {\n steps.push({\n subGoalId: sub.id, gatePassed: false, applied: false, reason: \"stray_files\",\n detail: `touched undeclared files: ${stray.join(\", \")}`,\n });\n break;\n }\n if (!(await options.apply(result.patch))) {\n steps.push({ subGoalId: sub.id, gatePassed: false, applied: false, reason: \"did_not_apply\" });\n break;\n }\n const gatePassed = await options.gate();\n steps.push({ subGoalId: sub.id, gatePassed, applied: true, ...(gatePassed ? {} : { reason: \"gate_failed\" as const }) });\n met = gatePassed;\n // A mid-plan gate failure is expected — the goal is only whole once every\n // sub-goal has landed — so integration continues. What stops it is a patch\n // that could not be applied at all, above.\n }\n\n return { met, steps, tokens, ...(priced && options.results.length > 0 ? { costUsd } : {}) };\n}\n\n/**\n * The check `assertDisjointPlan` cannot make: do the sub-goals collide through\n * what they *reach*, not merely through what they declare?\n *\n * Collision ids are graph ids (`file:src/a.ts`, `table:orders`), because the\n * answer is no longer only about files — a shared table is a real collision and\n * naming it as a bare path would be a lie about what it is.\n *\n * Declared files being disjoint is necessary and not sufficient. Two sub-goals\n * can own different modules and still both depend on a third; a change either\n * one makes to that shared module lands in one patch and is invisible to the\n * other's verification. The import graph is the only thing that can see this,\n * and it is cheap — the graph over 3,400 files builds in ~400ms, against the\n * cost of running two agents to completion and discovering it at integration.\n *\n * Reported rather than thrown, because a shared dependency is sometimes fine:\n * two sub-goals may both READ a types module neither intends to touch. The\n * caller decides whether the overlap is one it can live with.\n */\nexport async function planReachCollisions(\n plan: readonly SubGoal[],\n workspace: { paths: readonly string[]; read(path: string): Promise<string> },\n): Promise<PartitionVerdict> {\n const graph = await buildCodeGraph({ paths: workspace.paths, read: workspace.read });\n // Imports AND data, in one traversal. Two sub-goals that never import each\n // other can still both write the same table, and that collision is the one\n // structural analysis alone can never see.\n return partition(graph, plan.map((sub) => sub.files.map((file) => `${FILE}:${file}`)), {\n kinds: [IMPORTS, READS, WRITES], direction: \"out\",\n });\n}\n","// Choosing how to pursue a goal: sequentially, by racing, or by decomposing.\n//\n// The plan that produced this file assumed fan-out would usually win. Measured,\n// it usually loses. On every fixture tried, depth beat breadth on $/solve:\n//\n// racing, equal budget of 6 agent runs (two-stage-gate, gpt-5.6-terra)\n// 1 racer x 6 attempts 7,936 tokens/solve 16.9 s/solve\n// 2 racers x 3 attempts 14,055 tokens/solve 16.8 s/solve\n// 3 racers x 2 attempts 21,044 tokens/solve 19.7 s/solve\n//\n// decomposition (three-modules, gpt-5.6-terra)\n// sequential 6,474 tokens/solve 9.7 s/solve\n// decomposed x3 10,897 tokens/solve 8.5 s/solve\n//\n// One cause explains both: every parallel worker re-pays the orientation cost —\n// reading the project, taking the system prompt, finding its feet — while a\n// sequential run amortizes that across one conversation. Breadth buys wall\n// clock and nothing else.\n//\n// So this chooser is deliberately biased toward sequential, and every fan-out\n// branch has to justify itself against a measured premium.\n\n/** What is known about a goal before choosing how to pursue it. */\nexport interface StrategySignals {\n /** Prior attempts, oldest first. Empty on the first pass. */\n priorAttempts?: Array<{ gatePassed: boolean; feedback: string; error?: string }>;\n /** True when wall clock is the binding constraint — a human is waiting, or a\n * deadline is close — and paying more per solve to finish sooner is correct. */\n latencyBound?: boolean;\n /** Estimated tokens needed to hold the whole goal at once. */\n estimatedContextTokens?: number;\n /** What one agent can actually hold. */\n contextLimit?: number;\n /** A checked disjoint partition, when the planner produced one. */\n partitionable?: boolean;\n /** Racers/sub-agents available if fanning out. */\n width?: number;\n}\n\n/** How to spend a goal's budget: straight depth, N racers, or a split plan. */\nexport type GoalStrategy = \"sequential\" | \"race\" | \"decompose\";\n\n/** The chosen strategy, why, and what it is expected to cost against depth. */\nexport interface StrategyChoice {\n strategy: GoalStrategy;\n /** Why, in terms a human reviewing a run can check. */\n reason: string;\n /** Expected cost multiplier vs sequential, from the measurements above.\n * 1 for sequential; fan-out is charged what it was observed to cost. */\n expectedCostMultiplier: number;\n /** False when the branch rests on something not yet measured. */\n measured: boolean;\n}\n\n/** Observed premiums. Update these from the bench, not from expectation. */\nexport const MEASURED_PREMIUM = Object.freeze({\n /** 3 racers vs pure depth at equal budget: 21,044 / 7,936. */\n racePerRacer: 0.55,\n /** Decomposition across 3 sub-agents: 10,897 / 6,474. */\n decomposePerSubGoal: 0.23,\n});\n\nconst actionable = (feedback: string): boolean => {\n const text = feedback.trim();\n if (text.length < 12) return false;\n // Something the next attempt can act on names a place or an expectation. A\n // bare \"it failed\" leaves depth nothing to build on, which is the one case\n // where sampling another attempt beats trying again.\n return /\\b(expected|assert|error|fail(?:ed|ure)?|exit|line \\d+|\\.[a-z]{1,4}:\\d+)\\b/i.test(text)\n || /\\.(js|ts|tsx|jsx|mjs|cjs|py|go|rs|java|rb)\\b/i.test(text);\n};\n\n/**\n * Choose a strategy.\n *\n * Order matters and encodes the evidence: depth first, because it won every\n * measured comparison; fan-out only where depth demonstrably cannot help.\n */\nexport function chooseStrategy(signals: StrategySignals = {}): StrategyChoice {\n const width = Math.max(1, signals.width ?? 3);\n const prior = signals.priorAttempts ?? [];\n\n // 1. The goal does not fit. Depth cannot help — every attempt starts from the\n // same too-large problem — so this is the one case where decomposition is\n // not a premium but the only option.\n const overflows = signals.estimatedContextTokens !== undefined\n && signals.contextLimit !== undefined\n && signals.estimatedContextTokens > signals.contextLimit;\n if (overflows && signals.partitionable) {\n return {\n strategy: \"decompose\",\n reason: `the goal needs ~${signals.estimatedContextTokens} tokens against a ${signals.contextLimit} limit, and the plan partitions`,\n expectedCostMultiplier: 1 + MEASURED_PREMIUM.decomposePerSubGoal * (width - 1),\n // No fixture this large has been measured. This branch is reasoned, not\n // observed, and says so rather than borrowing the others' credibility.\n measured: false,\n };\n }\n if (overflows) {\n return {\n strategy: \"sequential\",\n reason: \"the goal exceeds one context but no disjoint partition was produced; decomposing without one would collide\",\n expectedCostMultiplier: 1,\n measured: true,\n };\n }\n\n // 2. Attempts are failing with nothing to act on. Re-prompting repeats itself;\n // another sample is the only thing that changes.\n const unhelpful = prior.length >= 2\n && prior.slice(-2).every((attempt) => !attempt.gatePassed && !attempt.error && !actionable(attempt.feedback));\n if (unhelpful) {\n return {\n strategy: \"race\",\n reason: `the last ${Math.min(2, prior.length)} gate failures carried nothing actionable, so depth has nothing to build on`,\n expectedCostMultiplier: 1 + MEASURED_PREMIUM.racePerRacer * (width - 1),\n measured: true,\n };\n }\n\n // 3. Latency is the constraint and the caller has accepted the premium.\n if (signals.latencyBound) {\n const strategy = signals.partitionable ? \"decompose\" : \"race\";\n const premium = strategy === \"decompose\"\n ? MEASURED_PREMIUM.decomposePerSubGoal\n : MEASURED_PREMIUM.racePerRacer;\n return {\n strategy,\n reason: `wall clock is the binding constraint; ${strategy} finishes sooner at a measured premium per solve`,\n expectedCostMultiplier: 1 + premium * (width - 1),\n measured: true,\n };\n }\n\n // 4. Everything else. Depth won every comparison run so far, so it is not a\n // fallback — it is the answer unless something above overrides it.\n return {\n strategy: \"sequential\",\n reason: prior.length === 0\n ? \"no evidence yet favours paying a fan-out premium\"\n : \"the gate is still saying something the next attempt can act on\",\n expectedCostMultiplier: 1,\n measured: true,\n };\n}\n\n/** Whether the gate's output gives the next attempt something to work with. */\nexport const feedbackIsActionable = actionable;\n","// Making installed dependencies available to a recipe, without copying them and\n// without exposing them to the agent.\n//\n// These are three separate questions that were previously answered by one list:\n//\n// 1. can the agent READ or PATCH it? No. `RESERVED` in code-patch already\n// refuses node_modules paths whatever is on disk, so the agent never sees\n// dependencies in `list`, `search`, `read`, or a candidate patch.\n// 2. is it COPIED into every staged tree? No, and it must not be: 70,834\n// files and 1.0GB against a 20,000-file cap, staged at least four times per\n// gated attempt (the attempt, each run_recipe, the verifier's base, and\n// each clean-verify recipe).\n// 3. is it PRESENT when the recipe runs? That is this file. Linked in for\n// the duration of one recipe and removed after, so the tree the verifier\n// diffs and digests is unchanged.\n//\n// The third answer is what makes a real test runner usable as a proof, which in\n// turn is what makes an end-to-end test expressible as a gate at all.\n\nimport { lstat, rm, symlink } from \"node:fs/promises\";\nimport { isAbsolute, join } from \"node:path\";\nimport type { CodeRecipeExecutor, CodeRecipeResult } from \"./code-tool-types\";\n\n/**\n * An installed dependency tree lent to a recipe for the length of one run.\n *\n * Mounted under a reserved name and removed in a finally, so the agent can run\n * a real build without the tree ever being addressable by a patch.\n */\nexport interface RecipeDependencies {\n /** Absolute path to an installed dependency tree on the host. */\n source: string;\n /** Where it appears inside the workspace. Must be a reserved name, so the\n * agent still cannot address it. */\n mountAs?: string;\n}\n\nconst RESERVED_MOUNTS = new Set([\"node_modules\", \"dist\", \"coverage\"]);\n\n/**\n * Wrap an executor so `dependencies.source` is present during each recipe run.\n *\n * It is linked, not copied. A link costs nothing per run, and dependencies are\n * identical across every attempt, racer and verification — copying them would\n * multiply the largest thing in the tree by the number of stages.\n *\n * The link is removed in a `finally`, so a recipe that times out or throws\n * cannot leave it behind for `workspace.patch()` to diff or\n * `digestStagedWorkspace` to hash.\n */\nexport function withRecipeDependencies(\n executor: CodeRecipeExecutor,\n dependencies: RecipeDependencies,\n): CodeRecipeExecutor {\n const mountAs = dependencies.mountAs ?? \"node_modules\";\n if (!isAbsolute(dependencies.source)) {\n throw new TypeError(\"recipe dependency source must be an absolute path\");\n }\n // Mounting at a name the agent CAN address would hand it a writable path into\n // the host's dependency tree through apply_patch.\n if (!RESERVED_MOUNTS.has(mountAs)) {\n throw new TypeError(`recipe dependencies must mount at a reserved name, not \"${mountAs}\"`);\n }\n\n return {\n run: async (input): Promise<CodeRecipeResult> => {\n const target = join(input.workspaceDir, mountAs);\n let linked = false;\n try {\n // An existing entry is left alone: a workspace that already carries its\n // own dependencies is not ours to replace.\n const existing = await lstat(target).catch(() => null);\n if (!existing) {\n await symlink(dependencies.source, target, \"dir\");\n linked = true;\n }\n return await executor.run(input);\n } finally {\n if (linked) await rm(target, { force: true, recursive: false }).catch(() => undefined);\n }\n },\n };\n}\n\n/** Resolve the dependency tree for a repository root, when it has one. */\nexport async function installedDependencies(repoRoot: string): Promise<RecipeDependencies | null> {\n const source = join(repoRoot, \"node_modules\");\n const info = await lstat(source).catch(() => null);\n return info?.isDirectory() ? { source } : null;\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAiBA,eAAe,WAAW,QAAwC;AAChE,QAAM,SAAS,GAAG,OAAO,OAAO,IAAI,OAAO,IAAI,IAAI,OAAO,IAAI;AAC9D,QAAM,QAAQ,MAAM,OAAO,OAAO,OAAO,WAAW,IAAI,YAAY,EAAE,OAAO,MAAM,CAAC;AACpF,SAAO,MAAM,KAAK,IAAI,WAAW,KAAK,GAAG,CAAC,SAAS,KAAK,SAAS,EAAE,EAAE,SAAS,GAAG,GAAG,CAAC,EAClF,KAAK,EAAE,EAAE,MAAM,GAAG,EAAE;AACzB;AAUO,SAAS,mBACd,SAAuC,WACf;AACxB,QAAM,SAAS,QAAQ;AACvB,QAAM,WAAW,QAAQ;AACzB,MAAI,CAAC,UAAU,CAAC,SAAU,QAAO;AACjC,SAAO;AAAA,IACL,QAAQ,OAAO,UAAU,UACtB,MAAM,OAAO,KAAK,SAAS,WAAW,UAAU,KAAK;AAAA,IACxD,UAAU,OAAO,WAA0B;AACzC,YAAM,SAAS,KAAK,SAAS,WAAW;AAAA,QACtC,SAAS,OAAO;AAAA,QAChB,MAAM,OAAO;AAAA,QACb,MAAM,OAAO;AAAA,QACb,GAAI,OAAO,WAAW,EAAE,UAAU,OAAO,SAAS,IAAI,CAAC;AAAA,QACvD,YAAY,MAAM,WAAW,MAAM;AAAA,MACrC,CAAC;AAID,aAAO,EAAE,GAAG,QAAQ,IAAI,IAAI,WAAW,KAAK,IAAI,EAAE;AAAA,IACpD;AAAA,EACF;AACF;;;ACVO,SAAS,cAAc,SAA+B;AAC3D,SAAO,QAAQ,OAAO,QAAQ,aAAa;AAC7C;AAWO,SAAS,cAAc,SAA8B;AAC1D,QAAM,QAAQ,QAAQ,YAAY,SAC9B,iBACA,IAAI,QAAQ,QAAQ,QAAQ,CAAC,CAAC;AAClC,QAAM,QAAQ,GAAG,QAAQ,QAAQ,gBAAgB,QAAQ,OAAO,eAAe,CAAC,YAAY,KAAK;AACjG,MAAI,CAAC,QAAQ,KAAK;AAChB,WAAO,iBAAiB,QAAQ,aAAa,WAAW,KAAK,YAAY,QAAQ,KAAK;AAAA,EACxF;AACA,MAAI,CAAC,QAAQ,UAAU;AAGrB,WAAO,2BAA2B,KAAK,2EACN,QAAQ,KAAK;AAAA,EAChD;AACA,SAAO,kBAAkB,KAAK,iBAAiB,QAAQ,SAAS,IAAI,IAAI,QAAQ,SAAS,GAAG,YAChF,QAAQ,KAAK;AAC3B;AASO,SAAS,cACd,SACA,SACA,UACe;AACf,SAAO;AAAA,IACL;AAAA,IACA,MAAM;AAAA,IACN,MAAM,cAAc,OAAO;AAAA,IAC3B,GAAI,QAAQ,WAAW,EAAE,UAAU,QAAQ,SAAS,IAAI,CAAC;AAAA,IACzD;AAAA,EACF;AACF;;;AC/CO,SAAS,aAAa,UAAwC;AACnE,QAAM,SAAS,CAAC,GAAG,QAAQ,EAAE,KAAK,CAAC,MAAM,UAAU;AACjD,QAAI,KAAK,eAAe,MAAM,WAAY,QAAO,KAAK,aAAa,KAAK;AAGxE,UAAM,aAAa,KAAK,QAAQ,IAAI;AACpC,UAAM,cAAc,MAAM,QAAQ,IAAI;AACtC,QAAI,eAAe,YAAa,QAAO,aAAa;AACpD,UAAM,SAAS,KAAK,SAAS,OAAO,qBAAqB,MAAM,SAAS,OAAO;AAC/E,QAAI,UAAU,EAAG,QAAO;AAKxB,UAAM,aAAa,KAAK,YAAY;AACpC,UAAM,cAAc,MAAM,YAAY;AACtC,QAAI,eAAe,YAAa,QAAO,aAAa,KAAK;AACzD,QAAI,cAAc,eAAe,KAAK,YAAY,MAAM,SAAS;AAC/D,aAAO,KAAK,UAAW,MAAM;AAAA,IAC/B;AACA,YAAQ,KAAK,cAAc,OAAO,qBAAqB,MAAM,cAAc,OAAO;AAAA,EACpF,CAAC;AACD,SAAO,OAAO,CAAC;AACjB;AAWO,SAAS,aAAa,SAA2C;AACtE,MAAI,CAAC,OAAO,cAAc,QAAQ,MAAM,KAAK,QAAQ,SAAS,GAAG;AAC/D,UAAM,IAAI,UAAU,mCAAmC;AAAA,EACzD;AACA,QAAM,SAAS,QAAQ,UAAU;AACjC,SAAO,OAAO,UAAU;AACtB,UAAM,WAAW,MAAM,QAAQ;AAAA,MAC7B,MAAM,KAAK,EAAE,QAAQ,QAAQ,OAAO,GAAG,OAAO,SAAS,UAAiC;AACtF,cAAM,QAAQ,QAAQ;AACtB,YAAI;AACF,iBAAO,EAAE,GAAI,MAAM,QAAQ,QAAQ,EAAE,GAAG,OAAO,MAAM,CAAC,GAAI,MAAM;AAAA,QAClE,SAAS,OAAO;AACd,iBAAO;AAAA,YACL;AAAA,YAAO,YAAY;AAAA,YAAO,UAAU;AAAA,YAAI,QAAQ;AAAA,YAChD,QAAQ,iBAAiB,QAAQ,MAAM,UAAU,OAAO,KAAK,GAAG,MAAM,GAAG,GAAG;AAAA,UAC9E;AAAA,QACF;AAAA,MACF,CAAC;AAAA,IACH;AACA,UAAM,SAAS,OAAO,QAAQ;AAC9B,YAAQ,UAAU,UAAU,MAAM;AAElC,UAAM,SAAS,SAAS,OAAO,CAAC,OAAO,YAAY,QAAQ,QAAQ,QAAQ,CAAC;AAC5E,UAAM,SAAS,SAAS,OAAO,CAAC,YAAY,QAAQ,YAAY,MAAS;AAGzE,UAAM,UAAU,OAAO,WAAW,SAAS,SACvC,OAAO,OAAO,CAAC,OAAO,YAAY,SAAS,QAAQ,WAAW,IAAI,CAAC,IACnE;AAEJ,WAAO;AAAA,MACL,YAAY,OAAO;AAAA,MACnB,UAAU,OAAO;AAAA,MACjB;AAAA,MACA,GAAI,YAAY,SAAY,CAAC,IAAI,EAAE,QAAQ;AAAA,MAC3C,GAAI,OAAO,UAAU,SAAY,CAAC,IAAI,EAAE,OAAO,OAAO,MAAM;AAAA,MAC5D,GAAI,OAAO,eAAe,SAAY,CAAC,IAAI,EAAE,YAAY,OAAO,WAAW;AAAA;AAAA,MAE3E,GAAI,SAAS,MAAM,CAAC,YAAY,QAAQ,KAAK,IAAI,EAAE,OAAO,OAAO,MAAM,IAAI,CAAC;AAAA,IAC9E;AAAA,EACF;AACF;;;AC3GA,SAAS,iBAAwC;AACjD,SAAS,gBAAgB,MAAM,SAAS,OAAO,cAAc;AAYtD,IAAM,qBAAN,cAAiC,MAAM;AAAA,EAC5C,YAAY,SAAiB;AAC3B,UAAM,OAAO;AACb,SAAK,OAAO;AAAA,EACd;AACF;AAUO,SAAS,mBAAmB,MAAgC;AACjE,MAAI,KAAK,WAAW,EAAG,OAAM,IAAI,mBAAmB,oCAAoC;AACxF,QAAM,MAAM,oBAAI,IAAY;AAC5B,QAAM,QAAQ,oBAAI,IAAoB;AACtC,aAAW,OAAO,MAAM;AACtB,QAAI,CAAC,IAAI,GAAG,KAAK,EAAG,OAAM,IAAI,mBAAmB,4BAA4B;AAC7E,QAAI,IAAI,IAAI,IAAI,EAAE,EAAG,OAAM,IAAI,mBAAmB,0BAA0B,IAAI,EAAE,GAAG;AACrF,QAAI,IAAI,IAAI,EAAE;AACd,QAAI,CAAC,IAAI,KAAK,KAAK,EAAG,OAAM,IAAI,mBAAmB,aAAa,IAAI,EAAE,sBAAsB;AAC5F,QAAI,IAAI,MAAM,WAAW,GAAG;AAC1B,YAAM,IAAI,mBAAmB,aAAa,IAAI,EAAE,0DAA0D;AAAA,IAC5G;AACA,eAAW,QAAQ,IAAI,OAAO;AAC5B,YAAM,UAAU,MAAM,IAAI,IAAI;AAC9B,UAAI,YAAY,QAAW;AACzB,cAAM,IAAI;AAAA,UACR,cAAc,OAAO,UAAU,IAAI,EAAE,mBAAmB,IAAI;AAAA,QAC9D;AAAA,MACF;AACA,YAAM,IAAI,MAAM,IAAI,EAAE;AAAA,IACxB;AAAA,EACF;AACF;AAGO,SAAS,WAAW,OAAe,WAAW,MAAM,MAAuB;AAChF,MAAI;AAAE,WAAO,kBAAkB,OAAO,QAAQ;AAAA,EAAG,QAC3C;AAAE,WAAO;AAAA,EAAM;AACvB;AAWO,SAAS,kBAAkB,KAAc,OAAyB;AACvE,QAAM,WAAW,IAAI,IAAI,IAAI,KAAK;AAClC,UAAQ,WAAW,KAAK,KAAK,CAAC,GAAG,OAAO,CAAC,SAAS,CAAC,SAAS,IAAI,IAAI,CAAC;AACvE;AAqDA,eAAsB,kBAAkB,SAAmD;AACzF,QAAM,QAA2B,CAAC;AAClC,MAAI,SAAS;AACb,MAAI,UAAU;AACd,MAAI,SAAS;AACb,MAAI,MAAM;AAEV,aAAW,OAAO,QAAQ,MAAM;AAC9B,UAAM,SAAS,QAAQ,QAAQ,KAAK,CAAC,UAAU,MAAM,IAAI,OAAO,IAAI,EAAE;AACtE,QAAI,CAAC,QAAQ;AACX,YAAM,KAAK,EAAE,WAAW,IAAI,IAAI,YAAY,OAAO,SAAS,OAAO,QAAQ,mBAAmB,QAAQ,YAAY,CAAC;AACnH;AAAA,IACF;AACA,cAAU,OAAO;AACjB,QAAI,OAAO,YAAY,OAAW,UAAS;AAAA,QACtC,YAAW,OAAO;AAEvB,QAAI,OAAO,OAAO;AAChB,YAAM,KAAK,EAAE,WAAW,IAAI,IAAI,YAAY,OAAO,SAAS,OAAO,QAAQ,mBAAmB,QAAQ,OAAO,MAAM,CAAC;AACpH;AAAA,IACF;AACA,QAAI,CAAC,OAAO,MAAM,KAAK,GAAG;AACxB,YAAM,KAAK,EAAE,WAAW,IAAI,IAAI,YAAY,OAAO,SAAS,OAAO,QAAQ,aAAa,CAAC;AACzF;AAAA,IACF;AACA,UAAM,QAAQ,kBAAkB,KAAK,OAAO,KAAK;AACjD,QAAI,MAAM,SAAS,GAAG;AACpB,YAAM,KAAK;AAAA,QACT,WAAW,IAAI;AAAA,QAAI,YAAY;AAAA,QAAO,SAAS;AAAA,QAAO,QAAQ;AAAA,QAC9D,QAAQ,6BAA6B,MAAM,KAAK,IAAI,CAAC;AAAA,MACvD,CAAC;AACD;AAAA,IACF;AACA,QAAI,CAAE,MAAM,QAAQ,MAAM,OAAO,KAAK,GAAI;AACxC,YAAM,KAAK,EAAE,WAAW,IAAI,IAAI,YAAY,OAAO,SAAS,OAAO,QAAQ,gBAAgB,CAAC;AAC5F;AAAA,IACF;AACA,UAAM,aAAa,MAAM,QAAQ,KAAK;AACtC,UAAM,KAAK,EAAE,WAAW,IAAI,IAAI,YAAY,SAAS,MAAM,GAAI,aAAa,CAAC,IAAI,EAAE,QAAQ,cAAuB,EAAG,CAAC;AACtH,UAAM;AAAA,EAIR;AAEA,SAAO,EAAE,KAAK,OAAO,QAAQ,GAAI,UAAU,QAAQ,QAAQ,SAAS,IAAI,EAAE,QAAQ,IAAI,CAAC,EAAG;AAC5F;AAqBA,eAAsB,oBACpB,MACA,WAC2B;AAC3B,QAAM,QAAQ,MAAM,eAAe,EAAE,OAAO,UAAU,OAAO,MAAM,UAAU,KAAK,CAAC;AAInF,SAAO,UAAU,OAAO,KAAK,IAAI,CAAC,QAAQ,IAAI,MAAM,IAAI,CAAC,SAAS,GAAG,IAAI,IAAI,IAAI,EAAE,CAAC,GAAG;AAAA,IACrF,OAAO,CAAC,SAAS,OAAO,MAAM;AAAA,IAAG,WAAW;AAAA,EAC9C,CAAC;AACH;;;AClKO,IAAM,mBAAmB,OAAO,OAAO;AAAA;AAAA,EAE5C,cAAc;AAAA;AAAA,EAEd,qBAAqB;AACvB,CAAC;AAED,IAAM,aAAa,CAAC,aAA8B;AAChD,QAAM,OAAO,SAAS,KAAK;AAC3B,MAAI,KAAK,SAAS,GAAI,QAAO;AAI7B,SAAO,8EAA8E,KAAK,IAAI,KACzF,gDAAgD,KAAK,IAAI;AAChE;AAQO,SAAS,eAAe,UAA2B,CAAC,GAAmB;AAC5E,QAAM,QAAQ,KAAK,IAAI,GAAG,QAAQ,SAAS,CAAC;AAC5C,QAAM,QAAQ,QAAQ,iBAAiB,CAAC;AAKxC,QAAM,YAAY,QAAQ,2BAA2B,UAChD,QAAQ,iBAAiB,UACzB,QAAQ,yBAAyB,QAAQ;AAC9C,MAAI,aAAa,QAAQ,eAAe;AACtC,WAAO;AAAA,MACL,UAAU;AAAA,MACV,QAAQ,mBAAmB,QAAQ,sBAAsB,qBAAqB,QAAQ,YAAY;AAAA,MAClG,wBAAwB,IAAI,iBAAiB,uBAAuB,QAAQ;AAAA;AAAA;AAAA,MAG5E,UAAU;AAAA,IACZ;AAAA,EACF;AACA,MAAI,WAAW;AACb,WAAO;AAAA,MACL,UAAU;AAAA,MACV,QAAQ;AAAA,MACR,wBAAwB;AAAA,MACxB,UAAU;AAAA,IACZ;AAAA,EACF;AAIA,QAAM,YAAY,MAAM,UAAU,KAC7B,MAAM,MAAM,EAAE,EAAE,MAAM,CAAC,YAAY,CAAC,QAAQ,cAAc,CAAC,QAAQ,SAAS,CAAC,WAAW,QAAQ,QAAQ,CAAC;AAC9G,MAAI,WAAW;AACb,WAAO;AAAA,MACL,UAAU;AAAA,MACV,QAAQ,YAAY,KAAK,IAAI,GAAG,MAAM,MAAM,CAAC;AAAA,MAC7C,wBAAwB,IAAI,iBAAiB,gBAAgB,QAAQ;AAAA,MACrE,UAAU;AAAA,IACZ;AAAA,EACF;AAGA,MAAI,QAAQ,cAAc;AACxB,UAAM,WAAW,QAAQ,gBAAgB,cAAc;AACvD,UAAM,UAAU,aAAa,cACzB,iBAAiB,sBACjB,iBAAiB;AACrB,WAAO;AAAA,MACL;AAAA,MACA,QAAQ,yCAAyC,QAAQ;AAAA,MACzD,wBAAwB,IAAI,WAAW,QAAQ;AAAA,MAC/C,UAAU;AAAA,IACZ;AAAA,EACF;AAIA,SAAO;AAAA,IACL,UAAU;AAAA,IACV,QAAQ,MAAM,WAAW,IACrB,qDACA;AAAA,IACJ,wBAAwB;AAAA,IACxB,UAAU;AAAA,EACZ;AACF;AAGO,IAAM,uBAAuB;;;AChIpC,SAAS,OAAO,IAAI,eAAe;AACnC,SAAS,YAAY,YAAY;AAiBjC,IAAM,kBAAkB,oBAAI,IAAI,CAAC,gBAAgB,QAAQ,UAAU,CAAC;AAa7D,SAAS,uBACd,UACA,cACoB;AACpB,QAAM,UAAU,aAAa,WAAW;AACxC,MAAI,CAAC,WAAW,aAAa,MAAM,GAAG;AACpC,UAAM,IAAI,UAAU,mDAAmD;AAAA,EACzE;AAGA,MAAI,CAAC,gBAAgB,IAAI,OAAO,GAAG;AACjC,UAAM,IAAI,UAAU,2DAA2D,OAAO,GAAG;AAAA,EAC3F;AAEA,SAAO;AAAA,IACL,KAAK,OAAO,UAAqC;AAC/C,YAAM,SAAS,KAAK,MAAM,cAAc,OAAO;AAC/C,UAAI,SAAS;AACb,UAAI;AAGF,cAAM,WAAW,MAAM,MAAM,MAAM,EAAE,MAAM,MAAM,IAAI;AACrD,YAAI,CAAC,UAAU;AACb,gBAAM,QAAQ,aAAa,QAAQ,QAAQ,KAAK;AAChD,mBAAS;AAAA,QACX;AACA,eAAO,MAAM,SAAS,IAAI,KAAK;AAAA,MACjC,UAAE;AACA,YAAI,OAAQ,OAAM,GAAG,QAAQ,EAAE,OAAO,MAAM,WAAW,MAAM,CAAC,EAAE,MAAM,MAAM,MAAS;AAAA,MACvF;AAAA,IACF;AAAA,EACF;AACF;AAGA,eAAsB,sBAAsB,UAAsD;AAChG,QAAM,SAAS,KAAK,UAAU,cAAc;AAC5C,QAAM,OAAO,MAAM,MAAM,MAAM,EAAE,MAAM,MAAM,IAAI;AACjD,SAAO,MAAM,YAAY,IAAI,EAAE,OAAO,IAAI;AAC5C;","names":[]}
1
+ {"version":3,"sources":["../src/code-runtime-memory.ts","../src/code-goal-outcome.ts","../src/code-goal-race.ts","../src/code-goal-decompose.ts","../src/code-goal-strategy.ts","../src/code-recipe-dependencies.ts"],"sourcesContent":["// The brokered control plane, presented as the memory store the runner wants.\n//\n// The Goal Runner takes a CodeMemoryStore; the runtime has a control plane that\n// speaks HTTP over a host credential. This is the few lines between them, and it\n// exists so neither side learns about the other: the runner never knows memory\n// crosses a network, and the control plane never learns what a hazard is.\n\nimport type { CodeMemory, CodeMemoryStore, NewCodeMemory } from \"./code-memory\";\nimport type { CodeRuntimeAgentControlPlane } from \"./code-runtime\";\n\n/**\n * A key derived from the memory itself, so a retry writes once.\n *\n * Content-derived rather than random: the retry that matters is the one after a\n * timeout, where the caller cannot know whether the first write landed — and a\n * fresh random key would guarantee a duplicate exactly then.\n */\nasync function mutationId(memory: NewCodeMemory): Promise<string> {\n const source = `${memory.subject} ${memory.kind} ${memory.body}`;\n const bytes = await crypto.subtle.digest(\"SHA-256\", new TextEncoder().encode(source));\n return Array.from(new Uint8Array(bytes), (byte) => byte.toString(16).padStart(2, \"0\"))\n .join(\"\").slice(0, 40);\n}\n\n/**\n * Present a session's brokered memory as a store, or null when the control\n * plane does not offer one.\n *\n * Null rather than a throwing stub: a Code host talking to a registry that\n * predates memory should quietly lose the feature, not fail every goal that\n * would have recorded something.\n */\nexport function runtimeMemoryStore(\n control: CodeRuntimeAgentControlPlane, sessionId: string,\n): CodeMemoryStore | null {\n const recall = control.recallMemories;\n const remember = control.rememberMemory;\n if (!recall || !remember) return null;\n return {\n recall: async (subjects, limit) =>\n (await recall.call(control, sessionId, subjects, limit)) as CodeMemory[],\n remember: async (memory: NewCodeMemory) => {\n await remember.call(control, sessionId, {\n subject: memory.subject,\n kind: memory.kind,\n body: memory.body,\n ...(memory.evidence ? { evidence: memory.evidence } : {}),\n mutationId: await mutationId(memory),\n });\n // The control plane owns the id and the author — attribution a writer\n // chooses is not attribution — so what comes back here is only enough to\n // satisfy the interface the runner holds.\n return { ...memory, id: \"\", createdAt: Date.now() };\n },\n };\n}\n","// Comparing what a goal achieved to what it was for.\n//\n// A PM goal already carries a `proof` — the sentence that says how you would\n// know it was met. The Goal Runner already produces a signed verification\n// receipt for every attempt. Nothing connected them, so a goal moved to met\n// because an agent said it was.\n//\n// That is precisely the edit the PM audit trail now flags as self-serving: the\n// principal doing the work moving its own goalposts. I did it myself half a\n// dozen times this session, by hand, and the board has no way to tell those\n// apart from a real completion.\n//\n// So an outcome is a THIRD thing, produced by neither the goal's author nor its\n// executor: the verifier's verdict, recorded against the goal, with the receipt\n// that backs it. A goal reaches met because evidence exists, or it does not\n// reach met.\n\nimport type { CodeMemoryEvidence, NewCodeMemory } from \"./code-memory\";\n\n/** What a run actually did, judged rather than claimed. */\nexport interface GoalOutcome {\n /** The PM entity this goal came from, when it came from one. */\n pmEntityId?: string;\n goal: string;\n /** How the goal said it would be judged. */\n proof: string;\n /** Whether the clean verifier agreed. */\n met: boolean;\n /** Why it stopped — `proof_passed`, or the budget that ran out. */\n stoppedReason: string;\n attempts: number;\n tokens: number;\n costUsd?: number;\n /** The receipt that backs a met verdict. Absent means nothing backs it. */\n evidence?: CodeMemoryEvidence;\n}\n\n/**\n * Whether an outcome may move a PM item to done.\n *\n * Two conditions, and the second is the one that matters: the verifier agreed,\n * AND it left a receipt. A `met` with no evidence is an agent's assertion\n * wearing a verdict's clothes, and admitting it would make the whole audit\n * trail decorative.\n */\nexport function outcomeCloses(outcome: GoalOutcome): boolean {\n return outcome.met && outcome.evidence !== undefined;\n}\n\n/**\n * Say plainly what happened, for the PM comment that records it.\n *\n * Written for a human deciding whether to trust the result, so the numbers that\n * bound the claim — attempts, tokens, cost — are in the sentence rather than in\n * a linked artifact nobody opens. Unknown cost is said as unknown; reporting\n * $0.00 for a run whose pricing was unavailable would be a lie in the direction\n * that flatters the agent.\n */\nexport function renderOutcome(outcome: GoalOutcome): string {\n const spend = outcome.costUsd === undefined\n ? \"cost unknown\"\n : `$${outcome.costUsd.toFixed(4)}`;\n const scale = `${outcome.attempts} attempt(s), ${outcome.tokens.toLocaleString()} tokens, ${spend}`;\n if (!outcome.met) {\n return `Goal not met (${outcome.stoppedReason}) after ${scale}. Proof: ${outcome.proof}`;\n }\n if (!outcome.evidence) {\n // Deliberately not phrased as success. A verdict with nothing behind it is\n // the exact shape of a self-certifying agent, and it should read as one.\n return `Goal reported met after ${scale}, but no verification receipt was produced. ` +\n `Treat as unverified. Proof: ${outcome.proof}`;\n }\n return `Goal met after ${scale}. Verified by ${outcome.evidence.kind}:${outcome.evidence.ref}. ` +\n `Proof: ${outcome.proof}`;\n}\n\n/**\n * Record the outcome as a memory, so the next run knows what this one settled.\n *\n * Filed against the goal's subject rather than a file: an outcome is about an\n * intention, and the useful question later is \"has anyone tried this before\",\n * not \"what happened to line 40\".\n */\nexport function outcomeMemory(\n outcome: GoalOutcome,\n subject: string,\n authorId: string,\n): NewCodeMemory {\n return {\n subject,\n kind: \"outcome\",\n body: renderOutcome(outcome),\n ...(outcome.evidence ? { evidence: outcome.evidence } : {}),\n authorId,\n };\n}\n","// Racing: N attempts at the same goal, in parallel, winner picked by receipt.\n//\n// This is an ATTEMPT STRATEGY, not a second runner. `racedAttempt` returns a\n// GoalAttempt, so runGoal is unchanged and racing composes with everything it\n// already does — budgets, deadlines, re-prompting on the winner's feedback.\n//\n// It answers a different failure than re-prompting does. Re-prompting helps when\n// the gate said something actionable; racing helps when attempts fail\n// DIFFERENTLY each time and there is nothing to act on, only variance to sample\n// out. Measured on this corpus: gpt-5.6-terra failed two-stage-gate once in a\n// single-attempt run, then passed it twice.\n\nimport type { GoalAttempt, GoalAttemptInput, GoalAttemptOutcome } from \"./code-goal-runner\";\n\n/** One racer's result, tagged with which racer produced it. */\nexport interface RacedOutcome extends GoalAttemptOutcome {\n racer: number;\n}\n\n/**\n * One race: N independent attempts at the same goal, cheapest winner taken.\n *\n * The run is charged for every racer, not just the winner. Racing answers\n * variance, not size; at equal budget, depth beat it on every axis measured.\n */\nexport interface RacedAttemptOptions {\n /** How many attempts run per round. */\n racers: number;\n /** Runs one racer. Each MUST get an isolated workspace — racers edit in\n * parallel, and a shared tree would have them overwrite each other. */\n attempt(input: GoalAttemptInput & { racer: number }): Promise<GoalAttemptOutcome>;\n /** Override the ranking. Default: {@link selectWinner}. */\n select?(outcomes: RacedOutcome[]): RacedOutcome;\n /** Observe the whole field, for reporting what racing cost and bought. */\n onRound?(outcomes: RacedOutcome[], winner: RacedOutcome): void;\n}\n\n/**\n * Rank a field of racers.\n *\n * Passing candidates first — that is the only thing that decides the goal. Then\n * fewest steps, then cheapest, then smallest patch: three tie-breaks that all\n * prefer the attempt which did LESS to get there, on the view that a smaller\n * change reaching the same verified state is the better one to keep.\n *\n * Unknown cost ranks LAST among passing candidates rather than cheapest. An\n * unpriced racer must not win a cost comparison it never entered.\n */\nexport function selectWinner(outcomes: RacedOutcome[]): RacedOutcome {\n const ranked = [...outcomes].sort((left, right) => {\n if (left.gatePassed !== right.gatePassed) return left.gatePassed ? -1 : 1;\n // An attempt that could not run is worse than one that ran and failed: it\n // produced no evidence at all.\n const leftBroken = left.error ? 1 : 0;\n const rightBroken = right.error ? 1 : 0;\n if (leftBroken !== rightBroken) return leftBroken - rightBroken;\n const steps = (left.steps ?? Number.MAX_SAFE_INTEGER) - (right.steps ?? Number.MAX_SAFE_INTEGER);\n if (steps !== 0) return steps;\n // Unknown cost is not a number to compare, so handle it as its own case:\n // both unknown is a tie, one unknown loses to any known value. Subtracting\n // sentinels here quietly produced Infinity for the one-unknown case, which a\n // finite-check then skipped — leaving the unpriced racer to win by position.\n const leftPriced = left.costUsd !== undefined;\n const rightPriced = right.costUsd !== undefined;\n if (leftPriced !== rightPriced) return leftPriced ? -1 : 1;\n if (leftPriced && rightPriced && left.costUsd !== right.costUsd) {\n return left.costUsd! - right.costUsd!;\n }\n return (left.patchBytes ?? Number.MAX_SAFE_INTEGER) - (right.patchBytes ?? Number.MAX_SAFE_INTEGER);\n });\n return ranked[0]!;\n}\n\n/**\n * Build a GoalAttempt that races `racers` attempts and returns the winner.\n *\n * The returned outcome reports the round's TOTAL tokens and cost, not the\n * winner's. The runner charges its budget from what an attempt reports, and\n * charging only the winner would make racing look free — three racers would\n * cost what one did, and every budget in the system would be wrong by a factor\n * of N. Racing trades depth for breadth at honest spend, or it is not a trade.\n */\nexport function racedAttempt(options: RacedAttemptOptions): GoalAttempt {\n if (!Number.isSafeInteger(options.racers) || options.racers < 1) {\n throw new TypeError(\"racers must be a positive integer\");\n }\n const select = options.select ?? selectWinner;\n return async (input) => {\n const outcomes = await Promise.all(\n Array.from({ length: options.racers }, async (_unused, index): Promise<RacedOutcome> => {\n const racer = index + 1;\n try {\n return { ...(await options.attempt({ ...input, racer })), racer };\n } catch (cause) {\n return {\n racer, gatePassed: false, feedback: \"\", tokens: 0,\n error: (cause instanceof Error ? cause.message : String(cause)).slice(0, 500),\n };\n }\n }),\n );\n const winner = select(outcomes);\n options.onRound?.(outcomes, winner);\n\n const tokens = outcomes.reduce((total, outcome) => total + outcome.tokens, 0);\n const priced = outcomes.filter((outcome) => outcome.costUsd !== undefined);\n // Only report a cost when EVERY racer priced. A partial sum would understate\n // the round and let an unpriced racer hide spend inside a budget check.\n const costUsd = priced.length === outcomes.length\n ? priced.reduce((total, outcome) => total + (outcome.costUsd ?? 0), 0)\n : undefined;\n\n return {\n gatePassed: winner.gatePassed,\n feedback: winner.feedback,\n tokens,\n ...(costUsd === undefined ? {} : { costUsd }),\n ...(winner.steps === undefined ? {} : { steps: winner.steps }),\n ...(winner.patchBytes === undefined ? {} : { patchBytes: winner.patchBytes }),\n // The round only failed to RUN if every racer did. One survivor is a round.\n ...(outcomes.every((outcome) => outcome.error) ? { error: winner.error } : {}),\n };\n };\n}\n","// Decomposition: split a goal into sub-goals that cannot collide, work them in\n// parallel, integrate them one at a time.\n//\n// The obstacle is specific. The workspace model is single-candidate: every\n// attempt owns one staged tree, and the verifier's contract is \"one patch, from\n// one trusted base, re-applied clean\". It cannot tell you that three partial\n// patches compose. So decomposition here does NOT merge.\n//\n// Instead the partition is made CHECKABLE. A plan declares, per sub-goal, the\n// files it will touch; overlapping declarations are rejected before any agent\n// starts; sub-agents work in isolated trees from the same base; and integration\n// applies each sub-patch to an accumulating tree, gating after each. Parallel\n// exploration, sequential application — no merge algorithm anywhere, and a\n// sub-agent that strays outside its declared files fails its own integration.\n\nimport { validateCodePatch } from \"./code-patch\";\nimport { partition, type PartitionVerdict } from \"@odla-ai/graph\";\nimport { buildCodeGraph, FILE, IMPORTS, READS, WRITES } from \"@odla-ai/graph/code\";\n\n/** One independent slice of a goal, scoped to the files it may touch. */\nexport interface SubGoal {\n id: string;\n /** What this sub-agent is asked to do. */\n goal: string;\n /** The files it declares it will touch. Its patch is rejected if it strays. */\n files: string[];\n}\n\n/** A plan that cannot be run in parallel — overlapping files, or a stray edit. */\nexport class DecompositionError extends Error {\n constructor(message: string) {\n super(message);\n this.name = \"DecompositionError\";\n }\n}\n\n/**\n * Reject a plan that cannot be worked in parallel, BEFORE any agent starts.\n *\n * Two sub-goals declaring the same file is the whole failure mode: they would\n * produce patches against the same lines and one of them could not be applied.\n * Catching it here costs nothing; catching it at integration costs every token\n * both sub-agents spent.\n */\nexport function assertDisjointPlan(plan: readonly SubGoal[]): void {\n if (plan.length === 0) throw new DecompositionError(\"a plan needs at least one sub-goal\");\n const ids = new Set<string>();\n const owner = new Map<string, string>();\n for (const sub of plan) {\n if (!sub.id.trim()) throw new DecompositionError(\"every sub-goal needs an id\");\n if (ids.has(sub.id)) throw new DecompositionError(`duplicate sub-goal id \"${sub.id}\"`);\n ids.add(sub.id);\n if (!sub.goal.trim()) throw new DecompositionError(`sub-goal \"${sub.id}\" has no instruction`);\n if (sub.files.length === 0) {\n throw new DecompositionError(`sub-goal \"${sub.id}\" declares no files; it cannot be checked for collisions`);\n }\n for (const file of sub.files) {\n const claimed = owner.get(file);\n if (claimed !== undefined) {\n throw new DecompositionError(\n `sub-goals \"${claimed}\" and \"${sub.id}\" both declare \"${file}\"; a plan must partition the files it touches`,\n );\n }\n owner.set(file, sub.id);\n }\n }\n}\n\n/** Files a patch actually touches, or null when it is not a valid patch. */\nexport function patchPaths(patch: string, maxBytes = 256 * 1024): string[] | null {\n try { return validateCodePatch(patch, maxBytes); }\n catch { return null; }\n}\n\n/**\n * Check a sub-agent stayed inside what it declared.\n *\n * This is what makes the partition a guarantee rather than a hope: the plan is\n * checked up front, and each patch is checked against the plan. A sub-agent that\n * wandered into a neighbour's file is refused even if its patch would have\n * applied cleanly, because the NEXT sub-patch was written against a tree where\n * that edit does not exist.\n */\nexport function straySubGoalFiles(sub: SubGoal, patch: string): string[] {\n const declared = new Set(sub.files);\n return (patchPaths(patch) ?? []).filter((path) => !declared.has(path));\n}\n\n/** What one sub-goal produced: its patch, its cost, and whether it worked. */\nexport interface SubGoalResult {\n sub: SubGoal;\n /** The candidate patch, or empty when the sub-agent changed nothing. */\n patch: string;\n tokens: number;\n costUsd?: number;\n error?: string;\n}\n\n/** Applying one sub-goal's patch to the accumulating tree, and the verdict. */\nexport interface IntegrationStep {\n subGoalId: string;\n /** Did the accumulated tree still satisfy the proof after applying this one? */\n gatePassed: boolean;\n applied: boolean;\n reason?: \"stray_files\" | \"did_not_apply\" | \"gate_failed\" | \"sub_goal_failed\" | \"no_changes\";\n detail?: string;\n}\n\n/**\n * A decomposed pursuit end to end.\n *\n * Measured at 1.68x the tokens of straight depth for 0.88x the wall clock, so\n * this buys latency, not efficiency — see chooseStrategy for when that trade\n * is worth making.\n */\nexport interface DecomposedRun {\n met: boolean;\n steps: IntegrationStep[];\n tokens: number;\n costUsd?: number;\n}\n\n/** The plan, its results, and the two callbacks that apply and judge them. */\nexport interface IntegrateOptions {\n plan: readonly SubGoal[];\n results: readonly SubGoalResult[];\n /** Apply one patch to the accumulating tree; reject means it did not apply. */\n apply(patch: string): Promise<boolean>;\n /** Run the proof against the accumulating tree. */\n gate(): Promise<boolean>;\n}\n\n/**\n * Integrate sub-results one at a time, gating after each.\n *\n * Order follows the plan, and the first failure stops integration. Continuing\n * past one would mean gating a tree whose earlier layer is already known bad,\n * so every later verdict would describe a state nobody intends to ship.\n */\nexport async function integrateSubGoals(options: IntegrateOptions): Promise<DecomposedRun> {\n const steps: IntegrationStep[] = [];\n let tokens = 0;\n let costUsd = 0;\n let priced = true;\n let met = false;\n\n for (const sub of options.plan) {\n const result = options.results.find((entry) => entry.sub.id === sub.id);\n if (!result) {\n steps.push({ subGoalId: sub.id, gatePassed: false, applied: false, reason: \"sub_goal_failed\", detail: \"no result\" });\n break;\n }\n tokens += result.tokens;\n if (result.costUsd === undefined) priced = false;\n else costUsd += result.costUsd;\n\n if (result.error) {\n steps.push({ subGoalId: sub.id, gatePassed: false, applied: false, reason: \"sub_goal_failed\", detail: result.error });\n break;\n }\n if (!result.patch.trim()) {\n steps.push({ subGoalId: sub.id, gatePassed: false, applied: false, reason: \"no_changes\" });\n break;\n }\n const stray = straySubGoalFiles(sub, result.patch);\n if (stray.length > 0) {\n steps.push({\n subGoalId: sub.id, gatePassed: false, applied: false, reason: \"stray_files\",\n detail: `touched undeclared files: ${stray.join(\", \")}`,\n });\n break;\n }\n if (!(await options.apply(result.patch))) {\n steps.push({ subGoalId: sub.id, gatePassed: false, applied: false, reason: \"did_not_apply\" });\n break;\n }\n const gatePassed = await options.gate();\n steps.push({ subGoalId: sub.id, gatePassed, applied: true, ...(gatePassed ? {} : { reason: \"gate_failed\" as const }) });\n met = gatePassed;\n // A mid-plan gate failure is expected — the goal is only whole once every\n // sub-goal has landed — so integration continues. What stops it is a patch\n // that could not be applied at all, above.\n }\n\n return { met, steps, tokens, ...(priced && options.results.length > 0 ? { costUsd } : {}) };\n}\n\n/**\n * The check `assertDisjointPlan` cannot make: do the sub-goals collide through\n * what they *reach*, not merely through what they declare?\n *\n * Collision ids are graph ids (`file:src/a.ts`, `table:orders`), because the\n * answer is no longer only about files — a shared table is a real collision and\n * naming it as a bare path would be a lie about what it is.\n *\n * Declared files being disjoint is necessary and not sufficient. Two sub-goals\n * can own different modules and still both depend on a third; a change either\n * one makes to that shared module lands in one patch and is invisible to the\n * other's verification. The import graph is the only thing that can see this,\n * and it is cheap — the graph over 3,400 files builds in ~400ms, against the\n * cost of running two agents to completion and discovering it at integration.\n *\n * Reported rather than thrown, because a shared dependency is sometimes fine:\n * two sub-goals may both READ a types module neither intends to touch. The\n * caller decides whether the overlap is one it can live with.\n */\nexport async function planReachCollisions(\n plan: readonly SubGoal[],\n workspace: { paths: readonly string[]; read(path: string): Promise<string> },\n): Promise<PartitionVerdict> {\n const graph = await buildCodeGraph({ paths: workspace.paths, read: workspace.read });\n // Imports AND data, in one traversal. Two sub-goals that never import each\n // other can still both write the same table, and that collision is the one\n // structural analysis alone can never see.\n return partition(graph, plan.map((sub) => sub.files.map((file) => `${FILE}:${file}`)), {\n kinds: [IMPORTS, READS, WRITES], direction: \"out\",\n });\n}\n","// Choosing how to pursue a goal: sequentially, by racing, or by decomposing.\n//\n// The plan that produced this file assumed fan-out would usually win. Measured,\n// it usually loses. On every fixture tried, depth beat breadth on $/solve:\n//\n// racing, equal budget of 6 agent runs (two-stage-gate, gpt-5.6-terra)\n// 1 racer x 6 attempts 7,936 tokens/solve 16.9 s/solve\n// 2 racers x 3 attempts 14,055 tokens/solve 16.8 s/solve\n// 3 racers x 2 attempts 21,044 tokens/solve 19.7 s/solve\n//\n// decomposition (three-modules, gpt-5.6-terra)\n// sequential 6,474 tokens/solve 9.7 s/solve\n// decomposed x3 10,897 tokens/solve 8.5 s/solve\n//\n// One cause explains both: every parallel worker re-pays the orientation cost —\n// reading the project, taking the system prompt, finding its feet — while a\n// sequential run amortizes that across one conversation. Breadth buys wall\n// clock and nothing else.\n//\n// So this chooser is deliberately biased toward sequential, and every fan-out\n// branch has to justify itself against a measured premium.\n\n/** What is known about a goal before choosing how to pursue it. */\nexport interface StrategySignals {\n /** Prior attempts, oldest first. Empty on the first pass. */\n priorAttempts?: Array<{ gatePassed: boolean; feedback: string; error?: string }>;\n /** True when wall clock is the binding constraint — a human is waiting, or a\n * deadline is close — and paying more per solve to finish sooner is correct. */\n latencyBound?: boolean;\n /** Estimated tokens needed to hold the whole goal at once. */\n estimatedContextTokens?: number;\n /** What one agent can actually hold. */\n contextLimit?: number;\n /** A checked disjoint partition, when the planner produced one. */\n partitionable?: boolean;\n /** Racers/sub-agents available if fanning out. */\n width?: number;\n}\n\n/** How to spend a goal's budget: straight depth, N racers, or a split plan. */\nexport type GoalStrategy = \"sequential\" | \"race\" | \"decompose\";\n\n/** The chosen strategy, why, and what it is expected to cost against depth. */\nexport interface StrategyChoice {\n strategy: GoalStrategy;\n /** Why, in terms a human reviewing a run can check. */\n reason: string;\n /** Expected cost multiplier vs sequential, from the measurements above.\n * 1 for sequential; fan-out is charged what it was observed to cost. */\n expectedCostMultiplier: number;\n /** False when the branch rests on something not yet measured. */\n measured: boolean;\n}\n\n/** Observed premiums. Update these from the bench, not from expectation. */\nexport const MEASURED_PREMIUM = Object.freeze({\n /** 3 racers vs pure depth at equal budget: 21,044 / 7,936. */\n racePerRacer: 0.55,\n /** Decomposition across 3 sub-agents: 10,897 / 6,474. */\n decomposePerSubGoal: 0.23,\n});\n\nconst actionable = (feedback: string): boolean => {\n const text = feedback.trim();\n if (text.length < 12) return false;\n // Something the next attempt can act on names a place or an expectation. A\n // bare \"it failed\" leaves depth nothing to build on, which is the one case\n // where sampling another attempt beats trying again.\n return /\\b(expected|assert|error|fail(?:ed|ure)?|exit|line \\d+|\\.[a-z]{1,4}:\\d+)\\b/i.test(text)\n || /\\.(js|ts|tsx|jsx|mjs|cjs|py|go|rs|java|rb)\\b/i.test(text);\n};\n\n/**\n * Choose a strategy.\n *\n * Order matters and encodes the evidence: depth first, because it won every\n * measured comparison; fan-out only where depth demonstrably cannot help.\n */\nexport function chooseStrategy(signals: StrategySignals = {}): StrategyChoice {\n const width = Math.max(1, signals.width ?? 3);\n const prior = signals.priorAttempts ?? [];\n\n // 1. The goal does not fit. Depth cannot help — every attempt starts from the\n // same too-large problem — so this is the one case where decomposition is\n // not a premium but the only option.\n const overflows = signals.estimatedContextTokens !== undefined\n && signals.contextLimit !== undefined\n && signals.estimatedContextTokens > signals.contextLimit;\n if (overflows && signals.partitionable) {\n return {\n strategy: \"decompose\",\n reason: `the goal needs ~${signals.estimatedContextTokens} tokens against a ${signals.contextLimit} limit, and the plan partitions`,\n expectedCostMultiplier: 1 + MEASURED_PREMIUM.decomposePerSubGoal * (width - 1),\n // No fixture this large has been measured. This branch is reasoned, not\n // observed, and says so rather than borrowing the others' credibility.\n measured: false,\n };\n }\n if (overflows) {\n return {\n strategy: \"sequential\",\n reason: \"the goal exceeds one context but no disjoint partition was produced; decomposing without one would collide\",\n expectedCostMultiplier: 1,\n measured: true,\n };\n }\n\n // 2. Attempts are failing with nothing to act on. Re-prompting repeats itself;\n // another sample is the only thing that changes.\n const unhelpful = prior.length >= 2\n && prior.slice(-2).every((attempt) => !attempt.gatePassed && !attempt.error && !actionable(attempt.feedback));\n if (unhelpful) {\n return {\n strategy: \"race\",\n reason: `the last ${Math.min(2, prior.length)} gate failures carried nothing actionable, so depth has nothing to build on`,\n expectedCostMultiplier: 1 + MEASURED_PREMIUM.racePerRacer * (width - 1),\n measured: true,\n };\n }\n\n // 3. Latency is the constraint and the caller has accepted the premium.\n if (signals.latencyBound) {\n const strategy = signals.partitionable ? \"decompose\" : \"race\";\n const premium = strategy === \"decompose\"\n ? MEASURED_PREMIUM.decomposePerSubGoal\n : MEASURED_PREMIUM.racePerRacer;\n return {\n strategy,\n reason: `wall clock is the binding constraint; ${strategy} finishes sooner at a measured premium per solve`,\n expectedCostMultiplier: 1 + premium * (width - 1),\n measured: true,\n };\n }\n\n // 4. Everything else. Depth won every comparison run so far, so it is not a\n // fallback — it is the answer unless something above overrides it.\n return {\n strategy: \"sequential\",\n reason: prior.length === 0\n ? \"no evidence yet favours paying a fan-out premium\"\n : \"the gate is still saying something the next attempt can act on\",\n expectedCostMultiplier: 1,\n measured: true,\n };\n}\n\n/** Whether the gate's output gives the next attempt something to work with. */\nexport const feedbackIsActionable = actionable;\n","// Making installed dependencies available to a recipe, without copying them and\n// without exposing them to the agent.\n//\n// These are three separate questions that were previously answered by one list:\n//\n// 1. can the agent READ or PATCH it? No. `RESERVED` in code-patch already\n// refuses node_modules paths whatever is on disk, so the agent never sees\n// dependencies in `list`, `search`, `read`, or a candidate patch.\n// 2. is it COPIED into every staged tree? No, and it must not be: 70,834\n// files and 1.0GB against a 20,000-file cap, staged at least four times per\n// gated attempt (the attempt, each run_recipe, the verifier's base, and\n// each clean-verify recipe).\n// 3. is it PRESENT when the recipe runs? That is this file. Linked in for\n// the duration of one recipe and removed after, so the tree the verifier\n// diffs and digests is unchanged.\n//\n// The third answer is what makes a real test runner usable as a proof, which in\n// turn is what makes an end-to-end test expressible as a gate at all.\n\nimport { lstat, rm, symlink } from \"node:fs/promises\";\nimport { isAbsolute, join } from \"node:path\";\nimport type { CodeRecipeExecutor, CodeRecipeResult } from \"./code-tool-types\";\n\n/**\n * An installed dependency tree lent to a recipe for the length of one run.\n *\n * Mounted under a reserved name and removed in a finally, so the agent can run\n * a real build without the tree ever being addressable by a patch.\n */\nexport interface RecipeDependencies {\n /** Absolute path to an installed dependency tree on the host. */\n source: string;\n /** Where it appears inside the workspace. Must be a reserved name, so the\n * agent still cannot address it. */\n mountAs?: string;\n}\n\nconst RESERVED_MOUNTS = new Set([\"node_modules\", \"dist\", \"coverage\"]);\n\n/**\n * Wrap an executor so `dependencies.source` is present during each recipe run.\n *\n * It is linked, not copied. A link costs nothing per run, and dependencies are\n * identical across every attempt, racer and verification — copying them would\n * multiply the largest thing in the tree by the number of stages.\n *\n * The link is removed in a `finally`, so a recipe that times out or throws\n * cannot leave it behind for `workspace.patch()` to diff or\n * `digestStagedWorkspace` to hash.\n */\nexport function withRecipeDependencies(\n executor: CodeRecipeExecutor,\n dependencies: RecipeDependencies,\n): CodeRecipeExecutor {\n const mountAs = dependencies.mountAs ?? \"node_modules\";\n if (!isAbsolute(dependencies.source)) {\n throw new TypeError(\"recipe dependency source must be an absolute path\");\n }\n // Mounting at a name the agent CAN address would hand it a writable path into\n // the host's dependency tree through apply_patch.\n if (!RESERVED_MOUNTS.has(mountAs)) {\n throw new TypeError(`recipe dependencies must mount at a reserved name, not \"${mountAs}\"`);\n }\n\n return {\n run: async (input): Promise<CodeRecipeResult> => {\n const target = join(input.workspaceDir, mountAs);\n let linked = false;\n try {\n // An existing entry is left alone: a workspace that already carries its\n // own dependencies is not ours to replace.\n const existing = await lstat(target).catch(() => null);\n if (!existing) {\n await symlink(dependencies.source, target, \"dir\");\n linked = true;\n }\n return await executor.run(input);\n } finally {\n if (linked) await rm(target, { force: true, recursive: false }).catch(() => undefined);\n }\n },\n };\n}\n\n/** Resolve the dependency tree for a repository root, when it has one. */\nexport async function installedDependencies(repoRoot: string): Promise<RecipeDependencies | null> {\n const source = join(repoRoot, \"node_modules\");\n const info = await lstat(source).catch(() => null);\n return info?.isDirectory() ? { source } : null;\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAiBA,eAAe,WAAW,QAAwC;AAChE,QAAM,SAAS,GAAG,OAAO,OAAO,IAAI,OAAO,IAAI,IAAI,OAAO,IAAI;AAC9D,QAAM,QAAQ,MAAM,OAAO,OAAO,OAAO,WAAW,IAAI,YAAY,EAAE,OAAO,MAAM,CAAC;AACpF,SAAO,MAAM,KAAK,IAAI,WAAW,KAAK,GAAG,CAAC,SAAS,KAAK,SAAS,EAAE,EAAE,SAAS,GAAG,GAAG,CAAC,EAClF,KAAK,EAAE,EAAE,MAAM,GAAG,EAAE;AACzB;AAUO,SAAS,mBACd,SAAuC,WACf;AACxB,QAAM,SAAS,QAAQ;AACvB,QAAM,WAAW,QAAQ;AACzB,MAAI,CAAC,UAAU,CAAC,SAAU,QAAO;AACjC,SAAO;AAAA,IACL,QAAQ,OAAO,UAAU,UACtB,MAAM,OAAO,KAAK,SAAS,WAAW,UAAU,KAAK;AAAA,IACxD,UAAU,OAAO,WAA0B;AACzC,YAAM,SAAS,KAAK,SAAS,WAAW;AAAA,QACtC,SAAS,OAAO;AAAA,QAChB,MAAM,OAAO;AAAA,QACb,MAAM,OAAO;AAAA,QACb,GAAI,OAAO,WAAW,EAAE,UAAU,OAAO,SAAS,IAAI,CAAC;AAAA,QACvD,YAAY,MAAM,WAAW,MAAM;AAAA,MACrC,CAAC;AAID,aAAO,EAAE,GAAG,QAAQ,IAAI,IAAI,WAAW,KAAK,IAAI,EAAE;AAAA,IACpD;AAAA,EACF;AACF;;;ACVO,SAAS,cAAc,SAA+B;AAC3D,SAAO,QAAQ,OAAO,QAAQ,aAAa;AAC7C;AAWO,SAAS,cAAc,SAA8B;AAC1D,QAAM,QAAQ,QAAQ,YAAY,SAC9B,iBACA,IAAI,QAAQ,QAAQ,QAAQ,CAAC,CAAC;AAClC,QAAM,QAAQ,GAAG,QAAQ,QAAQ,gBAAgB,QAAQ,OAAO,eAAe,CAAC,YAAY,KAAK;AACjG,MAAI,CAAC,QAAQ,KAAK;AAChB,WAAO,iBAAiB,QAAQ,aAAa,WAAW,KAAK,YAAY,QAAQ,KAAK;AAAA,EACxF;AACA,MAAI,CAAC,QAAQ,UAAU;AAGrB,WAAO,2BAA2B,KAAK,2EACN,QAAQ,KAAK;AAAA,EAChD;AACA,SAAO,kBAAkB,KAAK,iBAAiB,QAAQ,SAAS,IAAI,IAAI,QAAQ,SAAS,GAAG,YAChF,QAAQ,KAAK;AAC3B;AASO,SAAS,cACd,SACA,SACA,UACe;AACf,SAAO;AAAA,IACL;AAAA,IACA,MAAM;AAAA,IACN,MAAM,cAAc,OAAO;AAAA,IAC3B,GAAI,QAAQ,WAAW,EAAE,UAAU,QAAQ,SAAS,IAAI,CAAC;AAAA,IACzD;AAAA,EACF;AACF;;;AC/CO,SAAS,aAAa,UAAwC;AACnE,QAAM,SAAS,CAAC,GAAG,QAAQ,EAAE,KAAK,CAAC,MAAM,UAAU;AACjD,QAAI,KAAK,eAAe,MAAM,WAAY,QAAO,KAAK,aAAa,KAAK;AAGxE,UAAM,aAAa,KAAK,QAAQ,IAAI;AACpC,UAAM,cAAc,MAAM,QAAQ,IAAI;AACtC,QAAI,eAAe,YAAa,QAAO,aAAa;AACpD,UAAM,SAAS,KAAK,SAAS,OAAO,qBAAqB,MAAM,SAAS,OAAO;AAC/E,QAAI,UAAU,EAAG,QAAO;AAKxB,UAAM,aAAa,KAAK,YAAY;AACpC,UAAM,cAAc,MAAM,YAAY;AACtC,QAAI,eAAe,YAAa,QAAO,aAAa,KAAK;AACzD,QAAI,cAAc,eAAe,KAAK,YAAY,MAAM,SAAS;AAC/D,aAAO,KAAK,UAAW,MAAM;AAAA,IAC/B;AACA,YAAQ,KAAK,cAAc,OAAO,qBAAqB,MAAM,cAAc,OAAO;AAAA,EACpF,CAAC;AACD,SAAO,OAAO,CAAC;AACjB;AAWO,SAAS,aAAa,SAA2C;AACtE,MAAI,CAAC,OAAO,cAAc,QAAQ,MAAM,KAAK,QAAQ,SAAS,GAAG;AAC/D,UAAM,IAAI,UAAU,mCAAmC;AAAA,EACzD;AACA,QAAM,SAAS,QAAQ,UAAU;AACjC,SAAO,OAAO,UAAU;AACtB,UAAM,WAAW,MAAM,QAAQ;AAAA,MAC7B,MAAM,KAAK,EAAE,QAAQ,QAAQ,OAAO,GAAG,OAAO,SAAS,UAAiC;AACtF,cAAM,QAAQ,QAAQ;AACtB,YAAI;AACF,iBAAO,EAAE,GAAI,MAAM,QAAQ,QAAQ,EAAE,GAAG,OAAO,MAAM,CAAC,GAAI,MAAM;AAAA,QAClE,SAAS,OAAO;AACd,iBAAO;AAAA,YACL;AAAA,YAAO,YAAY;AAAA,YAAO,UAAU;AAAA,YAAI,QAAQ;AAAA,YAChD,QAAQ,iBAAiB,QAAQ,MAAM,UAAU,OAAO,KAAK,GAAG,MAAM,GAAG,GAAG;AAAA,UAC9E;AAAA,QACF;AAAA,MACF,CAAC;AAAA,IACH;AACA,UAAM,SAAS,OAAO,QAAQ;AAC9B,YAAQ,UAAU,UAAU,MAAM;AAElC,UAAM,SAAS,SAAS,OAAO,CAAC,OAAO,YAAY,QAAQ,QAAQ,QAAQ,CAAC;AAC5E,UAAM,SAAS,SAAS,OAAO,CAAC,YAAY,QAAQ,YAAY,MAAS;AAGzE,UAAM,UAAU,OAAO,WAAW,SAAS,SACvC,OAAO,OAAO,CAAC,OAAO,YAAY,SAAS,QAAQ,WAAW,IAAI,CAAC,IACnE;AAEJ,WAAO;AAAA,MACL,YAAY,OAAO;AAAA,MACnB,UAAU,OAAO;AAAA,MACjB;AAAA,MACA,GAAI,YAAY,SAAY,CAAC,IAAI,EAAE,QAAQ;AAAA,MAC3C,GAAI,OAAO,UAAU,SAAY,CAAC,IAAI,EAAE,OAAO,OAAO,MAAM;AAAA,MAC5D,GAAI,OAAO,eAAe,SAAY,CAAC,IAAI,EAAE,YAAY,OAAO,WAAW;AAAA;AAAA,MAE3E,GAAI,SAAS,MAAM,CAAC,YAAY,QAAQ,KAAK,IAAI,EAAE,OAAO,OAAO,MAAM,IAAI,CAAC;AAAA,IAC9E;AAAA,EACF;AACF;;;AC3GA,SAAS,iBAAwC;AACjD,SAAS,gBAAgB,MAAM,SAAS,OAAO,cAAc;AAYtD,IAAM,qBAAN,cAAiC,MAAM;AAAA,EAC5C,YAAY,SAAiB;AAC3B,UAAM,OAAO;AACb,SAAK,OAAO;AAAA,EACd;AACF;AAUO,SAAS,mBAAmB,MAAgC;AACjE,MAAI,KAAK,WAAW,EAAG,OAAM,IAAI,mBAAmB,oCAAoC;AACxF,QAAM,MAAM,oBAAI,IAAY;AAC5B,QAAM,QAAQ,oBAAI,IAAoB;AACtC,aAAW,OAAO,MAAM;AACtB,QAAI,CAAC,IAAI,GAAG,KAAK,EAAG,OAAM,IAAI,mBAAmB,4BAA4B;AAC7E,QAAI,IAAI,IAAI,IAAI,EAAE,EAAG,OAAM,IAAI,mBAAmB,0BAA0B,IAAI,EAAE,GAAG;AACrF,QAAI,IAAI,IAAI,EAAE;AACd,QAAI,CAAC,IAAI,KAAK,KAAK,EAAG,OAAM,IAAI,mBAAmB,aAAa,IAAI,EAAE,sBAAsB;AAC5F,QAAI,IAAI,MAAM,WAAW,GAAG;AAC1B,YAAM,IAAI,mBAAmB,aAAa,IAAI,EAAE,0DAA0D;AAAA,IAC5G;AACA,eAAW,QAAQ,IAAI,OAAO;AAC5B,YAAM,UAAU,MAAM,IAAI,IAAI;AAC9B,UAAI,YAAY,QAAW;AACzB,cAAM,IAAI;AAAA,UACR,cAAc,OAAO,UAAU,IAAI,EAAE,mBAAmB,IAAI;AAAA,QAC9D;AAAA,MACF;AACA,YAAM,IAAI,MAAM,IAAI,EAAE;AAAA,IACxB;AAAA,EACF;AACF;AAGO,SAAS,WAAW,OAAe,WAAW,MAAM,MAAuB;AAChF,MAAI;AAAE,WAAO,kBAAkB,OAAO,QAAQ;AAAA,EAAG,QAC3C;AAAE,WAAO;AAAA,EAAM;AACvB;AAWO,SAAS,kBAAkB,KAAc,OAAyB;AACvE,QAAM,WAAW,IAAI,IAAI,IAAI,KAAK;AAClC,UAAQ,WAAW,KAAK,KAAK,CAAC,GAAG,OAAO,CAAC,SAAS,CAAC,SAAS,IAAI,IAAI,CAAC;AACvE;AAqDA,eAAsB,kBAAkB,SAAmD;AACzF,QAAM,QAA2B,CAAC;AAClC,MAAI,SAAS;AACb,MAAI,UAAU;AACd,MAAI,SAAS;AACb,MAAI,MAAM;AAEV,aAAW,OAAO,QAAQ,MAAM;AAC9B,UAAM,SAAS,QAAQ,QAAQ,KAAK,CAAC,UAAU,MAAM,IAAI,OAAO,IAAI,EAAE;AACtE,QAAI,CAAC,QAAQ;AACX,YAAM,KAAK,EAAE,WAAW,IAAI,IAAI,YAAY,OAAO,SAAS,OAAO,QAAQ,mBAAmB,QAAQ,YAAY,CAAC;AACnH;AAAA,IACF;AACA,cAAU,OAAO;AACjB,QAAI,OAAO,YAAY,OAAW,UAAS;AAAA,QACtC,YAAW,OAAO;AAEvB,QAAI,OAAO,OAAO;AAChB,YAAM,KAAK,EAAE,WAAW,IAAI,IAAI,YAAY,OAAO,SAAS,OAAO,QAAQ,mBAAmB,QAAQ,OAAO,MAAM,CAAC;AACpH;AAAA,IACF;AACA,QAAI,CAAC,OAAO,MAAM,KAAK,GAAG;AACxB,YAAM,KAAK,EAAE,WAAW,IAAI,IAAI,YAAY,OAAO,SAAS,OAAO,QAAQ,aAAa,CAAC;AACzF;AAAA,IACF;AACA,UAAM,QAAQ,kBAAkB,KAAK,OAAO,KAAK;AACjD,QAAI,MAAM,SAAS,GAAG;AACpB,YAAM,KAAK;AAAA,QACT,WAAW,IAAI;AAAA,QAAI,YAAY;AAAA,QAAO,SAAS;AAAA,QAAO,QAAQ;AAAA,QAC9D,QAAQ,6BAA6B,MAAM,KAAK,IAAI,CAAC;AAAA,MACvD,CAAC;AACD;AAAA,IACF;AACA,QAAI,CAAE,MAAM,QAAQ,MAAM,OAAO,KAAK,GAAI;AACxC,YAAM,KAAK,EAAE,WAAW,IAAI,IAAI,YAAY,OAAO,SAAS,OAAO,QAAQ,gBAAgB,CAAC;AAC5F;AAAA,IACF;AACA,UAAM,aAAa,MAAM,QAAQ,KAAK;AACtC,UAAM,KAAK,EAAE,WAAW,IAAI,IAAI,YAAY,SAAS,MAAM,GAAI,aAAa,CAAC,IAAI,EAAE,QAAQ,cAAuB,EAAG,CAAC;AACtH,UAAM;AAAA,EAIR;AAEA,SAAO,EAAE,KAAK,OAAO,QAAQ,GAAI,UAAU,QAAQ,QAAQ,SAAS,IAAI,EAAE,QAAQ,IAAI,CAAC,EAAG;AAC5F;AAqBA,eAAsB,oBACpB,MACA,WAC2B;AAC3B,QAAM,QAAQ,MAAM,eAAe,EAAE,OAAO,UAAU,OAAO,MAAM,UAAU,KAAK,CAAC;AAInF,SAAO,UAAU,OAAO,KAAK,IAAI,CAAC,QAAQ,IAAI,MAAM,IAAI,CAAC,SAAS,GAAG,IAAI,IAAI,IAAI,EAAE,CAAC,GAAG;AAAA,IACrF,OAAO,CAAC,SAAS,OAAO,MAAM;AAAA,IAAG,WAAW;AAAA,EAC9C,CAAC;AACH;;;AClKO,IAAM,mBAAmB,OAAO,OAAO;AAAA;AAAA,EAE5C,cAAc;AAAA;AAAA,EAEd,qBAAqB;AACvB,CAAC;AAED,IAAM,aAAa,CAAC,aAA8B;AAChD,QAAM,OAAO,SAAS,KAAK;AAC3B,MAAI,KAAK,SAAS,GAAI,QAAO;AAI7B,SAAO,8EAA8E,KAAK,IAAI,KACzF,gDAAgD,KAAK,IAAI;AAChE;AAQO,SAAS,eAAe,UAA2B,CAAC,GAAmB;AAC5E,QAAM,QAAQ,KAAK,IAAI,GAAG,QAAQ,SAAS,CAAC;AAC5C,QAAM,QAAQ,QAAQ,iBAAiB,CAAC;AAKxC,QAAM,YAAY,QAAQ,2BAA2B,UAChD,QAAQ,iBAAiB,UACzB,QAAQ,yBAAyB,QAAQ;AAC9C,MAAI,aAAa,QAAQ,eAAe;AACtC,WAAO;AAAA,MACL,UAAU;AAAA,MACV,QAAQ,mBAAmB,QAAQ,sBAAsB,qBAAqB,QAAQ,YAAY;AAAA,MAClG,wBAAwB,IAAI,iBAAiB,uBAAuB,QAAQ;AAAA;AAAA;AAAA,MAG5E,UAAU;AAAA,IACZ;AAAA,EACF;AACA,MAAI,WAAW;AACb,WAAO;AAAA,MACL,UAAU;AAAA,MACV,QAAQ;AAAA,MACR,wBAAwB;AAAA,MACxB,UAAU;AAAA,IACZ;AAAA,EACF;AAIA,QAAM,YAAY,MAAM,UAAU,KAC7B,MAAM,MAAM,EAAE,EAAE,MAAM,CAAC,YAAY,CAAC,QAAQ,cAAc,CAAC,QAAQ,SAAS,CAAC,WAAW,QAAQ,QAAQ,CAAC;AAC9G,MAAI,WAAW;AACb,WAAO;AAAA,MACL,UAAU;AAAA,MACV,QAAQ,YAAY,KAAK,IAAI,GAAG,MAAM,MAAM,CAAC;AAAA,MAC7C,wBAAwB,IAAI,iBAAiB,gBAAgB,QAAQ;AAAA,MACrE,UAAU;AAAA,IACZ;AAAA,EACF;AAGA,MAAI,QAAQ,cAAc;AACxB,UAAM,WAAW,QAAQ,gBAAgB,cAAc;AACvD,UAAM,UAAU,aAAa,cACzB,iBAAiB,sBACjB,iBAAiB;AACrB,WAAO;AAAA,MACL;AAAA,MACA,QAAQ,yCAAyC,QAAQ;AAAA,MACzD,wBAAwB,IAAI,WAAW,QAAQ;AAAA,MAC/C,UAAU;AAAA,IACZ;AAAA,EACF;AAIA,SAAO;AAAA,IACL,UAAU;AAAA,IACV,QAAQ,MAAM,WAAW,IACrB,qDACA;AAAA,IACJ,wBAAwB;AAAA,IACxB,UAAU;AAAA,EACZ;AACF;AAGO,IAAM,uBAAuB;;;AChIpC,SAAS,OAAO,IAAI,eAAe;AACnC,SAAS,YAAY,YAAY;AAiBjC,IAAM,kBAAkB,oBAAI,IAAI,CAAC,gBAAgB,QAAQ,UAAU,CAAC;AAa7D,SAAS,uBACd,UACA,cACoB;AACpB,QAAM,UAAU,aAAa,WAAW;AACxC,MAAI,CAAC,WAAW,aAAa,MAAM,GAAG;AACpC,UAAM,IAAI,UAAU,mDAAmD;AAAA,EACzE;AAGA,MAAI,CAAC,gBAAgB,IAAI,OAAO,GAAG;AACjC,UAAM,IAAI,UAAU,2DAA2D,OAAO,GAAG;AAAA,EAC3F;AAEA,SAAO;AAAA,IACL,KAAK,OAAO,UAAqC;AAC/C,YAAM,SAAS,KAAK,MAAM,cAAc,OAAO;AAC/C,UAAI,SAAS;AACb,UAAI;AAGF,cAAM,WAAW,MAAM,MAAM,MAAM,EAAE,MAAM,MAAM,IAAI;AACrD,YAAI,CAAC,UAAU;AACb,gBAAM,QAAQ,aAAa,QAAQ,QAAQ,KAAK;AAChD,mBAAS;AAAA,QACX;AACA,eAAO,MAAM,SAAS,IAAI,KAAK;AAAA,MACjC,UAAE;AACA,YAAI,OAAQ,OAAM,GAAG,QAAQ,EAAE,OAAO,MAAM,WAAW,MAAM,CAAC,EAAE,MAAM,MAAM,MAAS;AAAA,MACvF;AAAA,IACF;AAAA,EACF;AACF;AAGA,eAAsB,sBAAsB,UAAsD;AAChG,QAAM,SAAS,KAAK,UAAU,cAAc;AAC5C,QAAM,OAAO,MAAM,MAAM,MAAM,EAAE,MAAM,MAAM,IAAI;AACjD,SAAO,MAAM,YAAY,IAAI,EAAE,OAAO,IAAI;AAC5C;","names":[]}
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@odla-ai/harness",
3
- "version": "0.8.1",
3
+ "version": "0.9.0",
4
4
  "description": "Safe, inspectable coding-task protocol and credentialless container runner for odla Studio.",
5
5
  "license": "MIT",
6
6
  "homepage": "https://odla.ai/docs/packages/harness",
@@ -65,7 +65,7 @@
65
65
  "test:live": "ODLA_CODE_LIVE_TEST=1 vitest run test/code-runtime-live.integration.test.ts"
66
66
  },
67
67
  "dependencies": {
68
- "@odla-ai/ai": "*",
68
+ "@odla-ai/ai": ">=0.17.0 <1.0.0",
69
69
  "@odla-ai/camel": "*",
70
70
  "@odla-ai/graph": "*"
71
71
  },