@tea-agent/loop-agent 0.44.0-next.8 → 0.44.0-next.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +97 -0
- package/dist/application/evaluation/budget.js +21 -0
- package/dist/application/evaluation/corpus-hash.js +10 -15
- package/dist/application/evaluation/corpus.js +2 -1
- package/dist/application/evaluation/frontend-browser-acceptance.js +77 -0
- package/dist/application/evaluation/frontend-corpus-browser-acceptance.js +175 -0
- package/dist/application/evaluation/frontend-gateway-evidence.js +226 -0
- package/dist/application/evaluation/frontend-pair-registration.js +143 -0
- package/dist/application/evaluation/frontend-paired-summary.js +150 -0
- package/dist/application/evaluation/frontend-run-observation.js +144 -0
- package/dist/application/evaluation/frontend-shared-evidence.js +105 -0
- package/dist/application/evaluation/types.js +45 -12
- package/dist/application/task-lifecycle/observe.js +26 -1
- package/dist/build-stamp.json +3 -3
- package/dist/cli/command-definitions.js +11 -0
- package/dist/cli/program.js +4 -0
- package/dist/commands/task-advance.js +3 -5
- package/dist/commands/task-source-prepare.js +13 -1
- package/dist/executors/dag-pi/tools/design-terminal-tools.js +17 -7
- package/dist/executors/dag-pi/tools/review-terminal-tools.js +17 -7
- package/dist/executors/dag-pi-executor.js +2048 -125
- package/dist/executors/pi-executor.js +6 -1
- package/dist/executors/pi-sdk-executor.js +76 -11
- package/dist/executors/shell-executor.js +88 -12
- package/dist/infrastructure/console/operation-store.js +2 -0
- package/dist/infrastructure/evaluation/corpus-store.js +37 -12
- package/dist/shared/operator/capabilities.js +10 -5
- package/dist/task/config-types.js +22 -9
- package/dist/task/contract/project.js +13 -0
- package/dist/task/contract/schema.js +2 -8
- package/dist/task/source-prepare/parse-intent.js +124 -5
- package/dist/task/source-prepare/prepare.js +6 -1
- package/dist/task/source-prepare/semantic-intake.js +17 -2
- package/dist/task/source-prepare/source-execution.js +65 -0
- package/dist/task/source-prepare/source-fidelity-pi.js +21 -8
- package/dist/task/source-prepare/source-provider-budget.js +290 -0
- package/dist/worker/console/operator-actions.js +0 -1
- package/dist/worker/console/operator-user-error.js +1 -1
- package/dist/worker/console/prd-intake-bridge.js +5 -15
- package/dist/workflows/dag/budget-enforcement.js +31 -2
- package/dist/workflows/dag/frontend-closeout.js +3 -1
- package/dist/workflows/dag/frontend-contract-facts.js +11 -8
- package/dist/workflows/dag/frontend-design-policy.js +6 -6
- package/dist/workflows/dag/frontend-durable-tools.js +90 -70
- package/dist/workflows/dag/frontend-implementation-contract.js +146 -59
- package/dist/workflows/dag/frontend-plan-canary.js +66 -16
- package/dist/workflows/dag/frontend-plan-decision-contract.js +13 -21
- package/dist/workflows/dag/frontend-plan-progress.js +92 -0
- package/dist/workflows/dag/frontend-recovery-plan.js +11 -10
- package/dist/workflows/dag/frontend-recovery-run.js +51 -46
- package/dist/workflows/dag/frontend-repair-assertions.js +36 -0
- package/dist/workflows/dag/frontend-review-context.js +29 -1
- package/dist/workflows/dag/frontend-review-scopes.js +92 -2
- package/dist/workflows/dag/frontend-risk.js +8 -3
- package/dist/workflows/dag/frontend-root-observation.js +261 -0
- package/dist/workflows/dag/frontend-session-budget.js +52 -39
- package/dist/workflows/dag/frontend-session-context.js +10 -0
- package/dist/workflows/dag/frontend-test-execution-evidence.js +206 -48
- package/dist/workflows/dag/frontend-typed-event-store.js +11 -6
- package/dist/workflows/dag/frontend-verification-trace.js +6 -0
- package/dist/workflows/dag/frontend-writer-admission.js +2 -1
- package/dist/workflows/dag/init-hybrid.js +108 -37
- package/dist/workflows/dag/node-execution.js +11 -11
- package/dist/workflows/dag/rerun-feedback.js +35 -4
- package/dist/workflows/dag/rerun-task.js +62 -67
- package/dist/workflows/dag/runner.js +58 -12
- package/dist/workflows/dag/types.js +15 -8
- package/docs/architecture/runtime-boundaries.md +1 -1
- package/docs/templates/backend-test-dag.json +4 -4
- package/docs/templates/frontend-implementation-contract.schema.json +31 -11
- package/package.json +1 -1
- package/skills/frontend-design-review/SKILL.md +8 -11
- package/skills/frontend-design-review/references/review-checklist.md +3 -4
- package/skills/frontend-review/SKILL.md +5 -1
- package/skills/loop-agent/references/command-reference.md +8 -0
|
@@ -790,7 +790,7 @@ export async function executePiStep(options) {
|
|
|
790
790
|
reuseRuntimeActive,
|
|
791
791
|
});
|
|
792
792
|
}
|
|
793
|
-
if (options.reserveProviderRequest || !shouldFallbackSdkToCli(sdkResult, validationIssues, options)) {
|
|
793
|
+
if (options.frontendPlanControl || options.reserveProviderRequest || !shouldFallbackSdkToCli(sdkResult, validationIssues, options)) {
|
|
794
794
|
const failureCategory = validationIssues.length > 0
|
|
795
795
|
? "invalid-output"
|
|
796
796
|
: sdkResult.failureCategory;
|
|
@@ -1104,6 +1104,11 @@ export function classifyPiFailure(input) {
|
|
|
1104
1104
|
// Provider/transport verdicts read error-shaped channels only; the raw stdout
|
|
1105
1105
|
// stream embeds assistant text, tool arguments and findings verbatim.
|
|
1106
1106
|
const combined = `${input.stderr}\n${extractProviderFailureText(input.stdout)}`.toLowerCase();
|
|
1107
|
+
// An explicit spend-budget stop is terminal even if a transport wrapper also
|
|
1108
|
+
// reports a closed connection. Read only the error channels above, never
|
|
1109
|
+
// assistant/tool prose; generic quota and context limits keep their categories.
|
|
1110
|
+
if (/(?<![\w-])(?:frontend_provider_budget_exhausted|budget_exhausted|budget_breach|budget exhausted)(?![\w-])/.test(combined))
|
|
1111
|
+
return "budget_breach";
|
|
1107
1112
|
// Context overflow (400 request-too-large / context window exceeded).
|
|
1108
1113
|
// Must be detected BEFORE the generic rate-limit/quota phrases: an overflow
|
|
1109
1114
|
// error often contains "too many tokens" / "maximum context" which would
|
|
@@ -36,7 +36,7 @@ export function createDagWizardLocalTelemetryHeadersExtension(telemetrySessionId
|
|
|
36
36
|
};
|
|
37
37
|
}
|
|
38
38
|
/** Wrap the actual stream seam: tool turns, retries and compaction all await it. */
|
|
39
|
-
export async function installPiProviderRequestBudget(session, reserve, onFailure, capacity) {
|
|
39
|
+
export async function installPiProviderRequestBudget(session, reserve, onFailure, capacity, plan) {
|
|
40
40
|
const agent = session.agent;
|
|
41
41
|
const original = agent?.streamFunction;
|
|
42
42
|
if (!agent || typeof original !== "function")
|
|
@@ -46,6 +46,9 @@ export async function installPiProviderRequestBudget(session, reserve, onFailure
|
|
|
46
46
|
try {
|
|
47
47
|
if (args[2]?.signal?.aborted)
|
|
48
48
|
throw new Error("FRONTEND_PROVIDER_REQUEST_ABORTED");
|
|
49
|
+
const protocolStop = plan?.readStop();
|
|
50
|
+
if (protocolStop)
|
|
51
|
+
throw Object.assign(new Error(`${protocolStop.code}: ${protocolStop.error}`), { frontendPlanStop: true });
|
|
49
52
|
if (capacity) {
|
|
50
53
|
const measurement = measureFrontendRequestEnvelope(args[1], args[0], capacity.policy);
|
|
51
54
|
capacity.observe(measurement);
|
|
@@ -55,10 +58,16 @@ export async function installPiProviderRequestBudget(session, reserve, onFailure
|
|
|
55
58
|
await reserve();
|
|
56
59
|
if (args[2]?.signal?.aborted)
|
|
57
60
|
throw new Error("FRONTEND_PROVIDER_REQUEST_ABORTED");
|
|
61
|
+
const stoppedDuringReservation = plan?.readStop();
|
|
62
|
+
if (stoppedDuringReservation)
|
|
63
|
+
throw Object.assign(new Error(`${stoppedDuringReservation.code}: ${stoppedDuringReservation.error}`), { frontendPlanStop: true });
|
|
58
64
|
}
|
|
59
65
|
catch (error) {
|
|
60
66
|
const message = error instanceof Error ? error.message : String(error);
|
|
61
|
-
|
|
67
|
+
if (error.frontendPlanStop)
|
|
68
|
+
plan?.onStop();
|
|
69
|
+
else
|
|
70
|
+
onFailure(message);
|
|
62
71
|
const stream = createAssistantMessageEventStream();
|
|
63
72
|
const model = args[0] ?? {};
|
|
64
73
|
stream.push({ type: "error", reason: "error", error: { role: "assistant", content: [], api: model.api ?? "unknown", provider: model.provider ?? "unknown", model: model.id ?? "unknown", usage: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, totalTokens: 0, cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 } }, stopReason: "error", errorMessage: message, timestamp: Date.now() } });
|
|
@@ -650,6 +659,34 @@ async function executeSingleSdkAttemptInternal(options) {
|
|
|
650
659
|
let terminationConfirmed = true;
|
|
651
660
|
let stderr = "";
|
|
652
661
|
let providerBudgetFailure;
|
|
662
|
+
let providerBudgetAbort = false;
|
|
663
|
+
let providerBudgetStopRequested = false;
|
|
664
|
+
let resolveProviderBudgetStop;
|
|
665
|
+
const providerBudgetStopPromise = new Promise(resolve => { resolveProviderBudgetStop = resolve; });
|
|
666
|
+
const frontendBudgetEnabled = Boolean(options.reserveProviderRequest || options.frontendExecutionPolicy || options.frontendPlanControl);
|
|
667
|
+
const observeProviderBudgetError = (message) => {
|
|
668
|
+
if (!frontendBudgetEnabled || typeof message !== "string" || providerBudgetStopRequested)
|
|
669
|
+
return;
|
|
670
|
+
if (classifyPiFailure({ assistantText: "", exitCode: 1, outputTooLarge: false, stderr: message, stdout: "", timedOut: false }) !== "budget_breach")
|
|
671
|
+
return;
|
|
672
|
+
providerBudgetFailure = message;
|
|
673
|
+
providerBudgetStopRequested = true;
|
|
674
|
+
if (!stderr.includes(message))
|
|
675
|
+
appendStderr(message);
|
|
676
|
+
resolveProviderBudgetStop("provider-budget");
|
|
677
|
+
};
|
|
678
|
+
let protocolStop;
|
|
679
|
+
let protocolAbort = false;
|
|
680
|
+
let unsubscribePlan;
|
|
681
|
+
let resolveProtocolStop;
|
|
682
|
+
const protocolStopPromise = new Promise(resolve => { resolveProtocolStop = resolve; });
|
|
683
|
+
const requestProtocolStop = () => {
|
|
684
|
+
const stop = options.frontendPlanControl?.readStop();
|
|
685
|
+
if (stop && stopReason !== "aborted" && !providerBudgetFailure) {
|
|
686
|
+
protocolStop = stop;
|
|
687
|
+
resolveProtocolStop("protocol-stop");
|
|
688
|
+
}
|
|
689
|
+
};
|
|
653
690
|
let capabilitiesObservation = frontendModelCapabilities(undefined);
|
|
654
691
|
const requestEnvelopeObservation = { peakBytes: 0, peakEstimatedTokens: 0, estimatorVersion: "utf8-bytes-div3-v1-estimate" };
|
|
655
692
|
const stdoutPreview = new BoundedTextPreview("stdout");
|
|
@@ -758,12 +795,14 @@ async function executeSingleSdkAttemptInternal(options) {
|
|
|
758
795
|
: {}),
|
|
759
796
|
});
|
|
760
797
|
capabilitiesObservation = frontendModelCapabilities(session.agent?.state?.model);
|
|
761
|
-
if (options.reserveProviderRequest || options.frontendExecutionPolicy)
|
|
762
|
-
await installPiProviderRequestBudget(session, options.reserveProviderRequest ?? (async () => { }), message => {
|
|
798
|
+
if (options.frontendPlanControl || options.reserveProviderRequest || options.frontendExecutionPolicy)
|
|
799
|
+
await installPiProviderRequestBudget(session, options.reserveProviderRequest ?? (async () => { }), message => { if (protocolAbort && message.includes("REQUEST_ABORTED"))
|
|
800
|
+
return; providerBudgetFailure = message; appendStderr(message); observeProviderBudgetError(message); }, options.frontendExecutionPolicy ? { policy: options.frontendExecutionPolicy, observe: measurement => {
|
|
763
801
|
capabilitiesObservation = measurement.capabilities;
|
|
764
802
|
requestEnvelopeObservation.peakBytes = Math.max(requestEnvelopeObservation.peakBytes, measurement.bytes);
|
|
765
803
|
requestEnvelopeObservation.peakEstimatedTokens = Math.max(requestEnvelopeObservation.peakEstimatedTokens, measurement.estimatedTokens);
|
|
766
|
-
} } : undefined);
|
|
804
|
+
} } : undefined, options.frontendPlanControl ? { readStop: options.frontendPlanControl.readStop, onStop: requestProtocolStop } : undefined);
|
|
805
|
+
unsubscribePlan = options.frontendPlanControl?.subscribe(requestProtocolStop);
|
|
767
806
|
let resolveStall;
|
|
768
807
|
let resolveSettlement;
|
|
769
808
|
let resolveContextBudget;
|
|
@@ -844,13 +883,23 @@ async function executeSingleSdkAttemptInternal(options) {
|
|
|
844
883
|
}
|
|
845
884
|
}
|
|
846
885
|
const observationMessage = isRecord(event.message) ? event.message : undefined;
|
|
847
|
-
|
|
886
|
+
observeProviderBudgetError(observationMessage?.errorMessage ?? (event.type === "auto_retry_start" ? event.errorMessage : undefined));
|
|
887
|
+
if (providerBudgetStopRequested && event.type === "auto_retry_start") {
|
|
888
|
+
// Pi emits the notification before installing its retry controller.
|
|
889
|
+
// Cancel after that synchronous setup, including when abort() already
|
|
890
|
+
// ran on message_end and is still waiting for the session to idle.
|
|
891
|
+
queueMicrotask(() => { try {
|
|
892
|
+
session?.abortRetry?.();
|
|
893
|
+
}
|
|
894
|
+
catch { /* abort/dispose confirmation remains authoritative */ } });
|
|
895
|
+
}
|
|
896
|
+
if (typeof observationMessage?.errorMessage === "string" && !(protocolStop && observationMessage.errorMessage === `${protocolStop.code}: ${protocolStop.error}`))
|
|
848
897
|
providerErrorMessage = observationMessage.errorMessage;
|
|
849
898
|
try {
|
|
850
899
|
readObservation.observe(event);
|
|
851
900
|
}
|
|
852
901
|
catch { /* observation must not change execution */ }
|
|
853
|
-
if (!(typeof observationMessage?.errorMessage === "string" && observationMessage.errorMessage
|
|
902
|
+
if (!(typeof observationMessage?.errorMessage === "string" && /^(FRONTEND_PROVIDER_|FRONTEND_PLAN_NO_PROGRESS|FRONTEND_CONTEXT_ESTIMATE_EXCEEDED)/.test(observationMessage.errorMessage)) && (event.usage || observationMessage?.usage || event.tokenUsage))
|
|
854
903
|
observationEvents.push({ responseId: event.responseId, messageId: event.messageId, usage: event.usage, tokenUsage: event.tokenUsage, message: observationMessage ? { role: observationMessage.role, id: observationMessage.id, responseId: observationMessage.responseId, usage: observationMessage.usage } : undefined });
|
|
855
904
|
const usageSample = extractSdkUsageSample(event);
|
|
856
905
|
if (usageSample)
|
|
@@ -898,8 +947,10 @@ async function executeSingleSdkAttemptInternal(options) {
|
|
|
898
947
|
promptPromise.then(() => "done"),
|
|
899
948
|
settlementPromise,
|
|
900
949
|
...(timeoutPromise ? [timeoutPromise] : []),
|
|
950
|
+
...(frontendBudgetEnabled ? [providerBudgetStopPromise] : []),
|
|
901
951
|
...(stallPromise ? [stallPromise] : []),
|
|
902
952
|
...(contextBudgetPromise ? [contextBudgetPromise] : []),
|
|
953
|
+
...(options.frontendPlanControl ? [protocolStopPromise] : []),
|
|
903
954
|
]);
|
|
904
955
|
};
|
|
905
956
|
const disposeSession = async () => {
|
|
@@ -928,7 +979,18 @@ async function executeSingleSdkAttemptInternal(options) {
|
|
|
928
979
|
let continuesSent = 0;
|
|
929
980
|
let raced = await runTurn(promptMessage);
|
|
930
981
|
while (true) {
|
|
931
|
-
if (
|
|
982
|
+
if (providerBudgetStopRequested) {
|
|
983
|
+
providerBudgetAbort = true;
|
|
984
|
+
abortConfirmed = await runSessionActionWithGrace("abort", () => session.abort());
|
|
985
|
+
await disposeSession();
|
|
986
|
+
}
|
|
987
|
+
else if ((raced === "protocol-stop" || ((raced === "done" || raced === "settled") && protocolStop)) && !providerBudgetFailure && !contextBudgetExhausted && stopReason !== "aborted") {
|
|
988
|
+
protocolAbort = true;
|
|
989
|
+
appendStderr(`${protocolStop.code}: ${protocolStop.error}; fingerprint=${protocolStop.fingerprint}`);
|
|
990
|
+
abortConfirmed = await runSessionActionWithGrace("abort", () => session.abort());
|
|
991
|
+
await disposeSession();
|
|
992
|
+
}
|
|
993
|
+
else if (raced === "context-budget" || contextBudgetExhausted) {
|
|
932
994
|
appendStderr(contextBudgetDetail);
|
|
933
995
|
abortConfirmed = await runSessionActionWithGrace("abort", () => session.abort());
|
|
934
996
|
await disposeSession();
|
|
@@ -950,6 +1012,8 @@ async function executeSingleSdkAttemptInternal(options) {
|
|
|
950
1012
|
maxContinues: PI_SDK_STALL_CONTINUE_MAX,
|
|
951
1013
|
});
|
|
952
1014
|
raced = await runTurn(PI_SDK_STALL_CONTINUE_PROMPT);
|
|
1015
|
+
if (raced === "protocol-stop" || raced === "context-budget" || raced === "provider-budget")
|
|
1016
|
+
continue;
|
|
953
1017
|
if (raced !== "done" && raced !== "settled") {
|
|
954
1018
|
await failClosedAfterWait(raced === "stall" ? "stall" : "absolute-timeout", raced === "stall"
|
|
955
1019
|
? "pi SDK stall continue did not recover; treating as timeout"
|
|
@@ -990,6 +1054,7 @@ async function executeSingleSdkAttemptInternal(options) {
|
|
|
990
1054
|
if (stallHandle)
|
|
991
1055
|
clearTimeout(stallHandle);
|
|
992
1056
|
unsubscribe?.();
|
|
1057
|
+
unsubscribePlan?.();
|
|
993
1058
|
if (session && !disposeAttempted) {
|
|
994
1059
|
disposeAttempted = true;
|
|
995
1060
|
const disposeConfirmed = await runSessionActionWithGrace("dispose", () => session.dispose());
|
|
@@ -1004,7 +1069,7 @@ async function executeSingleSdkAttemptInternal(options) {
|
|
|
1004
1069
|
}
|
|
1005
1070
|
}
|
|
1006
1071
|
}
|
|
1007
|
-
if (providerErrorMessage && stopReason === "error" && !stderr)
|
|
1072
|
+
if (providerErrorMessage && stopReason === "error" && !stderr.includes(providerErrorMessage))
|
|
1008
1073
|
appendStderr(providerErrorMessage);
|
|
1009
1074
|
if (options.outputLimitRecovery && stopReason === "length" && !committedTerminalObserved && outputContinuesSent >= PI_SDK_OUTPUT_LIMIT_CONTINUE_MAX) {
|
|
1010
1075
|
const note = serializeSessionEvent({ type: "loop-agent-output-limit-exhausted", maxContinues: PI_SDK_OUTPUT_LIMIT_CONTINUE_MAX });
|
|
@@ -1024,9 +1089,9 @@ async function executeSingleSdkAttemptInternal(options) {
|
|
|
1024
1089
|
? collected.parsedEvents
|
|
1025
1090
|
: (fallbackParsed?.parsedEvents ?? 0);
|
|
1026
1091
|
const tokensUsed = aggregateSdkTokenUsage(usageSamples);
|
|
1027
|
-
const cancelled = stopReason === "aborted";
|
|
1092
|
+
const cancelled = stopReason === "aborted" && !protocolAbort && !providerBudgetAbort;
|
|
1028
1093
|
const outputLimitExhausted = Boolean(options.outputLimitRecovery && stopReason === "length" && !committedTerminalObserved && !timedOut && !stderr);
|
|
1029
|
-
const failureCategory = !terminationConfirmed ? "termination-unconfirmed" :
|
|
1094
|
+
const failureCategory = !terminationConfirmed ? "termination-unconfirmed" : cancelled ? "cancelled" : outputLimitExhausted ? "output-limit" : providerBudgetFailure ? (providerBudgetFailure.includes("REQUEST_ABORTED") ? "interrupted" : providerBudgetFailure.startsWith("FRONTEND_CONTEXT_ESTIMATE_EXCEEDED") ? "context-budget-exhausted" : "budget_breach") : contextBudgetExhausted ? "context-budget-exhausted" : protocolStop && !providerErrorMessage && !timedOut ? "invalid-output" : terminationConfirmed
|
|
1030
1095
|
? classifyPiFailure({
|
|
1031
1096
|
assistantText,
|
|
1032
1097
|
exitCode: timedOut ? 1 : stderr ? 1 : 0,
|
|
@@ -1,6 +1,7 @@
|
|
|
1
|
+
import { collectExistingFrontendVerificationPaths } from "../workflows/dag/frontend-implementation-contract.js";
|
|
1
2
|
import { spawn } from "node:child_process";
|
|
2
3
|
import { createHash, randomUUID } from "node:crypto";
|
|
3
|
-
import {
|
|
4
|
+
import { inspectFrontendTestCommand, compareFrontendTestObservation, parseFrontendTestExecutionReport } from "../workflows/dag/frontend-test-execution-evidence.js";
|
|
4
5
|
import { appendFileSync, existsSync, mkdirSync, writeFileSync } from "node:fs";
|
|
5
6
|
import { mkdir, readFile, realpath, stat, writeFile } from "node:fs/promises";
|
|
6
7
|
import path from "node:path";
|
|
@@ -493,6 +494,25 @@ export function buildFrontendVerificationEvidence(input) {
|
|
|
493
494
|
lintCommandTexts: input.bundle.lintEvidence?.commandTexts ?? [],
|
|
494
495
|
});
|
|
495
496
|
const entryByCommand = new Map(directory.map((entry) => [entry.command.trim(), entry]));
|
|
497
|
+
const entryForExecutedCommand = (command) => {
|
|
498
|
+
const normalized = command.trim();
|
|
499
|
+
const exact = entryByCommand.get(normalized);
|
|
500
|
+
if (exact)
|
|
501
|
+
return exact;
|
|
502
|
+
// Behavior commands with Vitest JSON capability are executed with the
|
|
503
|
+
// runtime-owned reporter/outputFile suffix. Preserve their frozen command
|
|
504
|
+
// identity in review evidence, but accept *only* that exact suffix so an
|
|
505
|
+
// arbitrary command sharing a textual prefix is never attributed to a
|
|
506
|
+
// verification-directory entry.
|
|
507
|
+
return directory.find((entry) => {
|
|
508
|
+
if (entry.lane !== "behavior")
|
|
509
|
+
return false;
|
|
510
|
+
if (!normalized.startsWith(entry.command.trim()))
|
|
511
|
+
return false;
|
|
512
|
+
const suffix = normalized.slice(entry.command.trim().length);
|
|
513
|
+
return /^(?:\s+--)?\s+--reporter=json\s+--outputFile=(?:'[^']*'|"[^"]*"|\S+)$/u.test(suffix);
|
|
514
|
+
});
|
|
515
|
+
};
|
|
496
516
|
return {
|
|
497
517
|
schemaVersion: 1,
|
|
498
518
|
schemaId: "frontend-verification-evidence-v1",
|
|
@@ -503,7 +523,7 @@ export function buildFrontendVerificationEvidence(input) {
|
|
|
503
523
|
lintStatus: input.lintStatus,
|
|
504
524
|
lintConfigured: input.lintConfigured,
|
|
505
525
|
commands: input.results.map((result, index) => {
|
|
506
|
-
const entry =
|
|
526
|
+
const entry = entryForExecutedCommand(result.command);
|
|
507
527
|
return {
|
|
508
528
|
index: index + 1,
|
|
509
529
|
commandId: entry?.commandId ?? null,
|
|
@@ -1993,12 +2013,29 @@ async function deriveUnresolvedVerificationEntrypointRefs(runDir, frozenCommandI
|
|
|
1993
2013
|
return [];
|
|
1994
2014
|
}
|
|
1995
2015
|
}
|
|
2016
|
+
async function inspectFrontendBundleTests(bundle, cwd, workspaceRoot, phase, writeSet = []) {
|
|
2017
|
+
const inspected = new Map();
|
|
2018
|
+
const failures = [];
|
|
2019
|
+
for (const [index, command] of bundle.behaviorCommands.entries()) {
|
|
2020
|
+
const label = bundle.behaviorEvidence.commandLabels[index];
|
|
2021
|
+
const frozen = bundle.behaviorEvidence.preflight.find(entry => entry.label === label)?.testObservation;
|
|
2022
|
+
const observed = await inspectFrontendTestCommand({ command, cwd, workspaceRoot, allowedPaths: writeSet });
|
|
2023
|
+
const reason = frozen ? compareFrontendTestObservation(frozen, observed.observation, phase, writeSet) : "missing frozen test observation preflight; regenerate the DAG";
|
|
2024
|
+
if (reason)
|
|
2025
|
+
failures.push({ code: "FRONTEND_TEST_OBSERVATION_INVALID", commandId: frozen?.commandId, command, reason });
|
|
2026
|
+
else
|
|
2027
|
+
inspected.set(command, observed);
|
|
2028
|
+
}
|
|
2029
|
+
return { inspected, failures };
|
|
2030
|
+
}
|
|
1996
2031
|
async function executeFrontendVerificationBundle(input, meta) {
|
|
1997
2032
|
const started = Date.now();
|
|
1998
2033
|
const shell = input.task.shell;
|
|
1999
2034
|
const bundle = shell.frontendVerificationBundle;
|
|
2000
2035
|
const cwd = resolveShellCwd(input.cwd, shell.cwd);
|
|
2001
2036
|
const results = [];
|
|
2037
|
+
const testPreflight = await inspectFrontendBundleTests(bundle, cwd, input.cwd, "execution");
|
|
2038
|
+
let testObservationFailure = testPreflight.failures[0];
|
|
2002
2039
|
const commandResultsByKey = new Map();
|
|
2003
2040
|
const commandKey = (command) => `${cwd}\u0000${command.trim()}`;
|
|
2004
2041
|
const reuseCommandResult = (result, command) => {
|
|
@@ -2027,8 +2064,9 @@ async function executeFrontendVerificationBundle(input, meta) {
|
|
|
2027
2064
|
const successfulLabels = new Map();
|
|
2028
2065
|
const successfulCommandTexts = new Map();
|
|
2029
2066
|
const executedTests = new Map();
|
|
2067
|
+
const testExecutionEvidence = [];
|
|
2030
2068
|
const commandExecutionEvidence = new Map();
|
|
2031
|
-
for (const command of bundle.lintCommands ?? []) {
|
|
2069
|
+
for (const command of testObservationFailure ? [] : bundle.lintCommands ?? []) {
|
|
2032
2070
|
const key = commandKey(command);
|
|
2033
2071
|
const cached = commandResultsByKey.get(key);
|
|
2034
2072
|
let result = cached ? reuseCommandResult(cached, command) : undefined;
|
|
@@ -2085,17 +2123,26 @@ async function executeFrontendVerificationBundle(input, meta) {
|
|
|
2085
2123
|
];
|
|
2086
2124
|
const failedGroups = new Set();
|
|
2087
2125
|
const groupFailures = [];
|
|
2088
|
-
for (const group of lintBlocked ? [] : groups) {
|
|
2126
|
+
for (const group of lintBlocked || testObservationFailure ? [] : groups) {
|
|
2089
2127
|
for (const [index, command] of group.commands.entries()) {
|
|
2090
|
-
if (failedGroups.has(group.name))
|
|
2128
|
+
if (failedGroups.has(group.name) || testObservationFailure)
|
|
2091
2129
|
break;
|
|
2130
|
+
if (testPreflight.inspected.has(command)) {
|
|
2131
|
+
const fresh = await inspectFrontendBundleTests(bundle, cwd, input.cwd, "execution");
|
|
2132
|
+
testObservationFailure = fresh.failures[0];
|
|
2133
|
+
if (testObservationFailure)
|
|
2134
|
+
break;
|
|
2135
|
+
for (const [text, observed] of fresh.inspected)
|
|
2136
|
+
testPreflight.inspected.set(text, observed);
|
|
2137
|
+
}
|
|
2092
2138
|
const key = commandKey(command);
|
|
2093
2139
|
const cached = commandResultsByKey.get(key);
|
|
2094
2140
|
let result = cached ? reuseCommandResult(cached, command) : undefined;
|
|
2095
2141
|
if (!result) {
|
|
2096
2142
|
const commandNumber = results.length + 1;
|
|
2097
2143
|
const reportPath = path.join(meta.runDir, input.task.id, "commands", `${commandNumber}-${randomUUID()}.vitest.json`);
|
|
2098
|
-
const
|
|
2144
|
+
const observedTest = testPreflight.inspected.get(command);
|
|
2145
|
+
const observedCommand = observedTest?.withReporter?.(reportPath);
|
|
2099
2146
|
if (observedCommand)
|
|
2100
2147
|
await mkdir(path.dirname(reportPath), { recursive: true });
|
|
2101
2148
|
result = await executeShellCommand({
|
|
@@ -2116,7 +2163,17 @@ async function executeFrontendVerificationBundle(input, meta) {
|
|
|
2116
2163
|
// Vitest reports physical paths (e.g. /private/var on macOS).
|
|
2117
2164
|
// Match those in workspace coordinates even when shell.cwd is
|
|
2118
2165
|
// a subdirectory or the caller used a directory alias.
|
|
2119
|
-
|
|
2166
|
+
const rawReport = await readFile(reportPath, "utf8");
|
|
2167
|
+
const reportSha256 = createHash("sha256").update(rawReport).digest("hex");
|
|
2168
|
+
commandExecutionEvidence.set(key, parseFrontendTestExecutionReport(JSON.parse(rawReport), await realpath(observedTest?.effectiveCwd ?? cwd), await realpath(input.cwd)));
|
|
2169
|
+
const actual = commandExecutionEvidence.get(key);
|
|
2170
|
+
for (let offset = 0; offset < actual.length; offset += 24) {
|
|
2171
|
+
const relative = `${input.task.id}/commands/${commandNumber}-execution-${offset / 24}.json`;
|
|
2172
|
+
const excerpt = { schemaVersion: 1, commandSha256: observedTest.observation.commandSha256, cwd: observedTest.observation.cwd, reportSha256, offset, total: actual.length, behaviorCoverage: "unconfirmed", tests: actual.slice(offset, offset + 24).map(test => ({ file: test.file, status: test.status, title: test.title.slice(0, 500), ...(test.title.length > 500 ? { titleTruncated: true } : {}) })) };
|
|
2173
|
+
await writeDagRunJsonArtifact(meta.runDir, relative, excerpt);
|
|
2174
|
+
const bytes = await readFile(path.join(meta.runDir, relative));
|
|
2175
|
+
testExecutionEvidence.push({ path: relative, sha256: createHash("sha256").update(bytes).digest("hex"), reportPath: path.relative(meta.runDir, reportPath).replace(/\\/g, "/"), reportSha256 });
|
|
2176
|
+
}
|
|
2120
2177
|
}
|
|
2121
2178
|
catch (error) {
|
|
2122
2179
|
reportError = `frontend test execution report unavailable: ${error instanceof Error ? error.message : String(error)}`;
|
|
@@ -2212,6 +2269,8 @@ async function executeFrontendVerificationBundle(input, meta) {
|
|
|
2212
2269
|
: undefined;
|
|
2213
2270
|
let traceError;
|
|
2214
2271
|
try {
|
|
2272
|
+
if (testObservationFailure)
|
|
2273
|
+
throw new Error(`${testObservationFailure.code}: ${testObservationFailure.reason}`);
|
|
2215
2274
|
await runFrontendVerificationTraceGate({
|
|
2216
2275
|
runDir: meta.runDir,
|
|
2217
2276
|
workspaceRoot: input.cwd,
|
|
@@ -2229,6 +2288,7 @@ async function executeFrontendVerificationBundle(input, meta) {
|
|
|
2229
2288
|
nodeId: input.task.id,
|
|
2230
2289
|
commandLabels: successfulLabels.get("behavior") ?? [],
|
|
2231
2290
|
executedTests,
|
|
2291
|
+
testExecutionEvidence,
|
|
2232
2292
|
},
|
|
2233
2293
|
},
|
|
2234
2294
|
});
|
|
@@ -2261,9 +2321,10 @@ async function executeFrontendVerificationBundle(input, meta) {
|
|
|
2261
2321
|
nodeId: failedNodeId,
|
|
2262
2322
|
rawFailureCategory: failureCategory,
|
|
2263
2323
|
});
|
|
2264
|
-
const failureOwner =
|
|
2265
|
-
|
|
2266
|
-
|
|
2324
|
+
const failureOwner = testObservationFailure ? "verification-config" :
|
|
2325
|
+
classifiedOwner === "unknown" &&
|
|
2326
|
+
["typecheck", "build", "lint", "component-test", "unit-test", "trace"].includes(failureClass)
|
|
2327
|
+
? "implementation" : classifiedOwner;
|
|
2267
2328
|
const restartPhase = classifyRestartPhase(failureOwner);
|
|
2268
2329
|
// A+B (AC-008): verification-config routing. Produce structured
|
|
2269
2330
|
// unresolved entrypoint refs from committed plan facts and the frozen
|
|
@@ -2325,6 +2386,7 @@ async function executeFrontendVerificationBundle(input, meta) {
|
|
|
2325
2386
|
...(verificationConfigRoute
|
|
2326
2387
|
? { verificationConfigRoute }
|
|
2327
2388
|
: {}),
|
|
2389
|
+
...(testObservationFailure ? { testObservationFailure, commandsExecuted: results.some(result => !result.reused) } : {}),
|
|
2328
2390
|
commandResults,
|
|
2329
2391
|
failedAt: new Date().toISOString(),
|
|
2330
2392
|
};
|
|
@@ -2689,8 +2751,9 @@ async function executeFrontendDesignPolicy(input, meta) {
|
|
|
2689
2751
|
durationMs: Date.now() - started,
|
|
2690
2752
|
};
|
|
2691
2753
|
}
|
|
2754
|
+
const existingVerificationPaths = await collectExistingFrontendVerificationPaths(input.cwd, contract);
|
|
2692
2755
|
const uncoveredVTs = contract.verificationTargets
|
|
2693
|
-
.filter((vt) => vt.mode !== "static")
|
|
2756
|
+
.filter((vt) => vt.mode !== "static" && !existingVerificationPaths.includes(vt.file))
|
|
2694
2757
|
.filter((vt) => ![...writeSet].some((pattern) => pathMatchesPattern(vt.file, pattern) || vt.file === pattern));
|
|
2695
2758
|
if (uncoveredVTs.length > 0) {
|
|
2696
2759
|
return {
|
|
@@ -2757,6 +2820,7 @@ async function executeFrontendDesignPolicy(input, meta) {
|
|
|
2757
2820
|
[],
|
|
2758
2821
|
allowedDependencies: config.allowedDependencies,
|
|
2759
2822
|
writeSetPatterns: config.implementationWriteSet ?? [],
|
|
2823
|
+
existingVerificationPaths: await collectExistingFrontendVerificationPaths(input.cwd, contract),
|
|
2760
2824
|
// Fresh-existence evidence at gate time (Scout pathEvidence.fresh
|
|
2761
2825
|
// semantics): a reuse-existing evidencePath must name a file that
|
|
2762
2826
|
// exists NOW; greenfield paths are decision=new only.
|
|
@@ -2894,12 +2958,21 @@ async function executeFrontendWriterAdmission(input, meta) {
|
|
|
2894
2958
|
const workspaceFiles = await collectWorkspaceFiles(input.cwd);
|
|
2895
2959
|
const admission = deriveFrontendWriterAdmissionShell({
|
|
2896
2960
|
contract,
|
|
2961
|
+
existingVerificationPaths: await collectExistingFrontendVerificationPaths(input.cwd, contract),
|
|
2897
2962
|
designPolicyResult,
|
|
2898
2963
|
baseline,
|
|
2899
2964
|
frozenCommandLabels: config.frozenCommandLabels,
|
|
2900
2965
|
allowedMockStrategies: config.allowedMockStrategies,
|
|
2901
2966
|
workspaceFiles,
|
|
2902
2967
|
});
|
|
2968
|
+
const testObservationFindings = [];
|
|
2969
|
+
for (const node of meta.spec.tasks) {
|
|
2970
|
+
const verification = node.shell?.frontendVerificationBundle;
|
|
2971
|
+
if (!verification)
|
|
2972
|
+
continue;
|
|
2973
|
+
const checked = await inspectFrontendBundleTests(verification, resolveShellCwd(input.cwd, node.shell?.cwd), input.cwd, "admission", admission.writeSet);
|
|
2974
|
+
testObservationFindings.push(...checked.failures.map(failure => ({ code: failure.code, message: failure.reason })));
|
|
2975
|
+
}
|
|
2903
2976
|
// 6. Design verdict enforcement: request_design_changes is a blocked
|
|
2904
2977
|
// admission. This keeps the writer fail-closed until the recovery plan
|
|
2905
2978
|
// has incorporated the reviewer's concrete findings.
|
|
@@ -2907,9 +2980,10 @@ async function executeFrontendWriterAdmission(input, meta) {
|
|
|
2907
2980
|
? "frontend-design-policy-shell"
|
|
2908
2981
|
: "frontend-writer-admission-shell";
|
|
2909
2982
|
const findings = verdictPass
|
|
2910
|
-
? admission.findings
|
|
2983
|
+
? [...admission.findings, ...testObservationFindings]
|
|
2911
2984
|
: [
|
|
2912
2985
|
...admission.findings,
|
|
2986
|
+
...testObservationFindings,
|
|
2913
2987
|
{
|
|
2914
2988
|
code: designRequested
|
|
2915
2989
|
? "design-review-requested"
|
|
@@ -3101,6 +3175,8 @@ async function executeFrontendCloseout(input, meta) {
|
|
|
3101
3175
|
status: target.status === "ok" ? "passed" : "failed",
|
|
3102
3176
|
evidence: trace?.targets?.[index]?.evidence ?? "unavailable",
|
|
3103
3177
|
behaviorCoverage: "unconfirmed",
|
|
3178
|
+
plannedEvidenceLevel: contract.verificationTargets.find(entry => entry.id === trace?.targets?.[index]?.id)?.evidenceLevel ?? "unknown",
|
|
3179
|
+
boundary: contract.verificationTargets.find(entry => entry.id === trace?.targets?.[index]?.id)?.boundary,
|
|
3104
3180
|
})),
|
|
3105
3181
|
lintStatus: "unavailable",
|
|
3106
3182
|
integrationFacts,
|
|
@@ -11,6 +11,8 @@ export const TERMINAL_OPERATION_STATES = new Set(["succeeded", "failed", "timed-
|
|
|
11
11
|
* inspection but does not itself prove a live operation; callers still check
|
|
12
12
|
* DAG/Worker/chat activity separately. restarting is the terminal receipt of
|
|
13
13
|
* a successful Console handoff, retained across boots, not a live slot.
|
|
14
|
+
* A needs-reconcile record remains visible as a non-business outcome, but
|
|
15
|
+
* must never strand later mutations after its runtime is gone.
|
|
14
16
|
*/
|
|
15
17
|
export const OCCUPIED_OPERATION_STATES = new Set(["queued", "starting", "running"]);
|
|
16
18
|
/** True when the state can still make progress and therefore holds the slot. */
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { readFileBoundarySafe } from "../../shared/path-safety.js";
|
|
1
2
|
import { access, mkdir, readdir, readFile, rename, rm } from "node:fs/promises";
|
|
2
3
|
import path from "node:path";
|
|
3
4
|
import { mkdtemp } from "node:fs/promises";
|
|
@@ -30,28 +31,27 @@ async function pathExists(filePath) {
|
|
|
30
31
|
function serializeManifest(manifest) {
|
|
31
32
|
return `${JSON.stringify(manifest, null, 2)}\n`;
|
|
32
33
|
}
|
|
34
|
+
function compareArtifactPaths(a, b) {
|
|
35
|
+
return a.path < b.path ? -1 : a.path > b.path ? 1 : 0;
|
|
36
|
+
}
|
|
33
37
|
function normalizeTasksForStorage(tasks) {
|
|
34
38
|
return [...tasks]
|
|
35
39
|
.map((task) => ({
|
|
36
40
|
...task,
|
|
37
41
|
taskRef: task.taskRef.replace(/\\/g, "/"),
|
|
38
42
|
seeds: [...task.seeds].sort((a, b) => a - b),
|
|
43
|
+
...("category" in task ? {
|
|
44
|
+
requirements: [...task.requirements].sort(compareArtifactPaths),
|
|
45
|
+
initialFiles: [...task.initialFiles].sort(compareArtifactPaths),
|
|
46
|
+
oracleFiles: [...task.oracleFiles].sort(compareArtifactPaths),
|
|
47
|
+
} : {}),
|
|
39
48
|
}))
|
|
40
49
|
.sort((a, b) => a.taskRef < b.taskRef ? -1 : a.taskRef > b.taskRef ? 1 : 0);
|
|
41
50
|
}
|
|
42
51
|
export async function materializeCorpusManifest(raw) {
|
|
43
52
|
const input = corpusManifestInputSchema.parse(raw);
|
|
44
53
|
const tasks = normalizeTasksForStorage(input.tasks);
|
|
45
|
-
const
|
|
46
|
-
schemaVersion: 1,
|
|
47
|
-
corpusId: input.corpusId,
|
|
48
|
-
createdAt: input.createdAt,
|
|
49
|
-
...(input.description !== undefined
|
|
50
|
-
? { description: input.description }
|
|
51
|
-
: {}),
|
|
52
|
-
tasks,
|
|
53
|
-
};
|
|
54
|
-
const corpusHash = computeCorpusHash(withoutHash);
|
|
54
|
+
const corpusHash = computeCorpusHash(input);
|
|
55
55
|
if (input.corpusHash) {
|
|
56
56
|
const provided = formatContentSha(normalizeContentSha(input.corpusHash));
|
|
57
57
|
if (provided !== corpusHash) {
|
|
@@ -59,10 +59,30 @@ export async function materializeCorpusManifest(raw) {
|
|
|
59
59
|
}
|
|
60
60
|
}
|
|
61
61
|
return corpusManifestSchema.parse({
|
|
62
|
-
...
|
|
62
|
+
...input,
|
|
63
|
+
tasks,
|
|
63
64
|
corpusHash,
|
|
64
65
|
});
|
|
65
66
|
}
|
|
67
|
+
export async function assertCorpusArtifacts(repoRoot, manifest) {
|
|
68
|
+
if (manifest.schemaVersion !== 2)
|
|
69
|
+
return;
|
|
70
|
+
const snapshots = new Map();
|
|
71
|
+
for (const task of manifest.tasks)
|
|
72
|
+
for (const ref of [...task.requirements, ...task.initialFiles, ...task.oracleFiles]) {
|
|
73
|
+
const bytes = await readFileBoundarySafe({ boundaryRoot: repoRoot,
|
|
74
|
+
targetPath: path.resolve(repoRoot, ref.path), label: "frontend corpus artifact" });
|
|
75
|
+
if (sha256Hex(bytes) !== ref.sha256)
|
|
76
|
+
throw Error(`corpus artifact hash mismatch: ${ref.path}`);
|
|
77
|
+
snapshots.set(ref.path, bytes);
|
|
78
|
+
}
|
|
79
|
+
for (const [file, expected] of snapshots) {
|
|
80
|
+
const actual = await readFileBoundarySafe({ boundaryRoot: repoRoot,
|
|
81
|
+
targetPath: path.resolve(repoRoot, file), label: "frontend corpus artifact recheck" });
|
|
82
|
+
if (!actual.equals(expected))
|
|
83
|
+
throw Error(`corpus artifact changed during collection: ${file}`);
|
|
84
|
+
}
|
|
85
|
+
}
|
|
66
86
|
async function assertManifestIntegrity(repoRoot, corpusId, rawManifest) {
|
|
67
87
|
const integrityPath = path.join(corpusDir(repoRoot, corpusId), MANIFEST_INTEGRITY_FILE);
|
|
68
88
|
const expected = (await readFile(integrityPath, "utf-8")).trim();
|
|
@@ -98,6 +118,7 @@ export async function readCorpusManifest(repoRoot, corpusId) {
|
|
|
98
118
|
if (recomputed !== stored.corpusHash) {
|
|
99
119
|
throw new Error(`corpusHash mismatch for ${corpusId}: expected ${recomputed}, got ${stored.corpusHash}`);
|
|
100
120
|
}
|
|
121
|
+
await assertCorpusArtifacts(repoRoot, stored);
|
|
101
122
|
return stored;
|
|
102
123
|
}
|
|
103
124
|
export async function listCorpusIds(repoRoot) {
|
|
@@ -117,7 +138,11 @@ export async function listCorpusIds(repoRoot) {
|
|
|
117
138
|
}
|
|
118
139
|
}
|
|
119
140
|
export async function registerCorpusManifest(input) {
|
|
120
|
-
const
|
|
141
|
+
const repoRoot = input.repoRoot;
|
|
142
|
+
const manifest = corpusManifestSchema.parse(input.manifest);
|
|
143
|
+
if (computeCorpusHash(manifest) !== manifest.corpusHash)
|
|
144
|
+
throw Error("corpusHash mismatch");
|
|
145
|
+
await assertCorpusArtifacts(repoRoot, manifest);
|
|
121
146
|
assertCorpusId(manifest.corpusId);
|
|
122
147
|
const manifestPath = corpusManifestPath(repoRoot, manifest.corpusId);
|
|
123
148
|
if (await pathExists(manifestPath)) {
|
|
@@ -863,9 +863,9 @@ export function buildOperatorCapabilitiesDocument() {
|
|
|
863
863
|
},
|
|
864
864
|
{
|
|
865
865
|
name: "capabilityBoundary",
|
|
866
|
-
type: "
|
|
866
|
+
type: "object",
|
|
867
867
|
required: false,
|
|
868
|
-
description: "optional frozen test capability boundary {allowed?, forbidden?} composed into dependencyPolicy by the runtime",
|
|
868
|
+
description: "optional frozen test capability boundary {allowed?, forbidden?, requiredClauses?} composed into dependencyPolicy by the runtime",
|
|
869
869
|
},
|
|
870
870
|
{
|
|
871
871
|
name: "interactionIds",
|
|
@@ -1237,10 +1237,9 @@ export function buildOperatorCapabilitiesDocument() {
|
|
|
1237
1237
|
},
|
|
1238
1238
|
{
|
|
1239
1239
|
name: "capabilityBoundary",
|
|
1240
|
-
type: "
|
|
1241
|
-
itemsType: "string",
|
|
1240
|
+
type: "object",
|
|
1242
1241
|
required: false,
|
|
1243
|
-
description: "frozen test capability boundary {allowed?, forbidden?} composed into dependencyPolicy by the runtime",
|
|
1242
|
+
description: "frozen test capability boundary {allowed?, forbidden?, requiredClauses?} composed into dependencyPolicy by the runtime",
|
|
1244
1243
|
},
|
|
1245
1244
|
{
|
|
1246
1245
|
name: "interactionIds",
|
|
@@ -1950,6 +1949,12 @@ export const OPERATOR_COMMAND_COVERAGE = Object.freeze([
|
|
|
1950
1949
|
action: "doctor",
|
|
1951
1950
|
source: "loop-agent",
|
|
1952
1951
|
},
|
|
1952
|
+
{
|
|
1953
|
+
command: "loop-agent promote-run",
|
|
1954
|
+
coverage: "excluded",
|
|
1955
|
+
action: null,
|
|
1956
|
+
source: "loop-agent",
|
|
1957
|
+
},
|
|
1953
1958
|
{
|
|
1954
1959
|
command: "loop-agent task advance",
|
|
1955
1960
|
coverage: "model-callable",
|
|
@@ -1,6 +1,20 @@
|
|
|
1
1
|
import { z } from "zod";
|
|
2
2
|
import { frontendExecutionPolicySchema } from "../shared/frontend-execution-policy.js";
|
|
3
3
|
import { DEFAULT_FRONTEND_SPEC_ROOTS, isFrontendSpecFilePath, isValidFrontendSpecRoot, } from "../shared/openspec-spec.js";
|
|
4
|
+
/** Frozen user-authored clauses. Preserve wording/line breaks; normalize only
|
|
5
|
+
* line endings and surrounding whitespace, with explicit bounded payloads. */
|
|
6
|
+
const capabilityClauses = z.array(z.string().transform(value => value.replace(/\r\n?/g, "\n").trim()).pipe(z.string().min(1).max(4096))).max(64).transform(values => [...new Set(values)]);
|
|
7
|
+
export const capabilityBoundarySchema = z.object({
|
|
8
|
+
allowed: capabilityClauses.optional(),
|
|
9
|
+
forbidden: capabilityClauses.optional(),
|
|
10
|
+
requiredClauses: capabilityClauses.optional(),
|
|
11
|
+
}).strict().superRefine((boundary, ctx) => {
|
|
12
|
+
const clauses = [...(boundary.allowed ?? []), ...(boundary.forbidden ?? []), ...(boundary.requiredClauses ?? [])];
|
|
13
|
+
if (!clauses.length)
|
|
14
|
+
ctx.addIssue({ code: z.ZodIssueCode.custom, message: "capabilityBoundary needs allowed, forbidden or requiredClauses entries" });
|
|
15
|
+
if (clauses.reduce((total, clause) => total + clause.length, 0) > 32768)
|
|
16
|
+
ctx.addIssue({ code: z.ZodIssueCode.custom, message: "capabilityBoundary exceeds 32768 characters" });
|
|
17
|
+
});
|
|
4
18
|
export const taskFlowSchema = z.enum([
|
|
5
19
|
"auto",
|
|
6
20
|
"micro",
|
|
@@ -39,6 +53,13 @@ export const taskVerifyCommandSchema = z.object({
|
|
|
39
53
|
label: z.string().min(1),
|
|
40
54
|
command: z.string().min(1),
|
|
41
55
|
timeoutMs: z.number().int().positive().optional(),
|
|
56
|
+
/** Explicit verification lane for this command. Omit to keep automatic
|
|
57
|
+
* classification. Declaring `static` is the escape hatch for commands that
|
|
58
|
+
* provably cannot emit a test report (governance/lint scripts): the behavior
|
|
59
|
+
* lane requires test-observation capability at writer admission, so a
|
|
60
|
+
* non-test command misclassified into it blocks admission deterministically.
|
|
61
|
+
* Static commands still execute and their exit code is still checked. */
|
|
62
|
+
mode: z.enum(["static", "behavior"]).optional(),
|
|
42
63
|
});
|
|
43
64
|
export const dagVerifyStrategySchema = z
|
|
44
65
|
.object({
|
|
@@ -304,15 +325,7 @@ const taskConfigObjectSchema = z.object({
|
|
|
304
325
|
.optional(),
|
|
305
326
|
/** Generation-frozen test capability boundary composed into the canonical
|
|
306
327
|
* dependencyPolicy by the runtime (planner never maintains it). */
|
|
307
|
-
capabilityBoundary:
|
|
308
|
-
.object({
|
|
309
|
-
allowed: z.array(z.string().min(1)).optional(),
|
|
310
|
-
forbidden: z.array(z.string().min(1)).optional(),
|
|
311
|
-
requiredClauses: z.array(z.string().min(1)).optional(),
|
|
312
|
-
})
|
|
313
|
-
.strict()
|
|
314
|
-
.refine((entry) => (entry.allowed?.length ?? 0) > 0 || (entry.forbidden?.length ?? 0) > 0 || (entry.requiredClauses?.length ?? 0) > 0, { message: "capabilityBoundary needs allowed or forbidden entries" })
|
|
315
|
-
.optional(),
|
|
328
|
+
capabilityBoundary: capabilityBoundarySchema.optional(),
|
|
316
329
|
/** Generation-frozen interaction id vocabulary. Contract interaction names
|
|
317
330
|
* must come from this set (plan finalize deterministic check). */
|
|
318
331
|
interactionIds: z.array(z.string().min(1)).optional(),
|