@tea-agent/loop-agent 0.13.0-beta.0 → 0.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (127) hide show
  1. package/AGENTS.md +2 -0
  2. package/CHANGELOG.md +56 -305
  3. package/README.md +13 -19
  4. package/dist/commands/init.js +92 -23
  5. package/dist/executors/pi-event-serializer.js +33 -11
  6. package/dist/executors/shell-executor.js +200 -21
  7. package/dist/infrastructure/evaluation/candidate-store.js +5 -1
  8. package/dist/worker/observe/spec-evidence.js +19 -10
  9. package/dist/worker/observe/static/app.js +4 -3
  10. package/dist/worker/observe/static/constants.js +10 -2
  11. package/dist/worker/observe/static/dag-helpers.js +37 -8
  12. package/dist/worker/observe/static/dom.js +159 -0
  13. package/dist/worker/observe/static/format-pool.d.ts +71 -0
  14. package/dist/worker/observe/static/format-pool.js +67 -0
  15. package/dist/worker/observe/static/format.js +27 -2
  16. package/dist/worker/observe/static/index.html +76 -34
  17. package/dist/worker/observe/static/kpi.js +12 -6
  18. package/dist/worker/observe/static/markdown-render.js +124 -0
  19. package/dist/worker/observe/static/shell-chrome.js +8 -2
  20. package/dist/worker/observe/static/state.js +20 -0
  21. package/dist/worker/observe/static/styles.css +662 -60
  22. package/dist/worker/observe/static/views/dag-inspector.js +65 -142
  23. package/dist/worker/observe/static/views/dag.js +9 -0
  24. package/dist/worker/observe/static/views/dashboard.js +512 -269
  25. package/dist/worker/observe/static/views/pool.js +595 -237
  26. package/dist/worker/observe/static/views/session-timeline.js +577 -11
  27. package/dist/workflows/dag/backend-test-case-manifest.js +503 -0
  28. package/dist/workflows/dag/backend-test-execution-contract.js +353 -0
  29. package/dist/workflows/dag/backend-test-result-contract.js +568 -0
  30. package/dist/workflows/dag/decision-envelope.js +57 -2
  31. package/dist/workflows/dag/frontend-implementation-contract.js +240 -0
  32. package/dist/workflows/dag/frontend-project-capability.js +309 -0
  33. package/dist/workflows/dag/frontend-repair.js +341 -0
  34. package/dist/workflows/dag/frontend-risk.js +161 -0
  35. package/dist/workflows/dag/frontend-verification-trace.js +190 -0
  36. package/dist/workflows/dag/init-hybrid.js +1020 -125
  37. package/dist/workflows/dag/repair-artifact.js +43 -3
  38. package/dist/workflows/dag/skill-instructions.js +4 -2
  39. package/dist/workflows/dag/types.js +29 -8
  40. package/docs/README.md +2 -0
  41. package/docs/agent-dag-recovery-playbook.md +3 -3
  42. package/docs/agent-dag-runner.md +3 -3
  43. package/docs/architecture/README.md +3 -3
  44. package/docs/architecture/dag-execution.md +1 -1
  45. package/docs/architecture/evolution.md +13 -13
  46. package/docs/architecture/facts-and-state.md +1 -1
  47. package/docs/architecture/runtime-boundaries.md +7 -7
  48. package/docs/architecture/system-overview.md +3 -3
  49. package/docs/architecture/worker-and-feature.md +3 -3
  50. package/docs/design/README.md +7 -7
  51. package/docs/development-principles.md +4 -4
  52. package/docs/exec-plans/active/README.md +2 -4
  53. package/docs/exec-plans/completed/README.md +29 -6
  54. package/docs/feature-workflow.md +57 -32
  55. package/docs/init-surface.manifest.json +21 -3
  56. package/docs/loop-agent-harness.md +8 -8
  57. package/docs/production-readiness.md +1 -1
  58. package/docs/progress/README.md +20 -3
  59. package/docs/reports/README.md +53 -7
  60. package/docs/templates/agent-dag.supervised-implementation.json +127 -8
  61. package/docs/templates/backend-test-case-manifest.schema.json +190 -0
  62. package/docs/templates/backend-test-dag.classify.prompt.md +75 -0
  63. package/docs/templates/backend-test-dag.generate-pytest.prompt.md +6 -4
  64. package/docs/templates/backend-test-dag.json +269 -21
  65. package/docs/templates/backend-test-dag.retrospect.prompt.md +44 -30
  66. package/docs/templates/backend-test-dag.review-cases.prompt.md +6 -4
  67. package/docs/templates/backend-test-execution.schema.json +133 -0
  68. package/docs/templates/backend-test-result.schema.json +99 -0
  69. package/docs/templates/branch-merge-report.md +93 -0
  70. package/docs/templates/frontend-eval/fixtures/failures/01-type-build-error.md +17 -0
  71. package/docs/templates/frontend-eval/fixtures/failures/02-unit-component-test-fail.md +16 -0
  72. package/docs/templates/frontend-eval/fixtures/failures/03-fixture-schema-drift.md +16 -0
  73. package/docs/templates/frontend-eval/fixtures/failures/04-missing-loading-empty-error-state.md +16 -0
  74. package/docs/templates/frontend-eval/fixtures/failures/05-forbidden-write-writeset-expansion.md +16 -0
  75. package/docs/templates/frontend-eval/fixtures/failures/06-unapproved-dependency-add.md +16 -0
  76. package/docs/templates/frontend-eval/fixtures/failures/07-mock-production-on.md +21 -0
  77. package/docs/templates/frontend-eval/fixtures/functional/01-simple-component-style.md +29 -0
  78. package/docs/templates/frontend-eval/fixtures/functional/02-form-validation.md +28 -0
  79. package/docs/templates/frontend-eval/fixtures/functional/03-list-detail-page.md +28 -0
  80. package/docs/templates/frontend-eval/fixtures/functional/04-api-mock.md +29 -0
  81. package/docs/templates/frontend-eval/fixtures/functional/05-permission-auth-gated-ui.md +27 -0
  82. package/docs/templates/frontend-eval/fixtures/functional/06-ssr-server-client-boundary.md +28 -0
  83. package/docs/templates/frontend-eval/fixtures/functional/07-shared-public-component-api.md +28 -0
  84. package/docs/templates/frontend-eval/fixtures/functional/08-pure-local-no-remote.md +27 -0
  85. package/docs/templates/frontend-eval/metrics.md +138 -0
  86. package/docs/templates/frontend-eval/smoke-targets.md +53 -0
  87. package/docs/templates/frontend-implementation-contract.schema.json +27 -0
  88. package/docs/verification-matrix.md +1 -1
  89. package/examples/decision-gate-agent-dag.json +4 -4
  90. package/examples/hybrid-loop-agent-dag.json +1 -1
  91. package/package.json +2 -2
  92. package/skills/ai-engineering-context/SKILL.md +2 -2
  93. package/skills/browser-tools/SKILL.md +196 -0
  94. package/skills/browser-tools/browser-content.js +103 -0
  95. package/skills/browser-tools/browser-cookies.js +35 -0
  96. package/skills/browser-tools/browser-eval.js +53 -0
  97. package/skills/browser-tools/browser-hn-scraper.js +108 -0
  98. package/skills/browser-tools/browser-nav.js +44 -0
  99. package/skills/browser-tools/browser-pick.js +162 -0
  100. package/skills/browser-tools/browser-screenshot.js +34 -0
  101. package/skills/browser-tools/browser-start.js +86 -0
  102. package/skills/browser-tools/package-lock.json +2556 -0
  103. package/skills/browser-tools/package.json +19 -0
  104. package/skills/frontend-implementation/SKILL.md +3 -1
  105. package/skills/frontend-implementation/references/node-contracts.md +17 -66
  106. package/skills/frontend-verification/SKILL.md +1 -1
  107. package/skills/grill-with-docs/SKILL.md +5 -5
  108. package/skills/grill-with-docs/adr-format.md +3 -3
  109. package/skills/init-capability-evolution/SKILL.md +5 -5
  110. package/skills/loop-agent/SKILL.md +5 -5
  111. package/skills/loop-agent/references/README.md +3 -3
  112. package/skills/loop-agent/references/command-reference.md +39 -17
  113. package/skills/loop-agent/references/docs-converge.md +15 -15
  114. package/skills/loop-agent/references/harness-policy.md +2 -2
  115. package/skills/loop-agent/references/hybrid-dag.md +20 -15
  116. package/skills/loop-agent/references/multi-worktree.md +1 -1
  117. package/skills/loop-agent/references/orchestrator-and-interventions.md +8 -8
  118. package/skills/loop-agent/references/task-workflow.md +1 -1
  119. package/skills/loop-agent/references/verification-and-failure-handling.md +6 -4
  120. package/skills/requesting-code-review/SKILL.md +1 -1
  121. package/skills/systematic-debugging/CREATION-LOG.md +3 -3
  122. package/skills/systematic-debugging/SKILL.md +1 -1
  123. package/skills/systematic-debugging/test-academic.md +1 -1
  124. package/skills/systematic-debugging/test-pressure-1.md +1 -1
  125. package/skills/systematic-debugging/test-pressure-2.md +1 -1
  126. package/skills/systematic-debugging/test-pressure-3.md +1 -1
  127. package/skills/verification-before-completion/SKILL.md +1 -1
@@ -0,0 +1,568 @@
1
+ import { createHash } from "node:crypto";
2
+ import { readFile } from "node:fs/promises";
3
+ import path from "node:path";
4
+ import { z } from "zod";
5
+ import { writeDagRunJsonArtifact } from "../../infrastructure/harness/artifact-store.js";
6
+ export const BACKEND_TEST_RESULT_SCHEMA_ID = "backend-test-result-v1";
7
+ const SECRET_KEY = /(?:password|passwd|secret|token|api[_-]?key|private[_-]?key|credential|authorization)/i;
8
+ const SECRET_VALUE = /(?:-----BEGIN [A-Z ]*PRIVATE KEY-----|\b(?:sk|ghp|github_pat|xox[baprs]|AKIA)[-_A-Za-z0-9]{12,}\b)/;
9
+ export const BACKEND_TEST_CLASSIFICATION_CATEGORIES = [
10
+ "ProductBug",
11
+ "TestBug",
12
+ "EnvFailure",
13
+ "ContractMismatch",
14
+ "FlakyTest",
15
+ "Unknown",
16
+ ];
17
+ export const backendTestExecutionStatusSchema = z.enum([
18
+ "completed",
19
+ "collection-error",
20
+ "command-error",
21
+ "report-error",
22
+ ]);
23
+ export const backendTestCollectionStatusSchema = z.enum([
24
+ "ok",
25
+ "error",
26
+ "unknown",
27
+ "skipped",
28
+ ]);
29
+ /**
30
+ * Task Pool consumable outcome tokens (frozen for M2; no auto follow-up).
31
+ * - passed: all executed tests green
32
+ * - completed-with-failures: process completed with assertion/test failures
33
+ * - collection-error: import/collection failures
34
+ * - command-error: runner/command could not complete normally
35
+ * - report-error: junit missing/corrupt after execution attempt
36
+ */
37
+ export const backendTestOutcomeSchema = z.enum([
38
+ "passed",
39
+ "completed-with-failures",
40
+ "collection-error",
41
+ "command-error",
42
+ "report-error",
43
+ ]);
44
+ const junitRelativePathSchema = z
45
+ .string()
46
+ .min(1)
47
+ .refine((value) => !path.posix.isAbsolute(value) &&
48
+ !value.includes("\\") &&
49
+ value
50
+ .split("/")
51
+ .every((segment) => segment.length > 0 && segment !== "." && segment !== ".."), { message: "junit.relativePath must be a safe run-relative POSIX path" });
52
+ const failureSummarySchema = z
53
+ .object({
54
+ classname: z.string().min(1),
55
+ name: z.string().min(1),
56
+ message: z.string().min(1),
57
+ kind: z.enum(["failure", "error"]).default("failure"),
58
+ })
59
+ .strict();
60
+ export const backendTestResultContractSchema = z
61
+ .object({
62
+ schemaVersion: z.literal(1),
63
+ /** Process-level status of the pytest invocation. Task Pool: use with outcome. */
64
+ executionStatus: backendTestExecutionStatusSchema,
65
+ pytestExitCode: z.number().int().min(0).max(255),
66
+ collectionStatus: backendTestCollectionStatusSchema,
67
+ /** Total tests counted from JUnit (or 0 when report unavailable). */
68
+ tests: z.number().int().min(0),
69
+ passed: z.number().int().min(0),
70
+ failed: z.number().int().min(0),
71
+ error: z.number().int().min(0),
72
+ skipped: z.number().int().min(0),
73
+ durationMs: z.number().nonnegative().optional(),
74
+ junit: z
75
+ .object({
76
+ relativePath: junitRelativePathSchema,
77
+ sha256: z
78
+ .string()
79
+ .regex(/^[a-f0-9]{64}$/, "junit.sha256 must be lowercase hex sha256"),
80
+ })
81
+ .strict(),
82
+ /** Non-secret command summary only (no tokens/passwords). */
83
+ commandSummary: z.string().min(1),
84
+ /** Truncated failure/error summaries for Task Pool triage. */
85
+ failures: z.array(failureSummarySchema),
86
+ /**
87
+ * Authoritative shell-facing outcome for backend-test-outcome-gate-shell.
88
+ * Do not override from retrospective Markdown.
89
+ */
90
+ outcome: backendTestOutcomeSchema,
91
+ })
92
+ .strict()
93
+ .superRefine((value, ctx) => {
94
+ if (value.tests !==
95
+ value.passed + value.failed + value.error + value.skipped) {
96
+ ctx.addIssue({
97
+ code: z.ZodIssueCode.custom,
98
+ message: "tests must equal passed+failed+error+skipped",
99
+ path: ["tests"],
100
+ });
101
+ }
102
+ if (SECRET_KEY.test(value.commandSummary) ||
103
+ SECRET_VALUE.test(value.commandSummary)) {
104
+ ctx.addIssue({
105
+ code: z.ZodIssueCode.custom,
106
+ message: "commandSummary must not contain secret-like content",
107
+ path: ["commandSummary"],
108
+ });
109
+ }
110
+ for (const [index, failure] of value.failures.entries()) {
111
+ if (SECRET_VALUE.test(failure.message)) {
112
+ ctx.addIssue({
113
+ code: z.ZodIssueCode.custom,
114
+ message: "failure message must not contain secret-like content",
115
+ path: ["failures", index, "message"],
116
+ });
117
+ }
118
+ }
119
+ });
120
+ const MAX_FAILURE_MESSAGE = 500;
121
+ const MAX_FAILURES = 50;
122
+ function decodeXmlEntities(value) {
123
+ return value
124
+ .replace(/&lt;/g, "<")
125
+ .replace(/&gt;/g, ">")
126
+ .replace(/&quot;/g, '"')
127
+ .replace(/&apos;/g, "'")
128
+ .replace(/&amp;/g, "&");
129
+ }
130
+ const XML_ATTR_PATTERNS = {
131
+ tests: /\btests\s*=\s*"([^"]*)"/i,
132
+ failures: /\bfailures\s*=\s*"([^"]*)"/i,
133
+ errors: /\berrors\s*=\s*"([^"]*)"/i,
134
+ skipped: /\bskipped\s*=\s*"([^"]*)"/i,
135
+ time: /\btime\s*=\s*"([^"]*)"/i,
136
+ classname: /\bclassname\s*=\s*"([^"]*)"/i,
137
+ name: /\bname\s*=\s*"([^"]*)"/i,
138
+ message: /\bmessage\s*=\s*"([^"]*)"/i,
139
+ };
140
+ function attr(tag, name) {
141
+ const pattern = XML_ATTR_PATTERNS[name];
142
+ if (!pattern)
143
+ return undefined;
144
+ const match = tag.match(pattern);
145
+ return match ? decodeXmlEntities(match[1] ?? "") : undefined;
146
+ }
147
+ function truncate(value, max = MAX_FAILURE_MESSAGE) {
148
+ const normalized = value.replace(/\s+/g, " ").trim();
149
+ if (normalized.length <= max)
150
+ return normalized;
151
+ return `${normalized.slice(0, max - 1)}…`;
152
+ }
153
+ function assertNoSecrets(label, value) {
154
+ if (SECRET_KEY.test(value) || SECRET_VALUE.test(value)) {
155
+ throw new Error(`secret-like content rejected in ${label}`);
156
+ }
157
+ }
158
+ /**
159
+ * Minimal JUnit XML parser (no new npm deps). Fail-closed on non-JUnit markup.
160
+ */
161
+ export function parseJunitXml(xml) {
162
+ const trimmed = xml.trim();
163
+ if (!trimmed) {
164
+ throw new Error("invalid junit xml: empty");
165
+ }
166
+ if (!/<testsuites\b/i.test(trimmed) && !/<testsuite\b/i.test(trimmed)) {
167
+ throw new Error("invalid junit xml: missing testsuites/testsuite root");
168
+ }
169
+ const hasSuitesRoot = /<testsuites\b/i.test(trimmed);
170
+ if ((hasSuitesRoot && !/<\/testsuites>\s*$/i.test(trimmed)) ||
171
+ (!hasSuitesRoot && !/<\/testsuite>\s*$/i.test(trimmed))) {
172
+ throw new Error("invalid junit xml: unclosed root element");
173
+ }
174
+ const openedCases = (trimmed.match(/<testcase\b/gi) ?? []).length;
175
+ const closedCases = (trimmed.match(/<\/testcase>/gi) ?? []).length;
176
+ const selfClosingCases = (trimmed.match(/<testcase\b[^>]*\/>/gi) ?? [])
177
+ .length;
178
+ if (openedCases !== closedCases + selfClosingCases) {
179
+ throw new Error("invalid junit xml: unclosed testcase element");
180
+ }
181
+ // Prefer root testsuites aggregates when present.
182
+ const suitesOpen = trimmed.match(/<testsuites\b[^>]*>/i)?.[0];
183
+ let tests = 0;
184
+ let failed = 0;
185
+ let errors = 0;
186
+ let skipped = 0;
187
+ let timeSec;
188
+ if (suitesOpen) {
189
+ const t = attr(suitesOpen, "tests");
190
+ const f = attr(suitesOpen, "failures");
191
+ const e = attr(suitesOpen, "errors");
192
+ const s = attr(suitesOpen, "skipped");
193
+ const time = attr(suitesOpen, "time");
194
+ if (t !== undefined)
195
+ tests = Number(t);
196
+ if (f !== undefined)
197
+ failed = Number(f);
198
+ if (e !== undefined)
199
+ errors = Number(e);
200
+ if (s !== undefined)
201
+ skipped = Number(s);
202
+ if (time !== undefined && time !== "")
203
+ timeSec = Number(time);
204
+ }
205
+ // Self-closing first so empty cases ending with /> are not greedily paired with a later </testcase>.
206
+ const caseRe = /<testcase\b([^>]*?)\/>|<testcase\b([^>]*)>([\s\S]*?)<\/testcase>/gi;
207
+ const failures = [];
208
+ let caseCount = 0;
209
+ let caseFailed = 0;
210
+ let caseErrors = 0;
211
+ let caseSkipped = 0;
212
+ let match = caseRe.exec(trimmed);
213
+ while (match !== null) {
214
+ caseCount += 1;
215
+ const openAttrs = match[1] ?? match[2] ?? "";
216
+ const body = match[3] ?? "";
217
+ const classname = attr(openAttrs, "classname") || "unknown";
218
+ const name = attr(openAttrs, "name") || "unknown";
219
+ const failureTag = body.match(/<failure\b([^>]*)>([\s\S]*?)<\/failure>|<failure\b([^>]*)\/>/i);
220
+ const errorTag = body.match(/<error\b([^>]*)>([\s\S]*?)<\/error>|<error\b([^>]*)\/>/i);
221
+ const skippedTag = /<skipped\b/i.test(body);
222
+ if (failureTag) {
223
+ caseFailed += 1;
224
+ const fAttrs = failureTag[1] ?? failureTag[3] ?? "";
225
+ const fBody = failureTag[2] ?? "";
226
+ const message = attr(fAttrs, "message") || fBody || "failure";
227
+ failures.push({
228
+ classname,
229
+ name,
230
+ message: truncate(message),
231
+ kind: "failure",
232
+ });
233
+ }
234
+ else if (errorTag) {
235
+ caseErrors += 1;
236
+ const eAttrs = errorTag[1] ?? errorTag[3] ?? "";
237
+ const eBody = errorTag[2] ?? "";
238
+ const message = attr(eAttrs, "message") || eBody || "error";
239
+ failures.push({
240
+ classname,
241
+ name,
242
+ message: truncate(message),
243
+ kind: "error",
244
+ });
245
+ }
246
+ else if (skippedTag) {
247
+ caseSkipped += 1;
248
+ }
249
+ match = caseRe.exec(trimmed);
250
+ }
251
+ // Prefer case-level facts when testcase elements exist; never trust root counters
252
+ // that hide <failure>/<error> children (fail-closed on contradictory aggregates).
253
+ if (caseCount > 0) {
254
+ if (suitesOpen) {
255
+ const declaredTests = attr(suitesOpen, "tests") !== undefined
256
+ ? Number(attr(suitesOpen, "tests"))
257
+ : undefined;
258
+ const declaredFailed = attr(suitesOpen, "failures") !== undefined
259
+ ? Number(attr(suitesOpen, "failures"))
260
+ : undefined;
261
+ const declaredErrors = attr(suitesOpen, "errors") !== undefined
262
+ ? Number(attr(suitesOpen, "errors"))
263
+ : undefined;
264
+ const declaredSkipped = attr(suitesOpen, "skipped") !== undefined
265
+ ? Number(attr(suitesOpen, "skipped"))
266
+ : undefined;
267
+ // pytest collection reports often use tests="0" with a synthetic error testcase.
268
+ // Always fail-closed when root counters under-report failure/error children.
269
+ if (declaredTests !== undefined &&
270
+ !Number.isNaN(declaredTests) &&
271
+ declaredTests !== caseCount) {
272
+ const collectionStyle = declaredTests === 0 && caseErrors > 0 && caseFailed === 0;
273
+ if (!collectionStyle) {
274
+ throw new Error(`invalid junit xml: testsuites tests=${declaredTests} disagrees with testcase count=${caseCount}`);
275
+ }
276
+ }
277
+ if (declaredFailed !== undefined &&
278
+ !Number.isNaN(declaredFailed) &&
279
+ declaredFailed !== caseFailed) {
280
+ throw new Error(`invalid junit xml: testsuites failures=${declaredFailed} disagrees with case failures=${caseFailed}`);
281
+ }
282
+ if (declaredErrors !== undefined &&
283
+ !Number.isNaN(declaredErrors) &&
284
+ declaredErrors !== caseErrors) {
285
+ throw new Error(`invalid junit xml: testsuites errors=${declaredErrors} disagrees with case errors=${caseErrors}`);
286
+ }
287
+ if (declaredSkipped !== undefined &&
288
+ !Number.isNaN(declaredSkipped) &&
289
+ declaredSkipped !== caseSkipped) {
290
+ throw new Error(`invalid junit xml: testsuites skipped=${declaredSkipped} disagrees with case skipped=${caseSkipped}`);
291
+ }
292
+ }
293
+ tests = caseCount;
294
+ failed = caseFailed;
295
+ errors = caseErrors;
296
+ skipped = caseSkipped;
297
+ }
298
+ else if (!suitesOpen || Number.isNaN(tests) || tests === 0) {
299
+ // No testcase bodies: fall back to first testsuite attributes.
300
+ const suiteOpen = trimmed.match(/<testsuite\b[^>]*>/i)?.[0];
301
+ if (!suiteOpen) {
302
+ throw new Error("invalid junit xml: no testsuite data");
303
+ }
304
+ tests = Number(attr(suiteOpen, "tests") ?? "0");
305
+ failed = Number(attr(suiteOpen, "failures") ?? "0");
306
+ errors = Number(attr(suiteOpen, "errors") ?? "0");
307
+ skipped = Number(attr(suiteOpen, "skipped") ?? "0");
308
+ const time = attr(suiteOpen, "time");
309
+ if (time)
310
+ timeSec = Number(time);
311
+ }
312
+ if ([tests, failed, errors, skipped].some((n) => Number.isNaN(n) || n < 0)) {
313
+ throw new Error("invalid junit xml: non-numeric suite counters");
314
+ }
315
+ const passed = Math.max(0, tests - failed - errors - skipped);
316
+ return {
317
+ tests,
318
+ passed,
319
+ failed,
320
+ errors,
321
+ skipped,
322
+ durationMs: timeSec !== undefined && !Number.isNaN(timeSec)
323
+ ? Math.round(timeSec * 1000)
324
+ : undefined,
325
+ failures: failures.slice(0, MAX_FAILURES),
326
+ };
327
+ }
328
+ export function deriveBackendTestResult(input) {
329
+ assertNoSecrets("commandSummary", input.commandSummary);
330
+ const emptyJunitMeta = {
331
+ relativePath: input.junitRelativePath,
332
+ sha256: "0".repeat(64),
333
+ };
334
+ if (input.junitXml === null || input.junitXml === undefined) {
335
+ if (!input.allowMissingJunit) {
336
+ throw new Error("missing junit report");
337
+ }
338
+ const outcome = input.pytestExitCode === 0 ? "report-error" : "command-error";
339
+ const executionStatus = input.pytestExitCode === 0 ? "report-error" : "command-error";
340
+ const result = {
341
+ schemaVersion: 1,
342
+ executionStatus,
343
+ pytestExitCode: input.pytestExitCode,
344
+ collectionStatus: "unknown",
345
+ tests: 0,
346
+ passed: 0,
347
+ failed: 0,
348
+ error: 0,
349
+ skipped: 0,
350
+ junit: emptyJunitMeta,
351
+ commandSummary: input.commandSummary,
352
+ failures: [],
353
+ outcome,
354
+ };
355
+ return backendTestResultContractSchema.parse(result);
356
+ }
357
+ let parsed;
358
+ try {
359
+ parsed = parseJunitXml(input.junitXml);
360
+ }
361
+ catch (error) {
362
+ throw new Error(`invalid junit xml: ${error instanceof Error ? error.message : String(error)}`);
363
+ }
364
+ const sha256 = createHash("sha256").update(input.junitXml).digest("hex");
365
+ const exit = input.pytestExitCode;
366
+ let executionStatus = "completed";
367
+ let collectionStatus = "ok";
368
+ let outcome = "passed";
369
+ // Collection-heavy signals: pytest exit 2 is common for collection errors;
370
+ // also when error cases exist with zero/low completed tests.
371
+ const looksLikeCollection = exit === 2 ||
372
+ (parsed.errors > 0 && parsed.passed + parsed.failed === 0) ||
373
+ parsed.failures.some((f) => f.kind === "error" &&
374
+ /collect|import|syntax/i.test(`${f.name} ${f.message}`));
375
+ if (looksLikeCollection && (parsed.errors > 0 || exit >= 2)) {
376
+ executionStatus = "collection-error";
377
+ collectionStatus = "error";
378
+ outcome = "collection-error";
379
+ }
380
+ else if (exit >= 2 && parsed.failed === 0 && parsed.errors === 0) {
381
+ executionStatus = "command-error";
382
+ collectionStatus = "unknown";
383
+ outcome = "command-error";
384
+ }
385
+ else if (parsed.failed > 0 || parsed.errors > 0 || exit === 1) {
386
+ executionStatus = "completed";
387
+ collectionStatus = "ok";
388
+ outcome = "completed-with-failures";
389
+ }
390
+ else if (exit === 0 && parsed.failed === 0 && parsed.errors === 0) {
391
+ executionStatus = "completed";
392
+ collectionStatus = "ok";
393
+ outcome = "passed";
394
+ }
395
+ else {
396
+ executionStatus = "command-error";
397
+ collectionStatus = "unknown";
398
+ outcome = "command-error";
399
+ }
400
+ const result = {
401
+ schemaVersion: 1,
402
+ executionStatus,
403
+ pytestExitCode: exit,
404
+ collectionStatus,
405
+ tests: parsed.tests,
406
+ passed: parsed.passed,
407
+ failed: parsed.failed,
408
+ error: parsed.errors,
409
+ skipped: parsed.skipped,
410
+ durationMs: parsed.durationMs,
411
+ junit: {
412
+ relativePath: input.junitRelativePath,
413
+ sha256,
414
+ },
415
+ commandSummary: input.commandSummary,
416
+ failures: parsed.failures.map((f) => ({
417
+ classname: f.classname || "unknown",
418
+ name: f.name || "unknown",
419
+ message: truncate(f.message || "failure"),
420
+ kind: f.kind,
421
+ })),
422
+ outcome,
423
+ };
424
+ return backendTestResultContractSchema.parse(result);
425
+ }
426
+ export function validateBackendTestResultJunitIntegrity(result, junitXml) {
427
+ backendTestResultContractSchema.parse(result);
428
+ const actualSha = createHash("sha256").update(junitXml).digest("hex");
429
+ if (result.junit.sha256 !== actualSha) {
430
+ throw new Error("junit sha256 mismatch against report content");
431
+ }
432
+ }
433
+ /**
434
+ * Deterministic classification constraints for Pi classifier (not the final label).
435
+ * Single-run failures must never suggest FlakyTest; collection/command never ProductBug.
436
+ */
437
+ export function classifyCategoryHints(result) {
438
+ const forbidden = new Set();
439
+ const suggested = new Set();
440
+ // Single observation cannot prove flakiness.
441
+ forbidden.add("FlakyTest");
442
+ if (result.executionStatus === "collection-error" ||
443
+ result.executionStatus === "command-error" ||
444
+ result.executionStatus === "report-error" ||
445
+ result.outcome === "collection-error" ||
446
+ result.outcome === "command-error" ||
447
+ result.outcome === "report-error") {
448
+ forbidden.add("ProductBug");
449
+ suggested.add("EnvFailure");
450
+ suggested.add("Unknown");
451
+ if (result.executionStatus === "collection-error") {
452
+ suggested.add("TestBug");
453
+ }
454
+ return {
455
+ suggestedCategories: [...suggested],
456
+ forbiddenCategories: [...forbidden],
457
+ confidenceCap: 0.6,
458
+ };
459
+ }
460
+ if (result.outcome === "passed") {
461
+ return {
462
+ suggestedCategories: [],
463
+ forbiddenCategories: [
464
+ ...forbidden,
465
+ "ProductBug",
466
+ "TestBug",
467
+ "EnvFailure",
468
+ ],
469
+ confidenceCap: 1,
470
+ };
471
+ }
472
+ // Assertion failures: ProductBug / TestBug / Unknown allowed; not Flaky.
473
+ suggested.add("ProductBug");
474
+ suggested.add("TestBug");
475
+ suggested.add("Unknown");
476
+ return {
477
+ suggestedCategories: [...suggested],
478
+ forbiddenCategories: [...forbidden],
479
+ confidenceCap: 0.75,
480
+ };
481
+ }
482
+ async function readPytestExitCode(runDir, fromNodeId) {
483
+ const exitPath = path.join(runDir, "reports", "backend-test-pytest-exit.txt");
484
+ try {
485
+ const raw = (await readFile(exitPath, "utf8")).trim();
486
+ const code = Number(raw);
487
+ if (raw !== "" && Number.isInteger(code) && code >= 0 && code <= 255)
488
+ return code;
489
+ throw new Error(`invalid pytest exit evidence at ${exitPath}`);
490
+ }
491
+ catch (error) {
492
+ if (error.code !== "ENOENT")
493
+ throw error;
494
+ // Missing dedicated evidence may fall through to an explicit node marker.
495
+ }
496
+ const nodePath = path.join(runDir, `${fromNodeId}.json`);
497
+ try {
498
+ const record = JSON.parse(await readFile(nodePath, "utf8"));
499
+ const blob = `${record.stdout ?? ""}\n${record.stderr ?? ""}`;
500
+ const marker = blob.match(/pytestExitCode\s*=\s*(\d{1,3})/i);
501
+ if (marker) {
502
+ const code = Number(marker[1]);
503
+ if (Number.isInteger(code) && code >= 0 && code <= 255)
504
+ return code;
505
+ }
506
+ }
507
+ catch (error) {
508
+ if (error.code !== "ENOENT") {
509
+ throw new Error("invalid pytest exit evidence in execute node record");
510
+ }
511
+ }
512
+ throw new Error("missing valid pytest exit evidence");
513
+ }
514
+ export async function materializeBackendTestResultFromRunDir(input) {
515
+ if (!/^[a-z0-9][a-z0-9._-]*\.json$/.test(input.artifactName) ||
516
+ !/^[a-z0-9][a-z0-9._-]*$/.test(input.outputDir)) {
517
+ throw new Error("unsafe structured artifact path");
518
+ }
519
+ const junitRelativePath = input.junitRelativePath ?? "reports/backend-test-junit.xml";
520
+ const junitAbs = path.join(input.runDir, ...junitRelativePath.split("/"));
521
+ let junitXml = null;
522
+ try {
523
+ junitXml = await readFile(junitAbs, "utf8");
524
+ }
525
+ catch {
526
+ junitXml = null;
527
+ }
528
+ if (junitXml === null) {
529
+ throw new Error(`missing junit report at ${junitRelativePath} (fail-closed for Result v1)`);
530
+ }
531
+ if (!junitXml.trim()) {
532
+ throw new Error("invalid junit xml: empty report");
533
+ }
534
+ const pytestExitCode = await readPytestExitCode(input.runDir, input.fromNodeId);
535
+ const commandSummary = "PYTHONDONTWRITEBYTECODE=1 python -m pytest testcase/ -v -p no:cacheprovider --junitxml=reports/backend-test-junit.xml";
536
+ let result;
537
+ try {
538
+ result = deriveBackendTestResult({
539
+ pytestExitCode,
540
+ junitXml,
541
+ junitRelativePath,
542
+ commandSummary,
543
+ });
544
+ }
545
+ catch (error) {
546
+ throw new Error(`invalid junit/report: ${error instanceof Error ? error.message : String(error)}`);
547
+ }
548
+ // Integrity: stored sha must match bytes we read.
549
+ validateBackendTestResultJunitIntegrity(result, junitXml);
550
+ const relativePath = path.posix.join(input.outputDir, input.artifactName);
551
+ const artifactPath = await writeDagRunJsonArtifact(input.runDir, relativePath, result);
552
+ const serialized = `${JSON.stringify(result, null, 2)}\n`;
553
+ return {
554
+ path: artifactPath,
555
+ sha256: createHash("sha256").update(serialized).digest("hex"),
556
+ schemaId: BACKEND_TEST_RESULT_SCHEMA_ID,
557
+ };
558
+ }
559
+ /** Shell snippet for backend-test-outcome-gate-shell (result.outcome authoritative). */
560
+ export function buildBackendTestOutcomeGateShellSnippet(options) {
561
+ const resultRelativePath = options?.resultRelativePath ?? "contracts/backend-test-result.json";
562
+ return [
563
+ 'test -n "${HARNESS_DAG_RUN_DIR:-}" || { echo "missing HARNESS_DAG_RUN_DIR for backend-test outcome gate" >&2; exit 2; }',
564
+ `RESULT="\${HARNESS_DAG_RUN_DIR}/${resultRelativePath}"`,
565
+ 'test -f "${RESULT}" || { echo "missing backend-test result: ${RESULT}" >&2; exit 2; }',
566
+ `node -e 'const fs=require("fs");const r=JSON.parse(fs.readFileSync(process.argv[1],"utf8"));const outcome=String(r.outcome||"");const ok=outcome==="passed"&&Number(r.failed||0)===0&&Number(r.error||0)===0;console.log("backend-test outcome="+outcome+" passed="+r.passed+" failed="+r.failed+" error="+r.error+" executionStatus="+r.executionStatus);if(!ok){process.exit(1);}' "\${RESULT}"`,
567
+ ].join("; ");
568
+ }
@@ -93,14 +93,69 @@ export const decisionEnvelopeSchema = z
93
93
  .passthrough(),
94
94
  })
95
95
  .strict();
96
- const DECISION_ENVELOPE_FENCE_RE = /```DECISION_ENVELOPE_JSON[ \t]*\r?\n([\s\S]*?)\r?\n```/g;
96
+ /**
97
+ * Canonical schema-valid and semantic-valid payload embedded in supervised
98
+ * Decision Gate prompts. Keep this example minimal and verify it through the
99
+ * real parser so the prompt contract cannot silently drift from runtime rules.
100
+ */
101
+ export const DECISION_ENVELOPE_CANONICAL_EXAMPLE = {
102
+ schemaVersion: 1,
103
+ gateType: "acceptance-gate",
104
+ decisionScope: "dag-run",
105
+ decision: "auto-approve",
106
+ confidence: 0.9,
107
+ riskLevel: "low",
108
+ requiresHuman: false,
109
+ nextAction: "continue",
110
+ policyVersion: "agent-dag-decision-gate-v1",
111
+ policyChecks: {
112
+ mustEscalateFlags: [],
113
+ evidenceComplete: true,
114
+ allowedAutoApprove: true,
115
+ },
116
+ rationale: ["Deterministic verification passed."],
117
+ evidence: [
118
+ {
119
+ path: "docs/templates/agent-dag-decision-envelope.schema.json",
120
+ kind: "schema",
121
+ status: "verified",
122
+ summary: "Schema and deterministic verification evidence were reviewed.",
123
+ },
124
+ ],
125
+ blockingFindings: [],
126
+ requiredRevisions: [],
127
+ riskFlags: [],
128
+ humanEscalation: null,
129
+ audit: {
130
+ runId: "<current-run-id>",
131
+ nodeId: "decision-pi",
132
+ model: "<current-node-model>",
133
+ },
134
+ };
135
+ /** Build the strict structured-output contract injected into decision-pi. */
136
+ export function buildDecisionEnvelopePromptContract() {
137
+ return [
138
+ "Return exactly one Decision Envelope using the current strict schema.",
139
+ "Use only these decision values: auto-approve, approve-with-constraints, request-revision, run-more-verification, split-followup, reject, escalate-to-human, pause-wait-external.",
140
+ "Use only these nextAction values: continue, rerun-implement, rerun-verify, run-targeted-check, split-followup, pause-and-ask, abort.",
141
+ "When deterministic verification and review pass with no blocking findings, use decision=auto-approve, requiresHuman=false, nextAction=continue.",
142
+ "Replace audit.runId and audit.model placeholders with the current DAG run id and current node model. Keep audit.nodeId equal to the current Decision Gate node id.",
143
+ "Do not emit legacy root fields taskId, gate, verdict, summary, invariants, or findings. Do not use invented decision values such as proceed-to-closeout, accept, or proceed. Unknown fields and enum values fail validation.",
144
+ "The authoritative schema is docs/templates/agent-dag-decision-envelope.schema.json.",
145
+ "Output this exact fenced shape with a schema-valid payload:",
146
+ `\`\`\`${DECISION_ENVELOPE_FENCE_INFO}`,
147
+ JSON.stringify(DECISION_ENVELOPE_CANONICAL_EXAMPLE, null, 2),
148
+ "```",
149
+ ].join("\n");
150
+ }
151
+ const DECISION_ENVELOPE_FENCE_RE = /(?<!`)(`{3,})DECISION_ENVELOPE_JSON[ \t]*\r?\n([\s\S]*?)\r?\n\1[ \t]*(?=\r?\n|$)/g;
97
152
  export function extractDecisionEnvelopeFencedBlocks(text) {
98
153
  const blocks = [];
99
154
  if (!text)
100
155
  return blocks;
101
156
  const re = new RegExp(DECISION_ENVELOPE_FENCE_RE.source, DECISION_ENVELOPE_FENCE_RE.flags);
102
157
  for (const match of text.matchAll(re)) {
103
- const body = match[1]?.trim();
158
+ const body = match[2]?.trim();
104
159
  if (body)
105
160
  blocks.push(body);
106
161
  }