@tea-agent/loop-agent 0.13.0-beta.0 → 0.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (282) hide show
  1. package/AGENTS.md +157 -155
  2. package/CHANGELOG.md +301 -322
  3. package/README.md +335 -345
  4. package/bin/agent-worker.js +22 -22
  5. package/bin/loop-agent.js +21 -21
  6. package/dist/commands/cursor-prompt.js +6 -6
  7. package/dist/commands/init.js +597 -528
  8. package/dist/commands/loop-benchmark.js +11 -11
  9. package/dist/commands/pi-reuse-benchmark.js +16 -16
  10. package/dist/executors/shell-executor.js +200 -21
  11. package/dist/infrastructure/evaluation/candidate-store.js +5 -1
  12. package/dist/sidecars/cursor-prompt/executor.js +1 -1
  13. package/dist/task/runtime.js +27 -27
  14. package/dist/worker/observe/static/api.js +46 -46
  15. package/dist/worker/observe/static/app.js +150 -150
  16. package/dist/worker/observe/static/constants.js +148 -148
  17. package/dist/worker/observe/static/copy.js +67 -67
  18. package/dist/worker/observe/static/dag-helpers.js +172 -172
  19. package/dist/worker/observe/static/dag-layout.d.ts +31 -31
  20. package/dist/worker/observe/static/dag-layout.js +83 -83
  21. package/dist/worker/observe/static/dag-model.js +72 -72
  22. package/dist/worker/observe/static/dom.js +212 -53
  23. package/dist/worker/observe/static/format-pool.js +67 -67
  24. package/dist/worker/observe/static/format.js +292 -292
  25. package/dist/worker/observe/static/index.html +308 -308
  26. package/dist/worker/observe/static/kpi.js +94 -94
  27. package/dist/worker/observe/static/relations.js +133 -133
  28. package/dist/worker/observe/static/router.js +93 -93
  29. package/dist/worker/observe/static/run-processing.js +148 -148
  30. package/dist/worker/observe/static/shell-chrome.js +68 -68
  31. package/dist/worker/observe/static/state.js +267 -253
  32. package/dist/worker/observe/static/styles.css +1902 -1902
  33. package/dist/worker/observe/static/views/batch.js +227 -227
  34. package/dist/worker/observe/static/views/dag-graph.js +172 -172
  35. package/dist/worker/observe/static/views/dag-inspector.js +627 -607
  36. package/dist/worker/observe/static/views/dag.js +371 -362
  37. package/dist/worker/observe/static/views/dashboard.js +509 -252
  38. package/dist/worker/observe/static/views/failures.js +143 -143
  39. package/dist/worker/observe/static/views/feature.js +492 -492
  40. package/dist/worker/observe/static/views/pool.js +350 -350
  41. package/dist/worker/observe/static/views/run.js +453 -453
  42. package/dist/worker/observe/static/views/session-timeline.js +219 -205
  43. package/dist/worker/observe/static/views/shell.js +7 -7
  44. package/dist/worker/observe/static/views/task.js +314 -314
  45. package/dist/worker/observe/static/views/timeline.js +163 -163
  46. package/dist/workflows/dag/backend-test-case-manifest.js +503 -0
  47. package/dist/workflows/dag/backend-test-execution-contract.js +353 -0
  48. package/dist/workflows/dag/backend-test-result-contract.js +568 -0
  49. package/dist/workflows/dag/canvas-observer.js +275 -275
  50. package/dist/workflows/dag/decision-envelope.js +57 -2
  51. package/dist/workflows/dag/frontend-implementation-contract.js +240 -0
  52. package/dist/workflows/dag/frontend-project-capability.js +309 -0
  53. package/dist/workflows/dag/frontend-repair.js +341 -0
  54. package/dist/workflows/dag/frontend-risk.js +161 -0
  55. package/dist/workflows/dag/frontend-verification-trace.js +190 -0
  56. package/dist/workflows/dag/init-hybrid.js +1020 -125
  57. package/dist/workflows/dag/repair-artifact.js +43 -3
  58. package/dist/workflows/dag/skill-instructions.js +4 -2
  59. package/dist/workflows/dag/types.js +29 -8
  60. package/docs/README.md +105 -104
  61. package/docs/agent-dag-recovery-playbook.md +195 -195
  62. package/docs/agent-dag-runner.md +67 -67
  63. package/docs/architecture/README.md +26 -26
  64. package/docs/architecture/dag-execution.md +140 -140
  65. package/docs/architecture/evolution.md +54 -54
  66. package/docs/architecture/facts-and-state.md +71 -71
  67. package/docs/architecture/runtime-boundaries.md +191 -191
  68. package/docs/architecture/system-overview.md +93 -93
  69. package/docs/architecture/worker-and-feature.md +85 -85
  70. package/docs/cursor-prompt-sidecar.md +36 -36
  71. package/docs/decisions/README.md +18 -18
  72. package/docs/design/README.md +167 -167
  73. package/docs/development-principles.md +73 -73
  74. package/docs/exec-plans/README.md +6 -6
  75. package/docs/exec-plans/active/README.md +1 -4
  76. package/docs/exec-plans/completed/README.md +106 -84
  77. package/docs/feature-workflow.md +414 -389
  78. package/docs/harness-methodology-debugging.md +153 -153
  79. package/docs/harness-methodology-tdd.md +130 -130
  80. package/docs/harness-methodology-verification.md +27 -27
  81. package/docs/init-surface.manifest.json +307 -289
  82. package/docs/loop-agent-harness.md +142 -142
  83. package/docs/production-readiness.md +96 -96
  84. package/docs/progress/README.md +76 -60
  85. package/docs/reports/README.md +150 -108
  86. package/docs/skills/README.md +7 -7
  87. package/docs/skills/vetted-skill-registry.md +29 -29
  88. package/docs/templates/adr.md +60 -60
  89. package/docs/templates/agent-dag-authority-surface-audit.prompt.md +94 -94
  90. package/docs/templates/agent-dag-decision-envelope.schema.json +213 -213
  91. package/docs/templates/agent-dag-decision-gate-dogfood-report.md +117 -117
  92. package/docs/templates/agent-dag-decision-gate.prompt.md +246 -246
  93. package/docs/templates/agent-dag-process-supervisor.prompt.md +98 -98
  94. package/docs/templates/agent-dag-report.schema.json +473 -473
  95. package/docs/templates/agent-dag-review-verdict.prompt.md +68 -68
  96. package/docs/templates/agent-dag.base.json +190 -190
  97. package/docs/templates/agent-dag.final-verification.json +185 -185
  98. package/docs/templates/agent-dag.schema.json +411 -411
  99. package/docs/templates/agent-dag.supervised-implementation.json +620 -501
  100. package/docs/templates/backend-test-analysis.schema.json +44 -44
  101. package/docs/templates/backend-test-case-manifest.schema.json +190 -0
  102. package/docs/templates/backend-test-dag.classify.prompt.md +75 -0
  103. package/docs/templates/backend-test-dag.generate-pytest.prompt.md +204 -202
  104. package/docs/templates/backend-test-dag.json +559 -311
  105. package/docs/templates/backend-test-dag.retrospect.prompt.md +139 -125
  106. package/docs/templates/backend-test-dag.review-cases.prompt.md +83 -81
  107. package/docs/templates/backend-test-execution.schema.json +133 -0
  108. package/docs/templates/backend-test-result.schema.json +99 -0
  109. package/docs/templates/branch-merge-report.md +93 -0
  110. package/docs/templates/exec-plan.md +64 -64
  111. package/docs/templates/feature-spec.md +53 -53
  112. package/docs/templates/frontend-design-contract.md +42 -42
  113. package/docs/templates/frontend-eval/fixtures/failures/01-type-build-error.md +17 -0
  114. package/docs/templates/frontend-eval/fixtures/failures/02-unit-component-test-fail.md +16 -0
  115. package/docs/templates/frontend-eval/fixtures/failures/03-fixture-schema-drift.md +16 -0
  116. package/docs/templates/frontend-eval/fixtures/failures/04-missing-loading-empty-error-state.md +16 -0
  117. package/docs/templates/frontend-eval/fixtures/failures/05-forbidden-write-writeset-expansion.md +16 -0
  118. package/docs/templates/frontend-eval/fixtures/failures/06-unapproved-dependency-add.md +16 -0
  119. package/docs/templates/frontend-eval/fixtures/failures/07-mock-production-on.md +21 -0
  120. package/docs/templates/frontend-eval/fixtures/functional/01-simple-component-style.md +29 -0
  121. package/docs/templates/frontend-eval/fixtures/functional/02-form-validation.md +28 -0
  122. package/docs/templates/frontend-eval/fixtures/functional/03-list-detail-page.md +28 -0
  123. package/docs/templates/frontend-eval/fixtures/functional/04-api-mock.md +29 -0
  124. package/docs/templates/frontend-eval/fixtures/functional/05-permission-auth-gated-ui.md +27 -0
  125. package/docs/templates/frontend-eval/fixtures/functional/06-ssr-server-client-boundary.md +28 -0
  126. package/docs/templates/frontend-eval/fixtures/functional/07-shared-public-component-api.md +28 -0
  127. package/docs/templates/frontend-eval/fixtures/functional/08-pure-local-no-remote.md +27 -0
  128. package/docs/templates/frontend-eval/metrics.md +138 -0
  129. package/docs/templates/frontend-eval/smoke-targets.md +53 -0
  130. package/docs/templates/frontend-implementation-contract.schema.json +27 -0
  131. package/docs/templates/frontend-task-constraints.md +35 -35
  132. package/docs/templates/frontend-task-requirement.md +70 -70
  133. package/docs/templates/frontend-test-dag.generate-cases.prompt.md +5 -5
  134. package/docs/templates/frontend-test-dag.json +23 -23
  135. package/docs/templates/frontend-test-dag.retrieve-context.prompt.md +3 -3
  136. package/docs/templates/frontend-test-dag.retrospect.prompt.md +3 -3
  137. package/docs/templates/frontend-test-dag.review-cases.prompt.md +3 -3
  138. package/docs/templates/frontend-test-dag.review-execution.prompt.md +3 -3
  139. package/docs/templates/harness.schema.json +221 -221
  140. package/docs/templates/hybrid-dag.json +188 -188
  141. package/docs/templates/init-evolution-review.md +35 -35
  142. package/docs/templates/interactive-ui-round2-experiment.md +66 -66
  143. package/docs/templates/knowledge-graph-bootstrap-dag.json +118 -118
  144. package/docs/templates/knowledge-sync-dag.json +178 -178
  145. package/docs/templates/knowledge-sync-draft.schema.json +71 -71
  146. package/docs/templates/product-line/AGENTS.md +8 -8
  147. package/docs/templates/product-line/README.md +9 -9
  148. package/docs/templates/product-line/acceptance.yaml +14 -14
  149. package/docs/templates/product-line/closeout.yaml +9 -9
  150. package/docs/templates/product-line/design.md +13 -13
  151. package/docs/templates/product-line/links.md +10 -10
  152. package/docs/templates/product-line/requirement.md +17 -17
  153. package/docs/templates/product-line/task-graph.yaml +15 -15
  154. package/docs/templates/product-line/task.yaml +64 -64
  155. package/docs/templates/product-line/test-plan.md +7 -7
  156. package/docs/templates/production-readiness-checklist.md +57 -57
  157. package/docs/templates/progress-log.md +17 -17
  158. package/docs/templates/project-start-checklist.md +9 -9
  159. package/docs/templates/qa-report.md +48 -48
  160. package/docs/templates/sprint-contract.md +29 -29
  161. package/docs/templates/worker-dogfood-evidence.md +80 -80
  162. package/docs/templates/worker-dogfood-setup.md +68 -68
  163. package/docs/verification-matrix.md +70 -70
  164. package/examples/decision-gate-agent-dag.json +177 -177
  165. package/examples/example-dag.json +46 -46
  166. package/examples/hybrid-loop-agent-dag.json +189 -189
  167. package/harness.json +66 -66
  168. package/package.json +52 -88
  169. package/scripts/check-product-line-docs.sh +29 -29
  170. package/scripts/check-task-pool-root.sh +32 -32
  171. package/scripts/kb-bootstrap-init-skeleton.sh +240 -240
  172. package/scripts/kb-graph-incremental-prepare.mjs +386 -386
  173. package/scripts/kb-graph-incremental-prepare.sh +5 -5
  174. package/scripts/kb-graph-materialize.mjs +105 -105
  175. package/scripts/kb-graph-materialize.sh +4 -4
  176. package/scripts/kb-graph-promote.mjs +164 -164
  177. package/scripts/kb-graph-promote.sh +4 -4
  178. package/scripts/kb-query.mjs +554 -554
  179. package/scripts/kb-query.sh +5 -5
  180. package/skills/agent-worker/SKILL.md +39 -39
  181. package/skills/agent-worker/references/agent-worker-operator.md +60 -60
  182. package/skills/ai-engineering-context/SKILL.md +48 -48
  183. package/skills/analyze-product-dependencies/SKILL.md +67 -67
  184. package/skills/analyze-product-dependencies/agents/openai.yaml +4 -4
  185. package/skills/analyze-product-dependencies/references/api-documentation-schema.md +30 -30
  186. package/skills/analyze-product-dependencies/references/dependency-analysis-schema.md +28 -28
  187. package/skills/analyze-product-dependencies/references/example.md +76 -76
  188. package/skills/analyze-product-dependencies/references/forward-test-cases.md +35 -35
  189. package/skills/analyze-product-dependencies/references/input-contract.md +11 -11
  190. package/skills/analyze-product-dependencies/references/scouting-rules.md +61 -61
  191. package/skills/analyze-product-dependencies/scripts/test-validators.mjs +267 -267
  192. package/skills/analyze-product-dependencies/scripts/validate-api-documentation.mjs +101 -101
  193. package/skills/analyze-product-dependencies/scripts/validate-dependency-analysis.mjs +142 -142
  194. package/skills/analyze-product-dependencies/scripts/validate-product-requirement-input.mjs +76 -76
  195. package/skills/analyze-product-dependencies/scripts/validation-helpers.mjs +146 -146
  196. package/skills/analyze-product-requirements/SKILL.md +90 -90
  197. package/skills/analyze-product-requirements/agents/openai.yaml +4 -4
  198. package/skills/analyze-product-requirements/references/acceptance-criteria.md +91 -91
  199. package/skills/analyze-product-requirements/references/clarification-and-knowledge.md +56 -56
  200. package/skills/analyze-product-requirements/references/example.md +86 -86
  201. package/skills/analyze-product-requirements/references/forward-test-cases.md +66 -66
  202. package/skills/analyze-product-requirements/references/product-analysis-schema.md +32 -32
  203. package/skills/analyze-product-requirements/references/product-requirement-schema.md +33 -33
  204. package/skills/analyze-product-requirements/references/requirement-clarification-schema.md +35 -35
  205. package/skills/analyze-product-requirements/scripts/test-validators.mjs +193 -193
  206. package/skills/analyze-product-requirements/scripts/validate-product-analysis.mjs +69 -69
  207. package/skills/analyze-product-requirements/scripts/validate-product-requirement.mjs +97 -97
  208. package/skills/analyze-product-requirements/scripts/validate-requirement-clarification.mjs +98 -98
  209. package/skills/analyze-product-requirements/scripts/validation-helpers.mjs +156 -156
  210. package/skills/browser-tools/SKILL.md +196 -0
  211. package/skills/browser-tools/browser-content.js +103 -0
  212. package/skills/browser-tools/browser-cookies.js +35 -0
  213. package/skills/browser-tools/browser-eval.js +53 -0
  214. package/skills/browser-tools/browser-hn-scraper.js +108 -0
  215. package/skills/browser-tools/browser-nav.js +44 -0
  216. package/skills/browser-tools/browser-pick.js +162 -0
  217. package/skills/browser-tools/browser-screenshot.js +34 -0
  218. package/skills/browser-tools/browser-start.js +86 -0
  219. package/skills/browser-tools/package-lock.json +2556 -0
  220. package/skills/browser-tools/package.json +19 -0
  221. package/skills/code-review-core/SKILL.md +20 -20
  222. package/skills/codebase-scout/SKILL.md +19 -19
  223. package/skills/frontend-design-review/SKILL.md +66 -66
  224. package/skills/frontend-design-review/references/review-checklist.md +58 -58
  225. package/skills/frontend-implementation/SKILL.md +49 -47
  226. package/skills/frontend-implementation/references/code-standards.md +32 -32
  227. package/skills/frontend-implementation/references/design-spec.md +46 -46
  228. package/skills/frontend-implementation/references/node-contracts.md +27 -76
  229. package/skills/frontend-review/SKILL.md +59 -59
  230. package/skills/frontend-review/references/review-findings.md +47 -47
  231. package/skills/frontend-verification/SKILL.md +53 -53
  232. package/skills/frontend-verification/references/verification-checklist.md +68 -68
  233. package/skills/grill-me/SKILL.md +10 -10
  234. package/skills/grill-with-docs/SKILL.md +88 -88
  235. package/skills/grill-with-docs/adr-format.md +47 -47
  236. package/skills/grill-with-docs/context-format.md +60 -60
  237. package/skills/init-capability-evolution/SKILL.md +70 -70
  238. package/skills/loop-agent/SKILL.md +151 -151
  239. package/skills/loop-agent/references/README.md +67 -67
  240. package/skills/loop-agent/references/command-reference.md +527 -505
  241. package/skills/loop-agent/references/docs-converge.md +126 -126
  242. package/skills/loop-agent/references/harness-policy.md +263 -263
  243. package/skills/loop-agent/references/hybrid-dag.md +243 -238
  244. package/skills/loop-agent/references/learned/README.md +21 -21
  245. package/skills/loop-agent/references/long-running-loop.md +57 -57
  246. package/skills/loop-agent/references/model-routing.md +36 -36
  247. package/skills/loop-agent/references/multi-worktree.md +54 -54
  248. package/skills/loop-agent/references/one-shot-runs.md +85 -85
  249. package/skills/loop-agent/references/orchestrator-and-interventions.md +169 -169
  250. package/skills/loop-agent/references/pi-prompt.md +23 -23
  251. package/skills/loop-agent/references/pi-subagent-assisted-mode.md +84 -84
  252. package/skills/loop-agent/references/post-implementation-and-patterns.md +44 -44
  253. package/skills/loop-agent/references/task-workflow.md +89 -89
  254. package/skills/loop-agent/references/verification-and-failure-handling.md +141 -139
  255. package/skills/playwright-cli/SKILL.md +420 -420
  256. package/skills/playwright-cli/references/element-attributes.md +23 -23
  257. package/skills/playwright-cli/references/playwright-tests.md +39 -39
  258. package/skills/playwright-cli/references/request-mocking.md +87 -87
  259. package/skills/playwright-cli/references/running-code.md +241 -241
  260. package/skills/playwright-cli/references/session-management.md +225 -225
  261. package/skills/playwright-cli/references/storage-state.md +275 -275
  262. package/skills/playwright-cli/references/test-generation.md +433 -433
  263. package/skills/playwright-cli/references/tracing.md +139 -139
  264. package/skills/playwright-cli/references/video-recording.md +143 -143
  265. package/skills/playwright-cli-case-generator/SKILL.md +74 -74
  266. package/skills/requesting-code-review/SKILL.md +101 -101
  267. package/skills/requesting-code-review/code-reviewer.md +168 -168
  268. package/skills/systematic-debugging/CREATION-LOG.md +119 -119
  269. package/skills/systematic-debugging/SKILL.md +296 -296
  270. package/skills/systematic-debugging/condition-based-waiting-example.ts +158 -158
  271. package/skills/systematic-debugging/condition-based-waiting.md +115 -115
  272. package/skills/systematic-debugging/defense-in-depth.md +122 -122
  273. package/skills/systematic-debugging/find-polluter.sh +63 -63
  274. package/skills/systematic-debugging/root-cause-tracing.md +169 -169
  275. package/skills/systematic-debugging/test-academic.md +14 -14
  276. package/skills/systematic-debugging/test-pressure-1.md +58 -58
  277. package/skills/systematic-debugging/test-pressure-2.md +68 -68
  278. package/skills/systematic-debugging/test-pressure-3.md +69 -69
  279. package/skills/test-driven-development/SKILL.md +20 -20
  280. package/skills/using-git-worktrees/SKILL.md +215 -215
  281. package/skills/verification-before-completion/SKILL.md +154 -154
  282. package/skills/webapp-testing/SKILL.md +19 -19
@@ -0,0 +1,568 @@
1
+ import { createHash } from "node:crypto";
2
+ import { readFile } from "node:fs/promises";
3
+ import path from "node:path";
4
+ import { z } from "zod";
5
+ import { writeDagRunJsonArtifact } from "../../infrastructure/harness/artifact-store.js";
6
+ export const BACKEND_TEST_RESULT_SCHEMA_ID = "backend-test-result-v1";
7
+ const SECRET_KEY = /(?:password|passwd|secret|token|api[_-]?key|private[_-]?key|credential|authorization)/i;
8
+ const SECRET_VALUE = /(?:-----BEGIN [A-Z ]*PRIVATE KEY-----|\b(?:sk|ghp|github_pat|xox[baprs]|AKIA)[-_A-Za-z0-9]{12,}\b)/;
9
+ export const BACKEND_TEST_CLASSIFICATION_CATEGORIES = [
10
+ "ProductBug",
11
+ "TestBug",
12
+ "EnvFailure",
13
+ "ContractMismatch",
14
+ "FlakyTest",
15
+ "Unknown",
16
+ ];
17
+ export const backendTestExecutionStatusSchema = z.enum([
18
+ "completed",
19
+ "collection-error",
20
+ "command-error",
21
+ "report-error",
22
+ ]);
23
+ export const backendTestCollectionStatusSchema = z.enum([
24
+ "ok",
25
+ "error",
26
+ "unknown",
27
+ "skipped",
28
+ ]);
29
+ /**
30
+ * Task Pool consumable outcome tokens (frozen for M2; no auto follow-up).
31
+ * - passed: all executed tests green
32
+ * - completed-with-failures: process completed with assertion/test failures
33
+ * - collection-error: import/collection failures
34
+ * - command-error: runner/command could not complete normally
35
+ * - report-error: junit missing/corrupt after execution attempt
36
+ */
37
+ export const backendTestOutcomeSchema = z.enum([
38
+ "passed",
39
+ "completed-with-failures",
40
+ "collection-error",
41
+ "command-error",
42
+ "report-error",
43
+ ]);
44
+ const junitRelativePathSchema = z
45
+ .string()
46
+ .min(1)
47
+ .refine((value) => !path.posix.isAbsolute(value) &&
48
+ !value.includes("\\") &&
49
+ value
50
+ .split("/")
51
+ .every((segment) => segment.length > 0 && segment !== "." && segment !== ".."), { message: "junit.relativePath must be a safe run-relative POSIX path" });
52
+ const failureSummarySchema = z
53
+ .object({
54
+ classname: z.string().min(1),
55
+ name: z.string().min(1),
56
+ message: z.string().min(1),
57
+ kind: z.enum(["failure", "error"]).default("failure"),
58
+ })
59
+ .strict();
60
+ export const backendTestResultContractSchema = z
61
+ .object({
62
+ schemaVersion: z.literal(1),
63
+ /** Process-level status of the pytest invocation. Task Pool: use with outcome. */
64
+ executionStatus: backendTestExecutionStatusSchema,
65
+ pytestExitCode: z.number().int().min(0).max(255),
66
+ collectionStatus: backendTestCollectionStatusSchema,
67
+ /** Total tests counted from JUnit (or 0 when report unavailable). */
68
+ tests: z.number().int().min(0),
69
+ passed: z.number().int().min(0),
70
+ failed: z.number().int().min(0),
71
+ error: z.number().int().min(0),
72
+ skipped: z.number().int().min(0),
73
+ durationMs: z.number().nonnegative().optional(),
74
+ junit: z
75
+ .object({
76
+ relativePath: junitRelativePathSchema,
77
+ sha256: z
78
+ .string()
79
+ .regex(/^[a-f0-9]{64}$/, "junit.sha256 must be lowercase hex sha256"),
80
+ })
81
+ .strict(),
82
+ /** Non-secret command summary only (no tokens/passwords). */
83
+ commandSummary: z.string().min(1),
84
+ /** Truncated failure/error summaries for Task Pool triage. */
85
+ failures: z.array(failureSummarySchema),
86
+ /**
87
+ * Authoritative shell-facing outcome for backend-test-outcome-gate-shell.
88
+ * Do not override from retrospective Markdown.
89
+ */
90
+ outcome: backendTestOutcomeSchema,
91
+ })
92
+ .strict()
93
+ .superRefine((value, ctx) => {
94
+ if (value.tests !==
95
+ value.passed + value.failed + value.error + value.skipped) {
96
+ ctx.addIssue({
97
+ code: z.ZodIssueCode.custom,
98
+ message: "tests must equal passed+failed+error+skipped",
99
+ path: ["tests"],
100
+ });
101
+ }
102
+ if (SECRET_KEY.test(value.commandSummary) ||
103
+ SECRET_VALUE.test(value.commandSummary)) {
104
+ ctx.addIssue({
105
+ code: z.ZodIssueCode.custom,
106
+ message: "commandSummary must not contain secret-like content",
107
+ path: ["commandSummary"],
108
+ });
109
+ }
110
+ for (const [index, failure] of value.failures.entries()) {
111
+ if (SECRET_VALUE.test(failure.message)) {
112
+ ctx.addIssue({
113
+ code: z.ZodIssueCode.custom,
114
+ message: "failure message must not contain secret-like content",
115
+ path: ["failures", index, "message"],
116
+ });
117
+ }
118
+ }
119
+ });
120
+ const MAX_FAILURE_MESSAGE = 500;
121
+ const MAX_FAILURES = 50;
122
+ function decodeXmlEntities(value) {
123
+ return value
124
+ .replace(/&lt;/g, "<")
125
+ .replace(/&gt;/g, ">")
126
+ .replace(/&quot;/g, '"')
127
+ .replace(/&apos;/g, "'")
128
+ .replace(/&amp;/g, "&");
129
+ }
130
+ const XML_ATTR_PATTERNS = {
131
+ tests: /\btests\s*=\s*"([^"]*)"/i,
132
+ failures: /\bfailures\s*=\s*"([^"]*)"/i,
133
+ errors: /\berrors\s*=\s*"([^"]*)"/i,
134
+ skipped: /\bskipped\s*=\s*"([^"]*)"/i,
135
+ time: /\btime\s*=\s*"([^"]*)"/i,
136
+ classname: /\bclassname\s*=\s*"([^"]*)"/i,
137
+ name: /\bname\s*=\s*"([^"]*)"/i,
138
+ message: /\bmessage\s*=\s*"([^"]*)"/i,
139
+ };
140
+ function attr(tag, name) {
141
+ const pattern = XML_ATTR_PATTERNS[name];
142
+ if (!pattern)
143
+ return undefined;
144
+ const match = tag.match(pattern);
145
+ return match ? decodeXmlEntities(match[1] ?? "") : undefined;
146
+ }
147
+ function truncate(value, max = MAX_FAILURE_MESSAGE) {
148
+ const normalized = value.replace(/\s+/g, " ").trim();
149
+ if (normalized.length <= max)
150
+ return normalized;
151
+ return `${normalized.slice(0, max - 1)}…`;
152
+ }
153
+ function assertNoSecrets(label, value) {
154
+ if (SECRET_KEY.test(value) || SECRET_VALUE.test(value)) {
155
+ throw new Error(`secret-like content rejected in ${label}`);
156
+ }
157
+ }
158
+ /**
159
+ * Minimal JUnit XML parser (no new npm deps). Fail-closed on non-JUnit markup.
160
+ */
161
+ export function parseJunitXml(xml) {
162
+ const trimmed = xml.trim();
163
+ if (!trimmed) {
164
+ throw new Error("invalid junit xml: empty");
165
+ }
166
+ if (!/<testsuites\b/i.test(trimmed) && !/<testsuite\b/i.test(trimmed)) {
167
+ throw new Error("invalid junit xml: missing testsuites/testsuite root");
168
+ }
169
+ const hasSuitesRoot = /<testsuites\b/i.test(trimmed);
170
+ if ((hasSuitesRoot && !/<\/testsuites>\s*$/i.test(trimmed)) ||
171
+ (!hasSuitesRoot && !/<\/testsuite>\s*$/i.test(trimmed))) {
172
+ throw new Error("invalid junit xml: unclosed root element");
173
+ }
174
+ const openedCases = (trimmed.match(/<testcase\b/gi) ?? []).length;
175
+ const closedCases = (trimmed.match(/<\/testcase>/gi) ?? []).length;
176
+ const selfClosingCases = (trimmed.match(/<testcase\b[^>]*\/>/gi) ?? [])
177
+ .length;
178
+ if (openedCases !== closedCases + selfClosingCases) {
179
+ throw new Error("invalid junit xml: unclosed testcase element");
180
+ }
181
+ // Prefer root testsuites aggregates when present.
182
+ const suitesOpen = trimmed.match(/<testsuites\b[^>]*>/i)?.[0];
183
+ let tests = 0;
184
+ let failed = 0;
185
+ let errors = 0;
186
+ let skipped = 0;
187
+ let timeSec;
188
+ if (suitesOpen) {
189
+ const t = attr(suitesOpen, "tests");
190
+ const f = attr(suitesOpen, "failures");
191
+ const e = attr(suitesOpen, "errors");
192
+ const s = attr(suitesOpen, "skipped");
193
+ const time = attr(suitesOpen, "time");
194
+ if (t !== undefined)
195
+ tests = Number(t);
196
+ if (f !== undefined)
197
+ failed = Number(f);
198
+ if (e !== undefined)
199
+ errors = Number(e);
200
+ if (s !== undefined)
201
+ skipped = Number(s);
202
+ if (time !== undefined && time !== "")
203
+ timeSec = Number(time);
204
+ }
205
+ // Self-closing first so empty cases ending with /> are not greedily paired with a later </testcase>.
206
+ const caseRe = /<testcase\b([^>]*?)\/>|<testcase\b([^>]*)>([\s\S]*?)<\/testcase>/gi;
207
+ const failures = [];
208
+ let caseCount = 0;
209
+ let caseFailed = 0;
210
+ let caseErrors = 0;
211
+ let caseSkipped = 0;
212
+ let match = caseRe.exec(trimmed);
213
+ while (match !== null) {
214
+ caseCount += 1;
215
+ const openAttrs = match[1] ?? match[2] ?? "";
216
+ const body = match[3] ?? "";
217
+ const classname = attr(openAttrs, "classname") || "unknown";
218
+ const name = attr(openAttrs, "name") || "unknown";
219
+ const failureTag = body.match(/<failure\b([^>]*)>([\s\S]*?)<\/failure>|<failure\b([^>]*)\/>/i);
220
+ const errorTag = body.match(/<error\b([^>]*)>([\s\S]*?)<\/error>|<error\b([^>]*)\/>/i);
221
+ const skippedTag = /<skipped\b/i.test(body);
222
+ if (failureTag) {
223
+ caseFailed += 1;
224
+ const fAttrs = failureTag[1] ?? failureTag[3] ?? "";
225
+ const fBody = failureTag[2] ?? "";
226
+ const message = attr(fAttrs, "message") || fBody || "failure";
227
+ failures.push({
228
+ classname,
229
+ name,
230
+ message: truncate(message),
231
+ kind: "failure",
232
+ });
233
+ }
234
+ else if (errorTag) {
235
+ caseErrors += 1;
236
+ const eAttrs = errorTag[1] ?? errorTag[3] ?? "";
237
+ const eBody = errorTag[2] ?? "";
238
+ const message = attr(eAttrs, "message") || eBody || "error";
239
+ failures.push({
240
+ classname,
241
+ name,
242
+ message: truncate(message),
243
+ kind: "error",
244
+ });
245
+ }
246
+ else if (skippedTag) {
247
+ caseSkipped += 1;
248
+ }
249
+ match = caseRe.exec(trimmed);
250
+ }
251
+ // Prefer case-level facts when testcase elements exist; never trust root counters
252
+ // that hide <failure>/<error> children (fail-closed on contradictory aggregates).
253
+ if (caseCount > 0) {
254
+ if (suitesOpen) {
255
+ const declaredTests = attr(suitesOpen, "tests") !== undefined
256
+ ? Number(attr(suitesOpen, "tests"))
257
+ : undefined;
258
+ const declaredFailed = attr(suitesOpen, "failures") !== undefined
259
+ ? Number(attr(suitesOpen, "failures"))
260
+ : undefined;
261
+ const declaredErrors = attr(suitesOpen, "errors") !== undefined
262
+ ? Number(attr(suitesOpen, "errors"))
263
+ : undefined;
264
+ const declaredSkipped = attr(suitesOpen, "skipped") !== undefined
265
+ ? Number(attr(suitesOpen, "skipped"))
266
+ : undefined;
267
+ // pytest collection reports often use tests="0" with a synthetic error testcase.
268
+ // Always fail-closed when root counters under-report failure/error children.
269
+ if (declaredTests !== undefined &&
270
+ !Number.isNaN(declaredTests) &&
271
+ declaredTests !== caseCount) {
272
+ const collectionStyle = declaredTests === 0 && caseErrors > 0 && caseFailed === 0;
273
+ if (!collectionStyle) {
274
+ throw new Error(`invalid junit xml: testsuites tests=${declaredTests} disagrees with testcase count=${caseCount}`);
275
+ }
276
+ }
277
+ if (declaredFailed !== undefined &&
278
+ !Number.isNaN(declaredFailed) &&
279
+ declaredFailed !== caseFailed) {
280
+ throw new Error(`invalid junit xml: testsuites failures=${declaredFailed} disagrees with case failures=${caseFailed}`);
281
+ }
282
+ if (declaredErrors !== undefined &&
283
+ !Number.isNaN(declaredErrors) &&
284
+ declaredErrors !== caseErrors) {
285
+ throw new Error(`invalid junit xml: testsuites errors=${declaredErrors} disagrees with case errors=${caseErrors}`);
286
+ }
287
+ if (declaredSkipped !== undefined &&
288
+ !Number.isNaN(declaredSkipped) &&
289
+ declaredSkipped !== caseSkipped) {
290
+ throw new Error(`invalid junit xml: testsuites skipped=${declaredSkipped} disagrees with case skipped=${caseSkipped}`);
291
+ }
292
+ }
293
+ tests = caseCount;
294
+ failed = caseFailed;
295
+ errors = caseErrors;
296
+ skipped = caseSkipped;
297
+ }
298
+ else if (!suitesOpen || Number.isNaN(tests) || tests === 0) {
299
+ // No testcase bodies: fall back to first testsuite attributes.
300
+ const suiteOpen = trimmed.match(/<testsuite\b[^>]*>/i)?.[0];
301
+ if (!suiteOpen) {
302
+ throw new Error("invalid junit xml: no testsuite data");
303
+ }
304
+ tests = Number(attr(suiteOpen, "tests") ?? "0");
305
+ failed = Number(attr(suiteOpen, "failures") ?? "0");
306
+ errors = Number(attr(suiteOpen, "errors") ?? "0");
307
+ skipped = Number(attr(suiteOpen, "skipped") ?? "0");
308
+ const time = attr(suiteOpen, "time");
309
+ if (time)
310
+ timeSec = Number(time);
311
+ }
312
+ if ([tests, failed, errors, skipped].some((n) => Number.isNaN(n) || n < 0)) {
313
+ throw new Error("invalid junit xml: non-numeric suite counters");
314
+ }
315
+ const passed = Math.max(0, tests - failed - errors - skipped);
316
+ return {
317
+ tests,
318
+ passed,
319
+ failed,
320
+ errors,
321
+ skipped,
322
+ durationMs: timeSec !== undefined && !Number.isNaN(timeSec)
323
+ ? Math.round(timeSec * 1000)
324
+ : undefined,
325
+ failures: failures.slice(0, MAX_FAILURES),
326
+ };
327
+ }
328
+ export function deriveBackendTestResult(input) {
329
+ assertNoSecrets("commandSummary", input.commandSummary);
330
+ const emptyJunitMeta = {
331
+ relativePath: input.junitRelativePath,
332
+ sha256: "0".repeat(64),
333
+ };
334
+ if (input.junitXml === null || input.junitXml === undefined) {
335
+ if (!input.allowMissingJunit) {
336
+ throw new Error("missing junit report");
337
+ }
338
+ const outcome = input.pytestExitCode === 0 ? "report-error" : "command-error";
339
+ const executionStatus = input.pytestExitCode === 0 ? "report-error" : "command-error";
340
+ const result = {
341
+ schemaVersion: 1,
342
+ executionStatus,
343
+ pytestExitCode: input.pytestExitCode,
344
+ collectionStatus: "unknown",
345
+ tests: 0,
346
+ passed: 0,
347
+ failed: 0,
348
+ error: 0,
349
+ skipped: 0,
350
+ junit: emptyJunitMeta,
351
+ commandSummary: input.commandSummary,
352
+ failures: [],
353
+ outcome,
354
+ };
355
+ return backendTestResultContractSchema.parse(result);
356
+ }
357
+ let parsed;
358
+ try {
359
+ parsed = parseJunitXml(input.junitXml);
360
+ }
361
+ catch (error) {
362
+ throw new Error(`invalid junit xml: ${error instanceof Error ? error.message : String(error)}`);
363
+ }
364
+ const sha256 = createHash("sha256").update(input.junitXml).digest("hex");
365
+ const exit = input.pytestExitCode;
366
+ let executionStatus = "completed";
367
+ let collectionStatus = "ok";
368
+ let outcome = "passed";
369
+ // Collection-heavy signals: pytest exit 2 is common for collection errors;
370
+ // also when error cases exist with zero/low completed tests.
371
+ const looksLikeCollection = exit === 2 ||
372
+ (parsed.errors > 0 && parsed.passed + parsed.failed === 0) ||
373
+ parsed.failures.some((f) => f.kind === "error" &&
374
+ /collect|import|syntax/i.test(`${f.name} ${f.message}`));
375
+ if (looksLikeCollection && (parsed.errors > 0 || exit >= 2)) {
376
+ executionStatus = "collection-error";
377
+ collectionStatus = "error";
378
+ outcome = "collection-error";
379
+ }
380
+ else if (exit >= 2 && parsed.failed === 0 && parsed.errors === 0) {
381
+ executionStatus = "command-error";
382
+ collectionStatus = "unknown";
383
+ outcome = "command-error";
384
+ }
385
+ else if (parsed.failed > 0 || parsed.errors > 0 || exit === 1) {
386
+ executionStatus = "completed";
387
+ collectionStatus = "ok";
388
+ outcome = "completed-with-failures";
389
+ }
390
+ else if (exit === 0 && parsed.failed === 0 && parsed.errors === 0) {
391
+ executionStatus = "completed";
392
+ collectionStatus = "ok";
393
+ outcome = "passed";
394
+ }
395
+ else {
396
+ executionStatus = "command-error";
397
+ collectionStatus = "unknown";
398
+ outcome = "command-error";
399
+ }
400
+ const result = {
401
+ schemaVersion: 1,
402
+ executionStatus,
403
+ pytestExitCode: exit,
404
+ collectionStatus,
405
+ tests: parsed.tests,
406
+ passed: parsed.passed,
407
+ failed: parsed.failed,
408
+ error: parsed.errors,
409
+ skipped: parsed.skipped,
410
+ durationMs: parsed.durationMs,
411
+ junit: {
412
+ relativePath: input.junitRelativePath,
413
+ sha256,
414
+ },
415
+ commandSummary: input.commandSummary,
416
+ failures: parsed.failures.map((f) => ({
417
+ classname: f.classname || "unknown",
418
+ name: f.name || "unknown",
419
+ message: truncate(f.message || "failure"),
420
+ kind: f.kind,
421
+ })),
422
+ outcome,
423
+ };
424
+ return backendTestResultContractSchema.parse(result);
425
+ }
426
+ export function validateBackendTestResultJunitIntegrity(result, junitXml) {
427
+ backendTestResultContractSchema.parse(result);
428
+ const actualSha = createHash("sha256").update(junitXml).digest("hex");
429
+ if (result.junit.sha256 !== actualSha) {
430
+ throw new Error("junit sha256 mismatch against report content");
431
+ }
432
+ }
433
+ /**
434
+ * Deterministic classification constraints for Pi classifier (not the final label).
435
+ * Single-run failures must never suggest FlakyTest; collection/command never ProductBug.
436
+ */
437
+ export function classifyCategoryHints(result) {
438
+ const forbidden = new Set();
439
+ const suggested = new Set();
440
+ // Single observation cannot prove flakiness.
441
+ forbidden.add("FlakyTest");
442
+ if (result.executionStatus === "collection-error" ||
443
+ result.executionStatus === "command-error" ||
444
+ result.executionStatus === "report-error" ||
445
+ result.outcome === "collection-error" ||
446
+ result.outcome === "command-error" ||
447
+ result.outcome === "report-error") {
448
+ forbidden.add("ProductBug");
449
+ suggested.add("EnvFailure");
450
+ suggested.add("Unknown");
451
+ if (result.executionStatus === "collection-error") {
452
+ suggested.add("TestBug");
453
+ }
454
+ return {
455
+ suggestedCategories: [...suggested],
456
+ forbiddenCategories: [...forbidden],
457
+ confidenceCap: 0.6,
458
+ };
459
+ }
460
+ if (result.outcome === "passed") {
461
+ return {
462
+ suggestedCategories: [],
463
+ forbiddenCategories: [
464
+ ...forbidden,
465
+ "ProductBug",
466
+ "TestBug",
467
+ "EnvFailure",
468
+ ],
469
+ confidenceCap: 1,
470
+ };
471
+ }
472
+ // Assertion failures: ProductBug / TestBug / Unknown allowed; not Flaky.
473
+ suggested.add("ProductBug");
474
+ suggested.add("TestBug");
475
+ suggested.add("Unknown");
476
+ return {
477
+ suggestedCategories: [...suggested],
478
+ forbiddenCategories: [...forbidden],
479
+ confidenceCap: 0.75,
480
+ };
481
+ }
482
+ async function readPytestExitCode(runDir, fromNodeId) {
483
+ const exitPath = path.join(runDir, "reports", "backend-test-pytest-exit.txt");
484
+ try {
485
+ const raw = (await readFile(exitPath, "utf8")).trim();
486
+ const code = Number(raw);
487
+ if (raw !== "" && Number.isInteger(code) && code >= 0 && code <= 255)
488
+ return code;
489
+ throw new Error(`invalid pytest exit evidence at ${exitPath}`);
490
+ }
491
+ catch (error) {
492
+ if (error.code !== "ENOENT")
493
+ throw error;
494
+ // Missing dedicated evidence may fall through to an explicit node marker.
495
+ }
496
+ const nodePath = path.join(runDir, `${fromNodeId}.json`);
497
+ try {
498
+ const record = JSON.parse(await readFile(nodePath, "utf8"));
499
+ const blob = `${record.stdout ?? ""}\n${record.stderr ?? ""}`;
500
+ const marker = blob.match(/pytestExitCode\s*=\s*(\d{1,3})/i);
501
+ if (marker) {
502
+ const code = Number(marker[1]);
503
+ if (Number.isInteger(code) && code >= 0 && code <= 255)
504
+ return code;
505
+ }
506
+ }
507
+ catch (error) {
508
+ if (error.code !== "ENOENT") {
509
+ throw new Error("invalid pytest exit evidence in execute node record");
510
+ }
511
+ }
512
+ throw new Error("missing valid pytest exit evidence");
513
+ }
514
+ export async function materializeBackendTestResultFromRunDir(input) {
515
+ if (!/^[a-z0-9][a-z0-9._-]*\.json$/.test(input.artifactName) ||
516
+ !/^[a-z0-9][a-z0-9._-]*$/.test(input.outputDir)) {
517
+ throw new Error("unsafe structured artifact path");
518
+ }
519
+ const junitRelativePath = input.junitRelativePath ?? "reports/backend-test-junit.xml";
520
+ const junitAbs = path.join(input.runDir, ...junitRelativePath.split("/"));
521
+ let junitXml = null;
522
+ try {
523
+ junitXml = await readFile(junitAbs, "utf8");
524
+ }
525
+ catch {
526
+ junitXml = null;
527
+ }
528
+ if (junitXml === null) {
529
+ throw new Error(`missing junit report at ${junitRelativePath} (fail-closed for Result v1)`);
530
+ }
531
+ if (!junitXml.trim()) {
532
+ throw new Error("invalid junit xml: empty report");
533
+ }
534
+ const pytestExitCode = await readPytestExitCode(input.runDir, input.fromNodeId);
535
+ const commandSummary = "PYTHONDONTWRITEBYTECODE=1 python -m pytest testcase/ -v -p no:cacheprovider --junitxml=reports/backend-test-junit.xml";
536
+ let result;
537
+ try {
538
+ result = deriveBackendTestResult({
539
+ pytestExitCode,
540
+ junitXml,
541
+ junitRelativePath,
542
+ commandSummary,
543
+ });
544
+ }
545
+ catch (error) {
546
+ throw new Error(`invalid junit/report: ${error instanceof Error ? error.message : String(error)}`);
547
+ }
548
+ // Integrity: stored sha must match bytes we read.
549
+ validateBackendTestResultJunitIntegrity(result, junitXml);
550
+ const relativePath = path.posix.join(input.outputDir, input.artifactName);
551
+ const artifactPath = await writeDagRunJsonArtifact(input.runDir, relativePath, result);
552
+ const serialized = `${JSON.stringify(result, null, 2)}\n`;
553
+ return {
554
+ path: artifactPath,
555
+ sha256: createHash("sha256").update(serialized).digest("hex"),
556
+ schemaId: BACKEND_TEST_RESULT_SCHEMA_ID,
557
+ };
558
+ }
559
+ /** Shell snippet for backend-test-outcome-gate-shell (result.outcome authoritative). */
560
+ export function buildBackendTestOutcomeGateShellSnippet(options) {
561
+ const resultRelativePath = options?.resultRelativePath ?? "contracts/backend-test-result.json";
562
+ return [
563
+ 'test -n "${HARNESS_DAG_RUN_DIR:-}" || { echo "missing HARNESS_DAG_RUN_DIR for backend-test outcome gate" >&2; exit 2; }',
564
+ `RESULT="\${HARNESS_DAG_RUN_DIR}/${resultRelativePath}"`,
565
+ 'test -f "${RESULT}" || { echo "missing backend-test result: ${RESULT}" >&2; exit 2; }',
566
+ `node -e 'const fs=require("fs");const r=JSON.parse(fs.readFileSync(process.argv[1],"utf8"));const outcome=String(r.outcome||"");const ok=outcome==="passed"&&Number(r.failed||0)===0&&Number(r.error||0)===0;console.log("backend-test outcome="+outcome+" passed="+r.passed+" failed="+r.failed+" error="+r.error+" executionStatus="+r.executionStatus);if(!ok){process.exit(1);}' "\${RESULT}"`,
567
+ ].join("; ");
568
+ }