@tea-agent/loop-agent 0.13.0-beta.0 → 0.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (282) hide show
  1. package/AGENTS.md +157 -155
  2. package/CHANGELOG.md +301 -322
  3. package/README.md +335 -345
  4. package/bin/agent-worker.js +22 -22
  5. package/bin/loop-agent.js +21 -21
  6. package/dist/commands/cursor-prompt.js +6 -6
  7. package/dist/commands/init.js +597 -528
  8. package/dist/commands/loop-benchmark.js +11 -11
  9. package/dist/commands/pi-reuse-benchmark.js +16 -16
  10. package/dist/executors/shell-executor.js +200 -21
  11. package/dist/infrastructure/evaluation/candidate-store.js +5 -1
  12. package/dist/sidecars/cursor-prompt/executor.js +1 -1
  13. package/dist/task/runtime.js +27 -27
  14. package/dist/worker/observe/static/api.js +46 -46
  15. package/dist/worker/observe/static/app.js +150 -150
  16. package/dist/worker/observe/static/constants.js +148 -148
  17. package/dist/worker/observe/static/copy.js +67 -67
  18. package/dist/worker/observe/static/dag-helpers.js +172 -172
  19. package/dist/worker/observe/static/dag-layout.d.ts +31 -31
  20. package/dist/worker/observe/static/dag-layout.js +83 -83
  21. package/dist/worker/observe/static/dag-model.js +72 -72
  22. package/dist/worker/observe/static/dom.js +212 -53
  23. package/dist/worker/observe/static/format-pool.js +67 -67
  24. package/dist/worker/observe/static/format.js +292 -292
  25. package/dist/worker/observe/static/index.html +308 -308
  26. package/dist/worker/observe/static/kpi.js +94 -94
  27. package/dist/worker/observe/static/relations.js +133 -133
  28. package/dist/worker/observe/static/router.js +93 -93
  29. package/dist/worker/observe/static/run-processing.js +148 -148
  30. package/dist/worker/observe/static/shell-chrome.js +68 -68
  31. package/dist/worker/observe/static/state.js +267 -253
  32. package/dist/worker/observe/static/styles.css +1902 -1902
  33. package/dist/worker/observe/static/views/batch.js +227 -227
  34. package/dist/worker/observe/static/views/dag-graph.js +172 -172
  35. package/dist/worker/observe/static/views/dag-inspector.js +627 -607
  36. package/dist/worker/observe/static/views/dag.js +371 -362
  37. package/dist/worker/observe/static/views/dashboard.js +509 -252
  38. package/dist/worker/observe/static/views/failures.js +143 -143
  39. package/dist/worker/observe/static/views/feature.js +492 -492
  40. package/dist/worker/observe/static/views/pool.js +350 -350
  41. package/dist/worker/observe/static/views/run.js +453 -453
  42. package/dist/worker/observe/static/views/session-timeline.js +219 -205
  43. package/dist/worker/observe/static/views/shell.js +7 -7
  44. package/dist/worker/observe/static/views/task.js +314 -314
  45. package/dist/worker/observe/static/views/timeline.js +163 -163
  46. package/dist/workflows/dag/backend-test-case-manifest.js +503 -0
  47. package/dist/workflows/dag/backend-test-execution-contract.js +353 -0
  48. package/dist/workflows/dag/backend-test-result-contract.js +568 -0
  49. package/dist/workflows/dag/canvas-observer.js +275 -275
  50. package/dist/workflows/dag/decision-envelope.js +57 -2
  51. package/dist/workflows/dag/frontend-implementation-contract.js +240 -0
  52. package/dist/workflows/dag/frontend-project-capability.js +309 -0
  53. package/dist/workflows/dag/frontend-repair.js +341 -0
  54. package/dist/workflows/dag/frontend-risk.js +161 -0
  55. package/dist/workflows/dag/frontend-verification-trace.js +190 -0
  56. package/dist/workflows/dag/init-hybrid.js +1020 -125
  57. package/dist/workflows/dag/repair-artifact.js +43 -3
  58. package/dist/workflows/dag/skill-instructions.js +4 -2
  59. package/dist/workflows/dag/types.js +29 -8
  60. package/docs/README.md +105 -104
  61. package/docs/agent-dag-recovery-playbook.md +195 -195
  62. package/docs/agent-dag-runner.md +67 -67
  63. package/docs/architecture/README.md +26 -26
  64. package/docs/architecture/dag-execution.md +140 -140
  65. package/docs/architecture/evolution.md +54 -54
  66. package/docs/architecture/facts-and-state.md +71 -71
  67. package/docs/architecture/runtime-boundaries.md +191 -191
  68. package/docs/architecture/system-overview.md +93 -93
  69. package/docs/architecture/worker-and-feature.md +85 -85
  70. package/docs/cursor-prompt-sidecar.md +36 -36
  71. package/docs/decisions/README.md +18 -18
  72. package/docs/design/README.md +167 -167
  73. package/docs/development-principles.md +73 -73
  74. package/docs/exec-plans/README.md +6 -6
  75. package/docs/exec-plans/active/README.md +1 -4
  76. package/docs/exec-plans/completed/README.md +106 -84
  77. package/docs/feature-workflow.md +414 -389
  78. package/docs/harness-methodology-debugging.md +153 -153
  79. package/docs/harness-methodology-tdd.md +130 -130
  80. package/docs/harness-methodology-verification.md +27 -27
  81. package/docs/init-surface.manifest.json +307 -289
  82. package/docs/loop-agent-harness.md +142 -142
  83. package/docs/production-readiness.md +96 -96
  84. package/docs/progress/README.md +76 -60
  85. package/docs/reports/README.md +150 -108
  86. package/docs/skills/README.md +7 -7
  87. package/docs/skills/vetted-skill-registry.md +29 -29
  88. package/docs/templates/adr.md +60 -60
  89. package/docs/templates/agent-dag-authority-surface-audit.prompt.md +94 -94
  90. package/docs/templates/agent-dag-decision-envelope.schema.json +213 -213
  91. package/docs/templates/agent-dag-decision-gate-dogfood-report.md +117 -117
  92. package/docs/templates/agent-dag-decision-gate.prompt.md +246 -246
  93. package/docs/templates/agent-dag-process-supervisor.prompt.md +98 -98
  94. package/docs/templates/agent-dag-report.schema.json +473 -473
  95. package/docs/templates/agent-dag-review-verdict.prompt.md +68 -68
  96. package/docs/templates/agent-dag.base.json +190 -190
  97. package/docs/templates/agent-dag.final-verification.json +185 -185
  98. package/docs/templates/agent-dag.schema.json +411 -411
  99. package/docs/templates/agent-dag.supervised-implementation.json +620 -501
  100. package/docs/templates/backend-test-analysis.schema.json +44 -44
  101. package/docs/templates/backend-test-case-manifest.schema.json +190 -0
  102. package/docs/templates/backend-test-dag.classify.prompt.md +75 -0
  103. package/docs/templates/backend-test-dag.generate-pytest.prompt.md +204 -202
  104. package/docs/templates/backend-test-dag.json +559 -311
  105. package/docs/templates/backend-test-dag.retrospect.prompt.md +139 -125
  106. package/docs/templates/backend-test-dag.review-cases.prompt.md +83 -81
  107. package/docs/templates/backend-test-execution.schema.json +133 -0
  108. package/docs/templates/backend-test-result.schema.json +99 -0
  109. package/docs/templates/branch-merge-report.md +93 -0
  110. package/docs/templates/exec-plan.md +64 -64
  111. package/docs/templates/feature-spec.md +53 -53
  112. package/docs/templates/frontend-design-contract.md +42 -42
  113. package/docs/templates/frontend-eval/fixtures/failures/01-type-build-error.md +17 -0
  114. package/docs/templates/frontend-eval/fixtures/failures/02-unit-component-test-fail.md +16 -0
  115. package/docs/templates/frontend-eval/fixtures/failures/03-fixture-schema-drift.md +16 -0
  116. package/docs/templates/frontend-eval/fixtures/failures/04-missing-loading-empty-error-state.md +16 -0
  117. package/docs/templates/frontend-eval/fixtures/failures/05-forbidden-write-writeset-expansion.md +16 -0
  118. package/docs/templates/frontend-eval/fixtures/failures/06-unapproved-dependency-add.md +16 -0
  119. package/docs/templates/frontend-eval/fixtures/failures/07-mock-production-on.md +21 -0
  120. package/docs/templates/frontend-eval/fixtures/functional/01-simple-component-style.md +29 -0
  121. package/docs/templates/frontend-eval/fixtures/functional/02-form-validation.md +28 -0
  122. package/docs/templates/frontend-eval/fixtures/functional/03-list-detail-page.md +28 -0
  123. package/docs/templates/frontend-eval/fixtures/functional/04-api-mock.md +29 -0
  124. package/docs/templates/frontend-eval/fixtures/functional/05-permission-auth-gated-ui.md +27 -0
  125. package/docs/templates/frontend-eval/fixtures/functional/06-ssr-server-client-boundary.md +28 -0
  126. package/docs/templates/frontend-eval/fixtures/functional/07-shared-public-component-api.md +28 -0
  127. package/docs/templates/frontend-eval/fixtures/functional/08-pure-local-no-remote.md +27 -0
  128. package/docs/templates/frontend-eval/metrics.md +138 -0
  129. package/docs/templates/frontend-eval/smoke-targets.md +53 -0
  130. package/docs/templates/frontend-implementation-contract.schema.json +27 -0
  131. package/docs/templates/frontend-task-constraints.md +35 -35
  132. package/docs/templates/frontend-task-requirement.md +70 -70
  133. package/docs/templates/frontend-test-dag.generate-cases.prompt.md +5 -5
  134. package/docs/templates/frontend-test-dag.json +23 -23
  135. package/docs/templates/frontend-test-dag.retrieve-context.prompt.md +3 -3
  136. package/docs/templates/frontend-test-dag.retrospect.prompt.md +3 -3
  137. package/docs/templates/frontend-test-dag.review-cases.prompt.md +3 -3
  138. package/docs/templates/frontend-test-dag.review-execution.prompt.md +3 -3
  139. package/docs/templates/harness.schema.json +221 -221
  140. package/docs/templates/hybrid-dag.json +188 -188
  141. package/docs/templates/init-evolution-review.md +35 -35
  142. package/docs/templates/interactive-ui-round2-experiment.md +66 -66
  143. package/docs/templates/knowledge-graph-bootstrap-dag.json +118 -118
  144. package/docs/templates/knowledge-sync-dag.json +178 -178
  145. package/docs/templates/knowledge-sync-draft.schema.json +71 -71
  146. package/docs/templates/product-line/AGENTS.md +8 -8
  147. package/docs/templates/product-line/README.md +9 -9
  148. package/docs/templates/product-line/acceptance.yaml +14 -14
  149. package/docs/templates/product-line/closeout.yaml +9 -9
  150. package/docs/templates/product-line/design.md +13 -13
  151. package/docs/templates/product-line/links.md +10 -10
  152. package/docs/templates/product-line/requirement.md +17 -17
  153. package/docs/templates/product-line/task-graph.yaml +15 -15
  154. package/docs/templates/product-line/task.yaml +64 -64
  155. package/docs/templates/product-line/test-plan.md +7 -7
  156. package/docs/templates/production-readiness-checklist.md +57 -57
  157. package/docs/templates/progress-log.md +17 -17
  158. package/docs/templates/project-start-checklist.md +9 -9
  159. package/docs/templates/qa-report.md +48 -48
  160. package/docs/templates/sprint-contract.md +29 -29
  161. package/docs/templates/worker-dogfood-evidence.md +80 -80
  162. package/docs/templates/worker-dogfood-setup.md +68 -68
  163. package/docs/verification-matrix.md +70 -70
  164. package/examples/decision-gate-agent-dag.json +177 -177
  165. package/examples/example-dag.json +46 -46
  166. package/examples/hybrid-loop-agent-dag.json +189 -189
  167. package/harness.json +66 -66
  168. package/package.json +52 -88
  169. package/scripts/check-product-line-docs.sh +29 -29
  170. package/scripts/check-task-pool-root.sh +32 -32
  171. package/scripts/kb-bootstrap-init-skeleton.sh +240 -240
  172. package/scripts/kb-graph-incremental-prepare.mjs +386 -386
  173. package/scripts/kb-graph-incremental-prepare.sh +5 -5
  174. package/scripts/kb-graph-materialize.mjs +105 -105
  175. package/scripts/kb-graph-materialize.sh +4 -4
  176. package/scripts/kb-graph-promote.mjs +164 -164
  177. package/scripts/kb-graph-promote.sh +4 -4
  178. package/scripts/kb-query.mjs +554 -554
  179. package/scripts/kb-query.sh +5 -5
  180. package/skills/agent-worker/SKILL.md +39 -39
  181. package/skills/agent-worker/references/agent-worker-operator.md +60 -60
  182. package/skills/ai-engineering-context/SKILL.md +48 -48
  183. package/skills/analyze-product-dependencies/SKILL.md +67 -67
  184. package/skills/analyze-product-dependencies/agents/openai.yaml +4 -4
  185. package/skills/analyze-product-dependencies/references/api-documentation-schema.md +30 -30
  186. package/skills/analyze-product-dependencies/references/dependency-analysis-schema.md +28 -28
  187. package/skills/analyze-product-dependencies/references/example.md +76 -76
  188. package/skills/analyze-product-dependencies/references/forward-test-cases.md +35 -35
  189. package/skills/analyze-product-dependencies/references/input-contract.md +11 -11
  190. package/skills/analyze-product-dependencies/references/scouting-rules.md +61 -61
  191. package/skills/analyze-product-dependencies/scripts/test-validators.mjs +267 -267
  192. package/skills/analyze-product-dependencies/scripts/validate-api-documentation.mjs +101 -101
  193. package/skills/analyze-product-dependencies/scripts/validate-dependency-analysis.mjs +142 -142
  194. package/skills/analyze-product-dependencies/scripts/validate-product-requirement-input.mjs +76 -76
  195. package/skills/analyze-product-dependencies/scripts/validation-helpers.mjs +146 -146
  196. package/skills/analyze-product-requirements/SKILL.md +90 -90
  197. package/skills/analyze-product-requirements/agents/openai.yaml +4 -4
  198. package/skills/analyze-product-requirements/references/acceptance-criteria.md +91 -91
  199. package/skills/analyze-product-requirements/references/clarification-and-knowledge.md +56 -56
  200. package/skills/analyze-product-requirements/references/example.md +86 -86
  201. package/skills/analyze-product-requirements/references/forward-test-cases.md +66 -66
  202. package/skills/analyze-product-requirements/references/product-analysis-schema.md +32 -32
  203. package/skills/analyze-product-requirements/references/product-requirement-schema.md +33 -33
  204. package/skills/analyze-product-requirements/references/requirement-clarification-schema.md +35 -35
  205. package/skills/analyze-product-requirements/scripts/test-validators.mjs +193 -193
  206. package/skills/analyze-product-requirements/scripts/validate-product-analysis.mjs +69 -69
  207. package/skills/analyze-product-requirements/scripts/validate-product-requirement.mjs +97 -97
  208. package/skills/analyze-product-requirements/scripts/validate-requirement-clarification.mjs +98 -98
  209. package/skills/analyze-product-requirements/scripts/validation-helpers.mjs +156 -156
  210. package/skills/browser-tools/SKILL.md +196 -0
  211. package/skills/browser-tools/browser-content.js +103 -0
  212. package/skills/browser-tools/browser-cookies.js +35 -0
  213. package/skills/browser-tools/browser-eval.js +53 -0
  214. package/skills/browser-tools/browser-hn-scraper.js +108 -0
  215. package/skills/browser-tools/browser-nav.js +44 -0
  216. package/skills/browser-tools/browser-pick.js +162 -0
  217. package/skills/browser-tools/browser-screenshot.js +34 -0
  218. package/skills/browser-tools/browser-start.js +86 -0
  219. package/skills/browser-tools/package-lock.json +2556 -0
  220. package/skills/browser-tools/package.json +19 -0
  221. package/skills/code-review-core/SKILL.md +20 -20
  222. package/skills/codebase-scout/SKILL.md +19 -19
  223. package/skills/frontend-design-review/SKILL.md +66 -66
  224. package/skills/frontend-design-review/references/review-checklist.md +58 -58
  225. package/skills/frontend-implementation/SKILL.md +49 -47
  226. package/skills/frontend-implementation/references/code-standards.md +32 -32
  227. package/skills/frontend-implementation/references/design-spec.md +46 -46
  228. package/skills/frontend-implementation/references/node-contracts.md +27 -76
  229. package/skills/frontend-review/SKILL.md +59 -59
  230. package/skills/frontend-review/references/review-findings.md +47 -47
  231. package/skills/frontend-verification/SKILL.md +53 -53
  232. package/skills/frontend-verification/references/verification-checklist.md +68 -68
  233. package/skills/grill-me/SKILL.md +10 -10
  234. package/skills/grill-with-docs/SKILL.md +88 -88
  235. package/skills/grill-with-docs/adr-format.md +47 -47
  236. package/skills/grill-with-docs/context-format.md +60 -60
  237. package/skills/init-capability-evolution/SKILL.md +70 -70
  238. package/skills/loop-agent/SKILL.md +151 -151
  239. package/skills/loop-agent/references/README.md +67 -67
  240. package/skills/loop-agent/references/command-reference.md +527 -505
  241. package/skills/loop-agent/references/docs-converge.md +126 -126
  242. package/skills/loop-agent/references/harness-policy.md +263 -263
  243. package/skills/loop-agent/references/hybrid-dag.md +243 -238
  244. package/skills/loop-agent/references/learned/README.md +21 -21
  245. package/skills/loop-agent/references/long-running-loop.md +57 -57
  246. package/skills/loop-agent/references/model-routing.md +36 -36
  247. package/skills/loop-agent/references/multi-worktree.md +54 -54
  248. package/skills/loop-agent/references/one-shot-runs.md +85 -85
  249. package/skills/loop-agent/references/orchestrator-and-interventions.md +169 -169
  250. package/skills/loop-agent/references/pi-prompt.md +23 -23
  251. package/skills/loop-agent/references/pi-subagent-assisted-mode.md +84 -84
  252. package/skills/loop-agent/references/post-implementation-and-patterns.md +44 -44
  253. package/skills/loop-agent/references/task-workflow.md +89 -89
  254. package/skills/loop-agent/references/verification-and-failure-handling.md +141 -139
  255. package/skills/playwright-cli/SKILL.md +420 -420
  256. package/skills/playwright-cli/references/element-attributes.md +23 -23
  257. package/skills/playwright-cli/references/playwright-tests.md +39 -39
  258. package/skills/playwright-cli/references/request-mocking.md +87 -87
  259. package/skills/playwright-cli/references/running-code.md +241 -241
  260. package/skills/playwright-cli/references/session-management.md +225 -225
  261. package/skills/playwright-cli/references/storage-state.md +275 -275
  262. package/skills/playwright-cli/references/test-generation.md +433 -433
  263. package/skills/playwright-cli/references/tracing.md +139 -139
  264. package/skills/playwright-cli/references/video-recording.md +143 -143
  265. package/skills/playwright-cli-case-generator/SKILL.md +74 -74
  266. package/skills/requesting-code-review/SKILL.md +101 -101
  267. package/skills/requesting-code-review/code-reviewer.md +168 -168
  268. package/skills/systematic-debugging/CREATION-LOG.md +119 -119
  269. package/skills/systematic-debugging/SKILL.md +296 -296
  270. package/skills/systematic-debugging/condition-based-waiting-example.ts +158 -158
  271. package/skills/systematic-debugging/condition-based-waiting.md +115 -115
  272. package/skills/systematic-debugging/defense-in-depth.md +122 -122
  273. package/skills/systematic-debugging/find-polluter.sh +63 -63
  274. package/skills/systematic-debugging/root-cause-tracing.md +169 -169
  275. package/skills/systematic-debugging/test-academic.md +14 -14
  276. package/skills/systematic-debugging/test-pressure-1.md +58 -58
  277. package/skills/systematic-debugging/test-pressure-2.md +68 -68
  278. package/skills/systematic-debugging/test-pressure-3.md +69 -69
  279. package/skills/test-driven-development/SKILL.md +20 -20
  280. package/skills/using-git-worktrees/SKILL.md +215 -215
  281. package/skills/verification-before-completion/SKILL.md +154 -154
  282. package/skills/webapp-testing/SKILL.md +19 -19
@@ -1,125 +1,139 @@
1
- # Backend Test DAG Retrospect Prompt Template
2
-
3
- ## Purpose
4
-
5
- Use this prompt for a **test retrospective** node: `executor: "pi"`, `role: "closeout"`, `toolProfile: "write"`, `writePolicy: "exclusive"`. The closeout agent reads upstream review reports and pytest execution results, then generates a retrospective report with an objective maturity rating.
6
-
7
- Do **not** create a new executor type. This is a standard `executor: pi` writer node.
8
-
9
- ## Recommended DAG Node Shape
10
-
11
- ```json
12
- {
13
- "id": "test-retrospect-pi",
14
- "depends_on": ["execute-backend-pytest-shell"],
15
- "complexity": "MED",
16
- "executor": "pi",
17
- "role": "closeout",
18
- "toolProfile": "write",
19
- "writePolicy": "exclusive",
20
- "writeSet": ["docs/test-reports/**"],
21
- "allowedPaths": ["docs/test-reports/**"],
22
- "forbiddenPaths": [".harness/**", "artifacts/**"],
23
- "outputContract": "Markdown retrospective report under docs/test-reports/ with coverage summary, review findings, pytest results, and maturity rating (A/B/C/D).",
24
- "subtask_prompt_markdown": "./backend-test-dag.retrospect.prompt.md"
25
- }
26
- ```
27
-
28
- ## Prompt Body
29
-
30
- You are the Backend Test DAG **test retrospective** agent.
31
-
32
- Your job is to read upstream outputs (review report + pytest results) and generate a retrospective report with a maturity rating. Write the report under `docs/test-reports/` only. Stay within `writeSet`. Do not write root `artifacts/**`.
33
-
34
- ### Output Steps (do in order)
35
-
36
- 1. First, output the maturity rating on the first line: `Rating: A/B/C/D`
37
- 2. Then write the full report under `docs/test-reports/`
38
-
39
- ### Inputs
40
-
41
- 1. **Review report** — `review-backend-cases-pi` output (VERDICT, findings, coverage assessment).
42
- 2. **Pytest output** — `execute-backend-pytest-shell` stdout/stderr and exit code.
43
- 3. **Machine-readable report** — `$HARNESS_DAG_RUN_DIR/reports/backend-test-junit.xml` (runner-owned evidence; use the path reported by `execute-backend-pytest-shell`).
44
-
45
- Do NOT re-read source documents. Use upstream outputs only.
46
-
47
- ### Maturity Rating Criteria
48
-
49
- | Rating | Coverage | Pass Rate | Review Findings |
50
- |--------|----------|-----------|-----------------|
51
- | **A** | 100% acceptance criteria covered | 100% pytest pass | No Critical or Important findings |
52
- | **B** | ≥80% acceptance criteria covered | ≥90% pytest pass | Only Informational findings |
53
- | **C** | ≥60% acceptance criteria covered | ≥70% pytest pass | No Critical findings (Important allowed) |
54
- | **D** | Below C thresholds | Below C thresholds | Or any Critical finding unresolved |
55
-
56
- #### Rating Rules
57
-
58
- - **Coverage** = (acceptance criteria with ≥1 covering test case) / (total acceptance criteria) × 100%
59
- - **Pass rate** = (passed pytest functions) / (total non-skipped pytest functions) × 100%
60
- - If `review-backend-cases-pi` returned `VERDICT: request-revision` and revision was not completed, cap at **D**.
61
- - If pytest exit code is non-zero and >30% tests failed, cap at **D** regardless of coverage.
62
- - Skipped tests (`@pytest.mark.skip`) count as "not covered" for pass rate but not as failures.
63
-
64
- ### Report Structure
65
-
66
- Write the report as a Markdown file named `backend-test-retrospect-<date>.md` under `docs/test-reports/`.
67
-
68
- ```markdown
69
- # Backend Test Retrospective Report
70
-
71
- **Date:** <YYYY-MM-DD>
72
- **Task:** <task-id>
73
- **Maturity Rating:** <A|B|C|D>
74
-
75
- ## 1. Test Coverage Summary
76
-
77
- | Metric | Value |
78
- |--------|-------|
79
- | Total acceptance criteria | N |
80
- | Covered by test cases | N (X%) |
81
- | Total functional test cases | N |
82
- | Positive path cases | N |
83
- | Negative path cases | N |
84
- | Boundary cases | N |
85
-
86
- ## 2. Automation Results
87
-
88
- | Metric | Value |
89
- |--------|-------|
90
- | Total pytest functions | N |
91
- | Passed | N |
92
- | Failed | N |
93
- | Skipped | N |
94
- | Pass rate | X% |
95
- | Pytest exit code | N |
96
-
97
- ### Failed Test Analysis
98
-
99
- | Test Case ID | Function | Failure Reason | Root Cause |
100
- |--------------|----------|----------------|------------|
101
- | ... | ... | ... | ... |
102
-
103
- ## 3. Review Findings
104
-
105
- | Severity | Finding | Status |
106
- |----------|---------|--------|
107
- | Critical | ... | Resolved / Unresolved |
108
- | Important | ... | Resolved / Unresolved |
109
- | Informational | ... | Resolved / Unresolved |
110
-
111
- ## 4. Maturity Rating Rationale
112
-
113
- Explain which threshold was met or missed, and why the specific rating was assigned.
114
-
115
- ## 5. Recommendations
116
-
117
- - Actionable items for improving the rating in the next iteration.
118
- - Specific gaps to close (uncovered criteria, flaky tests, missing negative paths).
119
- ```
120
-
121
- ### Output Shape (after rating line)
122
-
123
- After the mandatory maturity rating line, provide a brief summary paragraph before writing the full report file.
124
-
125
- Do not include chain-of-thought. Do not write root `artifacts/**`.
1
+ # Backend Test DAG Retrospect Prompt Template
2
+
3
+ ## Purpose
4
+
5
+ Use this prompt for a **test retrospective** node: `executor: "pi"`, `role: "closeout"`, `toolProfile: "write"`, `writePolicy: "exclusive"`. The closeout agent reads **Backend Test Result v1** (and classification), then generates a retrospective report with an objective maturity rating.
6
+
7
+ Do **not** create a new executor type. This is a standard `executor: pi` writer node.
8
+
9
+ ## Recommended DAG Node Shape
10
+
11
+ ```json
12
+ {
13
+ "id": "test-retrospect-pi",
14
+ "depends_on": ["classify-backend-test-result-pi"],
15
+ "complexity": "MED",
16
+ "executor": "pi",
17
+ "role": "closeout",
18
+ "toolProfile": "write",
19
+ "writePolicy": "exclusive",
20
+ "writeSet": ["docs/test-reports/**"],
21
+ "allowedPaths": ["docs/test-reports/**"],
22
+ "forbiddenPaths": [".harness/**", "artifacts/**"],
23
+ "outputContract": "Markdown retrospective report under docs/test-reports/ with coverage summary, review findings, Result v1 stats, classification, and maturity rating (A/B/C/D).",
24
+ "subtask_prompt_markdown": "./backend-test-dag.retrospect.prompt.md"
25
+ }
26
+ ```
27
+
28
+ ## Prompt Body
29
+
30
+ You are the Backend Test DAG **test retrospective** agent.
31
+
32
+ Your job is to read upstream Result v1 + classification (+ review report) and generate a retrospective report with a maturity rating. Write the report under `docs/test-reports/` only. Stay within `writeSet`. Do not write root `artifacts/**`.
33
+
34
+ This node runs on **both pass and assertion-fail** paths (after parse + classify). Final task success is decided later by `backend-test-outcome-gate-shell` using Result v1 shell facts only — **never** rewrite a failed result as passed in this report.
35
+
36
+ ### Output Steps (do in order)
37
+
38
+ 1. First, output the maturity rating on the first line: `Rating: A/B/C/D`
39
+ 2. Then write the full report under `docs/test-reports/`
40
+
41
+ ### Inputs
42
+
43
+ 1. **Result v1 (authoritative stats)** — `$HARNESS_DAG_RUN_DIR/contracts/backend-test-result.json`
44
+ Use `passed` / `failed` / `error` / `skipped` / `outcome` / `failures[]` / `pytestExitCode` only from this artifact.
45
+ 2. **Case Manifest v1 (authoritative AC coverage)** `$HARNESS_DAG_RUN_DIR/contracts/backend-test-case-manifest.json`
46
+ Use `coverageSummary.acCoverageRatio`, `coveredAcCount`, `explicitAcCount`, case counts only from this artifact.
47
+ 3. **Classification** `classify-backend-test-result-pi` JSON (`category`, `confidence`, `evidence`). Interpretive only; does not override outcome.
48
+ 4. **Review report** — `review-backend-cases-pi` output (VERDICT, findings, coverage assessment).
49
+ 5. Optional secondary: execute stdout markers / JUnit path (do not re-parse logs for counts when Result v1 exists).
50
+
51
+ Do NOT re-read source documents. Use upstream outputs only.
52
+
53
+ ### Stats authority
54
+
55
+ - Pass rate = `passed / (passed + failed + error)` when denominator > 0 (skipped excluded from denominator unless Result documents otherwise) — **Result v1 only**.
56
+ - AC coverage = `coverageSummary.acCoverageRatio` from Case Manifest v1 only (do **not** recompute or invent percentages).
57
+ - Failed case table rows must match `failures[]` from Result v1.
58
+ - If Result v1 `outcome` is not `passed`, the retrospective **must not** claim overall success.
59
+
60
+ ### Maturity Rating Criteria
61
+
62
+ | Rating | Coverage (manifest) | Pass Rate (Result v1) | Review Findings |
63
+ |--------|---------------------|------------------------|-----------------|
64
+ | **A** | `acCoverageRatio` = 1 | 100% pytest pass | No Critical or Important findings |
65
+ | **B** | `acCoverageRatio` ≥ 0.8 | ≥90% pytest pass | Only Informational findings |
66
+ | **C** | `acCoverageRatio` 0.6 | ≥70% pytest pass | No Critical findings (Important allowed) |
67
+ | **D** | Below C thresholds | Below C thresholds | Or any Critical finding unresolved |
68
+
69
+ #### Rating Rules
70
+
71
+ - **Coverage** from Case Manifest `coverageSummary` only (deterministic gate product).
72
+ - **Pass rate** from Result v1 only (not guessed from logs).
73
+ - If `review-backend-cases-pi` returned `VERDICT: request-revision` and revision was not completed, cap at **D**.
74
+ - If Result v1 shows >30% failed+error among executed tests, cap at **D** regardless of coverage.
75
+ - Skipped tests count as "not covered" for pass rate but not as failures.
76
+ - Collection/command/report errors → cap at **D** and record classification (not ProductBug by default).
77
+
78
+ ### Report Structure
79
+
80
+ Write the report as a Markdown file named `backend-test-retrospect-<date>.md` under `docs/test-reports/`.
81
+
82
+ ```markdown
83
+ # Backend Test Retrospective Report
84
+
85
+ **Date:** <YYYY-MM-DD>
86
+ **Task:** <task-id>
87
+ **Maturity Rating:** <A|B|C|D>
88
+ **Result outcome:** <from Result v1>
89
+ **Classification:** <from classify JSON>
90
+
91
+ ## 1. Test Coverage Summary
92
+
93
+ | Metric | Value |
94
+ |--------|-------|
95
+ | Total acceptance criteria | N |
96
+ | Covered by test cases | N (X%) |
97
+ | Total functional test cases | N |
98
+
99
+ ## 2. Automation Results (from Result v1)
100
+
101
+ | Metric | Value |
102
+ |--------|-------|
103
+ | Total tests | N |
104
+ | Passed | N |
105
+ | Failed | N |
106
+ | Error | N |
107
+ | Skipped | N |
108
+ | Pass rate | X% |
109
+ | Pytest exit code | N |
110
+ | Outcome | … |
111
+ | Execution status | … |
112
+
113
+ ### Failed Test Analysis
114
+
115
+ | Test Case / Function | Message (truncated) | Classification |
116
+ |----------------------|-------------------|----------------|
117
+ | ... | ... | ... |
118
+
119
+ ## 3. Review Findings
120
+
121
+ | Severity | Finding | Status |
122
+ |----------|---------|--------|
123
+ | | | |
124
+
125
+ ## 4. Maturity Rating Rationale
126
+
127
+ Explain which threshold was met or missed.
128
+
129
+ ## 5. Recommendations
130
+
131
+ - Actionable items for the next iteration.
132
+ - Do not propose changing production code solely to greenwash tests.
133
+ ```
134
+
135
+ ### Output Shape (after rating line)
136
+
137
+ After the mandatory maturity rating line, provide a brief summary paragraph before writing the full report file.
138
+
139
+ Do not include chain-of-thought. Do not write root `artifacts/**`.
@@ -1,81 +1,83 @@
1
- # Backend Test DAG Review Cases Prompt Template
2
-
3
- ## Purpose
4
-
5
- Use this prompt for a read-only **backend test case review** node: `executor: "pi"`, `role: "reviewer"`, `writePolicy: "read-only"`. The reviewer audits generated backend functional test cases for completeness, format compliance, and traceability to source requirements. Downstream `generate-backend-pytest-pi` depends on a `VERDICT: pass` to proceed.
6
-
7
- Do **not** create `executor: reviewer`. Reviewer is a **role** on `executor: pi`.
8
-
9
- ## Recommended DAG Node Shape
10
-
11
- ```json
12
- {
13
- "id": "review-backend-cases-pi",
14
- "depends_on": ["generate-backend-functional-cases-pi", "backend-test-analysis-contract-shell"],
15
- "complexity": "HIGH",
16
- "executor": "pi",
17
- "role": "reviewer",
18
- "writePolicy": "read-only",
19
- "allowedPaths": ["**"],
20
- "forbiddenPaths": [".harness/**", "artifacts/**"],
21
- "outputContract": "Plain Markdown whose first non-empty line is VERDICT: pass or VERDICT: request-revision; followed by Findings and Coverage Assessment. No file writes.",
22
- "subtask_prompt_markdown": "./backend-test-dag.review-cases.prompt.md"
23
- }
24
- ```
25
-
26
- ## Prompt Body
27
-
28
- You are the Backend Test DAG **test case reviewer** (read-only).
29
-
30
- Your job is to audit the generated backend functional test cases for completeness, format compliance, requirement coverage, and traceability. You are **not** an implementer or test generator. Do not edit repository files, including root `artifacts/**`.
31
-
32
- ### Mandatory First Line
33
-
34
- The **first non-empty line** of your response must be exactly one of:
35
-
36
- - `VERDICT: pass`
37
- - `VERDICT: request-revision`
38
-
39
- No preamble, heading, or blank lines before the verdict line.
40
-
41
- ### Inputs to Review
42
-
43
- 1. **Acceptance criteria / analysis** — from the validated Backend Test Analysis v1 artifact materialized by `backend-test-analysis-contract-shell` (`contracts/backend-test-analysis.json` under the current DAG run). Do not treat free-form Markdown from `analyze-inputs-pi` as the contract.
44
- 2. **Generated test cases** — files under `testcase/md/`.
45
-
46
- Do NOT re-read source documents. Use the validated analysis artifact and generated cases only.
47
-
48
- ### Review Checklist
49
-
50
- | Area | Check | Severity if Missing |
51
- |------|-------|---------------------|
52
- | **ID format** | Every test case ID matches `BE-<MODULE>-<NNN>` (e.g. `BE-ORDER-001`) | Critical |
53
- | **Positive path coverage** | Happy-path scenarios for each acceptance criterion | Critical |
54
- | **Negative path coverage** | Error/exception scenarios (invalid input, not found, state violations) | Important |
55
- | **Boundary conditions** | Edge cases (empty input, max length, edge values) | Important |
56
- | **State transitions** | Illegal state changes covered | Important |
57
- | **Requirement traceability** | Each acceptance criterion (AC-xxx) maps to at least one test case ID | Critical |
58
- | **Case structure** | Each case has: ID, Title, Precondition, Steps, Expected Result | Important |
59
- | **No duplicate IDs** | All test case IDs are unique across files | Critical |
60
-
61
- ### Conditional Coverage (check ONLY if mentioned in upstream analysis)
62
-
63
- - **Authentication coverage**: check ONLY if the validated analysis artifact mentions auth mechanism (JWT, OAuth2, API Key, etc.)
64
- - **Timeout coverage**: check ONLY if the validated analysis artifact mentions timeout handling or degradation strategy
65
- - If not mentioned in the validated analysis artifact, do NOT flag as missing
66
-
67
- ### Verdict Rules
68
-
69
- | Condition | Verdict |
70
- |-----------|---------|
71
- | All Critical checks pass, Important checks have no more than 2 findings | `VERDICT: pass` |
72
- | Any Critical check fails | `VERDICT: request-revision` |
73
- | More than 2 Important findings | `VERDICT: request-revision` |
74
- | Only Informational findings | `VERDICT: pass` (with findings listed) |
75
-
76
- ### Output Shape (after verdict line)
77
-
78
- 1. **Coverage Assessment** table mapping each AC to covering test case IDs (or "uncovered").
79
- 2. **Findings** — bullet list tagged `Critical`, `Important`, or `Informational`.
80
- 3. **Statistics** — total case count, positive/negative/boundary breakdown, module distribution.
81
- 4. **Required revisions** (only when `request-revision`) numbered items for the upstream generator to fix.
1
+ # Backend Test DAG Review Cases Prompt Template
2
+
3
+ ## Purpose
4
+
5
+ Use this prompt for a read-only **backend test case review** node: `executor: "pi"`, `role: "reviewer"`, `writePolicy: "read-only"`. The reviewer audits generated backend functional test cases for completeness, format compliance, and traceability to source requirements. Downstream `generate-backend-pytest-pi` depends on a `VERDICT: pass` to proceed.
6
+
7
+ Do **not** create `executor: reviewer`. Reviewer is a **role** on `executor: pi`.
8
+
9
+ ## Recommended DAG Node Shape
10
+
11
+ ```json
12
+ {
13
+ "id": "review-backend-cases-pi",
14
+ "depends_on": ["backend-test-case-manifest-shell", "backend-test-analysis-contract-shell"],
15
+ "complexity": "HIGH",
16
+ "executor": "pi",
17
+ "role": "reviewer",
18
+ "writePolicy": "read-only",
19
+ "allowedPaths": ["**"],
20
+ "forbiddenPaths": [".harness/**", "artifacts/**"],
21
+ "outputContract": "Plain Markdown whose first non-empty line is VERDICT: pass or VERDICT: request-revision; followed by Findings and Coverage Assessment. No file writes.",
22
+ "subtask_prompt_markdown": "./backend-test-dag.review-cases.prompt.md"
23
+ }
24
+ ```
25
+
26
+ ## Prompt Body
27
+
28
+ You are the Backend Test DAG **test case reviewer** (read-only).
29
+
30
+ Your job is to audit the generated backend functional test cases for completeness, format compliance, requirement coverage, and traceability. You are **not** an implementer or test generator. Do not edit repository files, including root `artifacts/**`.
31
+
32
+ ### Mandatory First Line
33
+
34
+ The **first non-empty line** of your response must be exactly one of:
35
+
36
+ - `VERDICT: pass`
37
+ - `VERDICT: request-revision`
38
+
39
+ No preamble, heading, or blank lines before the verdict line.
40
+
41
+ ### Inputs to Review
42
+
43
+ 1. **Acceptance criteria / analysis** — from the validated Backend Test Analysis v1 artifact materialized by `backend-test-analysis-contract-shell` (`contracts/backend-test-analysis.json` under the current DAG run). Do not treat free-form Markdown from `analyze-inputs-pi` as the contract.
44
+ 2. **Case Manifest v1** — `contracts/backend-test-case-manifest.json` (schemaId `backend-test-case-manifest-v1`). Prefer `coverageSummary` and caseId↔acIds from this artifact; do not invent coverage percentages.
45
+ 3. **Generated test cases** — files under `testcase/md/`.
46
+
47
+ Do NOT re-read source documents. Use the validated analysis artifact, case manifest, and generated cases only.
48
+
49
+ ### Review Checklist
50
+
51
+ | Area | Check | Severity if Missing |
52
+ |------|-------|---------------------|
53
+ | **ID format** | Every test case ID matches `BE-<MODULE>-<NNN>` (e.g. `BE-ORDER-001`) | Critical |
54
+ | **Positive path coverage** | Happy-path scenarios for each acceptance criterion | Critical |
55
+ | **Negative path coverage** | Error/exception scenarios (invalid input, not found, state violations) | Important |
56
+ | **Boundary conditions** | Edge cases (empty input, max length, edge values) | Important |
57
+ | **State transitions** | Illegal state changes covered | Important |
58
+ | **Requirement traceability** | Each acceptance criterion (AC-xxx) maps to at least one test case ID (manifest coverageSummary or evidenceGaps) | Critical |
59
+ | **Manifest consistency** | Markdown cases align with Case Manifest v1 caseId/acIds | Critical |
60
+ | **Case structure** | Each case has: ID, Title, Precondition, Steps, Expected Result | Important |
61
+ | **No duplicate IDs** | All test case IDs are unique across files | Critical |
62
+
63
+ ### Conditional Coverage (check ONLY if mentioned in upstream analysis)
64
+
65
+ - **Authentication coverage**: check ONLY if the validated analysis artifact mentions auth mechanism (JWT, OAuth2, API Key, etc.)
66
+ - **Timeout coverage**: check ONLY if the validated analysis artifact mentions timeout handling or degradation strategy
67
+ - If not mentioned in the validated analysis artifact, do NOT flag as missing
68
+
69
+ ### Verdict Rules
70
+
71
+ | Condition | Verdict |
72
+ |-----------|---------|
73
+ | All Critical checks pass, Important checks have no more than 2 findings | `VERDICT: pass` |
74
+ | Any Critical check fails | `VERDICT: request-revision` |
75
+ | More than 2 Important findings | `VERDICT: request-revision` |
76
+ | Only Informational findings | `VERDICT: pass` (with findings listed) |
77
+
78
+ ### Output Shape (after verdict line)
79
+
80
+ 1. **Coverage Assessment** — table mapping each AC to covering test case IDs (or "uncovered").
81
+ 2. **Findings** — bullet list tagged `Critical`, `Important`, or `Informational`.
82
+ 3. **Statistics** — total case count, positive/negative/boundary breakdown, module distribution.
83
+ 4. **Required revisions** (only when `request-revision`) — numbered items for the upstream generator to fix.
@@ -0,0 +1,133 @@
1
+ {
2
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
3
+ "$id": "https://tea-agent.dev/schemas/backend-test-execution-v1.json",
4
+ "title": "Backend Test Execution v1",
5
+ "type": "object",
6
+ "additionalProperties": false,
7
+ "required": [
8
+ "schemaVersion",
9
+ "framework",
10
+ "runner",
11
+ "testRoot",
12
+ "workingDirectory",
13
+ "report",
14
+ "targetMode",
15
+ "existingFixtures",
16
+ "authenticationMode",
17
+ "requiredEnvNames",
18
+ "dataIsolation",
19
+ "evidenceGaps",
20
+ "evidenceRefs"
21
+ ],
22
+ "properties": {
23
+ "schemaVersion": { "const": 1 },
24
+ "framework": { "const": "pytest" },
25
+ "runner": {
26
+ "type": "object",
27
+ "additionalProperties": false,
28
+ "properties": {
29
+ "commandParts": {
30
+ "type": "array",
31
+ "items": { "type": "string", "minLength": 1 }
32
+ },
33
+ "frozenCommandHints": {
34
+ "type": "array",
35
+ "items": { "type": "string", "minLength": 1 }
36
+ }
37
+ }
38
+ },
39
+ "testRoot": {
40
+ "type": "string",
41
+ "minLength": 1,
42
+ "description": "Repo-relative posix path; no absolute form or .. segments"
43
+ },
44
+ "workingDirectory": {
45
+ "type": "string",
46
+ "minLength": 1,
47
+ "default": "."
48
+ },
49
+ "report": {
50
+ "type": "object",
51
+ "additionalProperties": false,
52
+ "required": ["format", "relativeHint"],
53
+ "properties": {
54
+ "format": { "const": "junit" },
55
+ "relativeHint": { "type": "string", "minLength": 1 }
56
+ }
57
+ },
58
+ "targetMode": {
59
+ "enum": ["in-process", "external-running-service", "managed-command"]
60
+ },
61
+ "baseUrlEnvName": {
62
+ "type": "string",
63
+ "pattern": "^[A-Z_][A-Z0-9_]*$"
64
+ },
65
+ "readiness": {
66
+ "type": "array",
67
+ "items": {
68
+ "type": "object",
69
+ "additionalProperties": false,
70
+ "required": ["path", "description"],
71
+ "properties": {
72
+ "path": { "type": "string", "minLength": 1 },
73
+ "description": { "type": "string", "minLength": 1 }
74
+ }
75
+ }
76
+ },
77
+ "existingFixtures": {
78
+ "type": "array",
79
+ "items": {
80
+ "type": "object",
81
+ "additionalProperties": false,
82
+ "required": ["name", "sourcePath", "kind"],
83
+ "properties": {
84
+ "name": { "type": "string", "minLength": 1 },
85
+ "sourcePath": { "type": "string", "minLength": 1 },
86
+ "kind": { "type": "string", "minLength": 1 }
87
+ }
88
+ }
89
+ },
90
+ "authenticationMode": { "type": "string", "minLength": 1 },
91
+ "requiredEnvNames": {
92
+ "type": "array",
93
+ "items": {
94
+ "type": "string",
95
+ "pattern": "^[A-Z_][A-Z0-9_]*$"
96
+ }
97
+ },
98
+ "dataIsolation": {
99
+ "type": "object",
100
+ "additionalProperties": false,
101
+ "required": ["mode"],
102
+ "properties": {
103
+ "mode": { "type": "string", "minLength": 1 },
104
+ "evidence": { "type": "string", "minLength": 1 }
105
+ }
106
+ },
107
+ "managedCommand": {
108
+ "type": "object",
109
+ "additionalProperties": false,
110
+ "properties": {
111
+ "start": { "type": "string", "minLength": 1 },
112
+ "stop": { "type": "string", "minLength": 1 },
113
+ "sourceRef": { "type": "string", "minLength": 1 }
114
+ }
115
+ },
116
+ "evidenceGaps": {
117
+ "type": "array",
118
+ "items": {
119
+ "type": "object",
120
+ "additionalProperties": false,
121
+ "required": ["description"],
122
+ "properties": {
123
+ "description": { "type": "string", "minLength": 1 },
124
+ "sourceRef": { "type": "string", "minLength": 1 }
125
+ }
126
+ }
127
+ },
128
+ "evidenceRefs": {
129
+ "type": "array",
130
+ "items": { "type": "string", "minLength": 1 }
131
+ }
132
+ }
133
+ }