@tea-agent/loop-agent 0.12.0 → 0.13.0-beta.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +155 -153
- package/CHANGELOG.md +338 -265
- package/README.md +345 -298
- package/bin/agent-worker.js +22 -22
- package/bin/loop-agent.js +21 -21
- package/dist/application/dag/generate-task-dag.js +28 -28
- package/dist/application/evaluation/candidate-hash.js +75 -0
- package/dist/application/evaluation/candidate.js +52 -0
- package/dist/application/evaluation/replay.js +289 -0
- package/dist/application/evaluation/types.js +130 -0
- package/dist/cli/command-definitions.js +27 -7
- package/dist/cli/program.js +8 -4
- package/dist/commands/cursor-prompt.js +6 -6
- package/dist/commands/eval.js +235 -0
- package/dist/commands/init.js +544 -506
- package/dist/commands/knowledge.js +129 -31
- package/dist/commands/loop-benchmark.js +11 -11
- package/dist/commands/pi-reuse-benchmark.js +16 -16
- package/dist/executors/pi-sdk-executor.js +38 -24
- package/dist/executors/shell-executor.js +34 -2
- package/dist/executors/shell-presets.js +20 -0
- package/dist/executors/shell-verification.js +7 -0
- package/dist/governance/manifest-types.js +4 -0
- package/dist/infrastructure/evaluation/candidate-store.js +435 -0
- package/dist/infrastructure/evaluation/store.js +40 -0
- package/dist/sidecars/cursor-prompt/executor.js +1 -1
- package/dist/task/config-types.js +28 -1
- package/dist/task/runtime.js +27 -27
- package/dist/worker/cli.js +96 -1
- package/dist/worker/delivery/package.js +3 -3
- package/dist/worker/feature/decision-loader.js +37 -6
- package/dist/worker/feature/next-action.js +10 -2
- package/dist/worker/feature/ready-plan-projection.js +81 -0
- package/dist/worker/feature/reducer.js +2 -1
- package/dist/worker/feature/review.js +19 -2
- package/dist/worker/feature/run.js +27 -2
- package/dist/worker/follow-up/approve.js +5 -2
- package/dist/worker/follow-up/factory.js +1 -1
- package/dist/worker/observability/read-model.js +246 -41
- package/dist/worker/observe/routes.js +173 -15
- package/dist/worker/observe/spec-evidence.js +281 -0
- package/dist/worker/observe/static/api.js +46 -27
- package/dist/worker/observe/static/app.js +150 -150
- package/dist/worker/observe/static/constants.js +148 -148
- package/dist/worker/observe/static/copy.js +67 -67
- package/dist/worker/observe/static/dag-helpers.js +172 -172
- package/dist/worker/observe/static/dag-layout.d.ts +31 -31
- package/dist/worker/observe/static/dag-layout.js +83 -83
- package/dist/worker/observe/static/dag-model.js +72 -72
- package/dist/worker/observe/static/dom.js +61 -61
- package/dist/worker/observe/static/format-pool.js +67 -67
- package/dist/worker/observe/static/format.js +292 -292
- package/dist/worker/observe/static/index.html +308 -308
- package/dist/worker/observe/static/kpi.js +94 -94
- package/dist/worker/observe/static/relations.js +133 -128
- package/dist/worker/observe/static/router.js +93 -85
- package/dist/worker/observe/static/run-processing.js +148 -148
- package/dist/worker/observe/static/shell-chrome.js +68 -68
- package/dist/worker/observe/static/state.js +253 -253
- package/dist/worker/observe/static/styles.css +1902 -1890
- package/dist/worker/observe/static/views/batch.js +227 -226
- package/dist/worker/observe/static/views/dag-graph.js +172 -172
- package/dist/worker/observe/static/views/dag-inspector.js +607 -477
- package/dist/worker/observe/static/views/dag.js +362 -362
- package/dist/worker/observe/static/views/dashboard.js +445 -442
- package/dist/worker/observe/static/views/failures.js +143 -143
- package/dist/worker/observe/static/views/feature.js +492 -453
- package/dist/worker/observe/static/views/pool.js +350 -347
- package/dist/worker/observe/static/views/run.js +453 -453
- package/dist/worker/observe/static/views/session-timeline.js +205 -205
- package/dist/worker/observe/static/views/shell.js +7 -7
- package/dist/worker/observe/static/views/task.js +314 -260
- package/dist/worker/observe/static/views/timeline.js +163 -163
- package/dist/worker/pool/doctor.js +165 -0
- package/dist/worker/pool/migrate-state.js +303 -0
- package/dist/worker/pool/run-store.js +205 -17
- package/dist/worker/pool/types.js +17 -1
- package/dist/worker/pool/validation.js +100 -15
- package/dist/worker/report/morning-report.js +12 -2
- package/dist/worker/runner/run-ready.js +41 -26
- package/dist/worker/task-graph/ready-planner.js +136 -0
- package/dist/workflows/dag/backend-test-analysis-contract.js +120 -0
- package/dist/workflows/dag/canvas-observer.js +275 -275
- package/dist/workflows/dag/convergence/controller.js +16 -8
- package/dist/workflows/dag/dynamic-runtime/map.js +90 -2
- package/dist/workflows/dag/failure-routing.js +12 -1
- package/dist/workflows/dag/init-hybrid.js +2404 -360
- package/dist/workflows/dag/node-execution.js +9 -0
- package/dist/workflows/dag/prompt.js +9 -0
- package/dist/workflows/dag/report.js +35 -1
- package/dist/workflows/dag/runner.js +28 -2
- package/dist/workflows/dag/task-demand-routing.js +383 -0
- package/dist/workflows/dag/types.js +51 -13
- package/dist/workflows/dag/upstream-artifacts.js +1 -0
- package/dist/workflows/dag/validate.js +59 -1
- package/docs/README.md +106 -104
- package/docs/agent-dag-recovery-playbook.md +195 -184
- package/docs/agent-dag-runner.md +67 -67
- package/docs/architecture/README.md +26 -26
- package/docs/architecture/dag-execution.md +140 -140
- package/docs/architecture/evolution.md +54 -53
- package/docs/architecture/facts-and-state.md +71 -58
- package/docs/architecture/runtime-boundaries.md +191 -191
- package/docs/architecture/system-overview.md +93 -93
- package/docs/architecture/worker-and-feature.md +85 -81
- package/docs/cursor-prompt-sidecar.md +36 -36
- package/docs/decisions/README.md +18 -15
- package/docs/design/README.md +167 -77
- package/docs/development-principles.md +73 -73
- package/docs/exec-plans/README.md +6 -6
- package/docs/exec-plans/active/README.md +15 -9
- package/docs/exec-plans/completed/README.md +85 -73
- package/docs/feature-workflow.md +389 -261
- package/docs/harness-methodology-debugging.md +153 -153
- package/docs/harness-methodology-tdd.md +130 -130
- package/docs/harness-methodology-verification.md +27 -27
- package/docs/init-surface.manifest.json +289 -280
- package/docs/loop-agent-harness.md +142 -130
- package/docs/production-readiness.md +96 -96
- package/docs/progress/README.md +64 -54
- package/docs/reports/README.md +117 -94
- package/docs/skills/README.md +7 -7
- package/docs/skills/vetted-skill-registry.md +29 -27
- package/docs/templates/adr.md +60 -60
- package/docs/templates/agent-dag-authority-surface-audit.prompt.md +94 -94
- package/docs/templates/agent-dag-decision-envelope.schema.json +213 -213
- package/docs/templates/agent-dag-decision-gate-dogfood-report.md +117 -117
- package/docs/templates/agent-dag-decision-gate.prompt.md +246 -246
- package/docs/templates/agent-dag-process-supervisor.prompt.md +98 -98
- package/docs/templates/agent-dag-report.schema.json +473 -473
- package/docs/templates/agent-dag-review-verdict.prompt.md +68 -68
- package/docs/templates/agent-dag.base.json +190 -190
- package/docs/templates/agent-dag.final-verification.json +185 -185
- package/docs/templates/agent-dag.schema.json +411 -383
- package/docs/templates/agent-dag.supervised-implementation.json +501 -501
- package/docs/templates/backend-test-analysis.schema.json +44 -0
- package/docs/templates/backend-test-dag.generate-pytest.prompt.md +202 -139
- package/docs/templates/backend-test-dag.json +311 -276
- package/docs/templates/backend-test-dag.retrospect.prompt.md +125 -125
- package/docs/templates/backend-test-dag.review-cases.prompt.md +81 -81
- package/docs/templates/exec-plan.md +64 -64
- package/docs/templates/feature-spec.md +53 -53
- package/docs/templates/frontend-design-contract.md +42 -33
- package/docs/templates/frontend-task-constraints.md +35 -25
- package/docs/templates/frontend-task-requirement.md +70 -61
- package/docs/templates/frontend-test-dag.generate-cases.prompt.md +5 -0
- package/docs/templates/frontend-test-dag.json +23 -0
- package/docs/templates/frontend-test-dag.retrieve-context.prompt.md +3 -0
- package/docs/templates/frontend-test-dag.retrospect.prompt.md +3 -0
- package/docs/templates/frontend-test-dag.review-cases.prompt.md +3 -0
- package/docs/templates/frontend-test-dag.review-execution.prompt.md +3 -0
- package/docs/templates/harness.schema.json +221 -221
- package/docs/templates/hybrid-dag.json +188 -188
- package/docs/templates/init-evolution-review.md +35 -35
- package/docs/templates/interactive-ui-round2-experiment.md +66 -66
- package/docs/templates/knowledge-graph-bootstrap-dag.json +118 -0
- package/docs/templates/knowledge-sync-dag.json +178 -0
- package/docs/templates/knowledge-sync-draft.schema.json +71 -0
- package/docs/templates/product-line/AGENTS.md +8 -8
- package/docs/templates/product-line/README.md +9 -9
- package/docs/templates/product-line/acceptance.yaml +14 -14
- package/docs/templates/product-line/closeout.yaml +9 -9
- package/docs/templates/product-line/design.md +13 -13
- package/docs/templates/product-line/links.md +10 -10
- package/docs/templates/product-line/requirement.md +17 -17
- package/docs/templates/product-line/task-graph.yaml +15 -15
- package/docs/templates/product-line/task.yaml +64 -64
- package/docs/templates/product-line/test-plan.md +7 -7
- package/docs/templates/production-readiness-checklist.md +57 -57
- package/docs/templates/progress-log.md +17 -17
- package/docs/templates/project-start-checklist.md +9 -9
- package/docs/templates/qa-report.md +48 -48
- package/docs/templates/sprint-contract.md +29 -29
- package/docs/templates/worker-dogfood-evidence.md +80 -80
- package/docs/templates/worker-dogfood-setup.md +68 -68
- package/docs/verification-matrix.md +70 -66
- package/examples/decision-gate-agent-dag.json +177 -177
- package/examples/example-dag.json +46 -46
- package/examples/hybrid-loop-agent-dag.json +189 -189
- package/harness.json +66 -66
- package/package.json +88 -46
- package/scripts/check-product-line-docs.sh +29 -29
- package/scripts/check-task-pool-root.sh +32 -32
- package/scripts/kb-bootstrap-init-skeleton.sh +240 -0
- package/scripts/kb-graph-incremental-prepare.mjs +386 -0
- package/scripts/kb-graph-incremental-prepare.sh +5 -0
- package/scripts/kb-graph-materialize.mjs +105 -0
- package/scripts/kb-graph-materialize.sh +4 -0
- package/scripts/kb-graph-promote.mjs +164 -0
- package/scripts/kb-graph-promote.sh +4 -0
- package/scripts/kb-query.mjs +554 -0
- package/scripts/kb-query.sh +5 -0
- package/skills/agent-worker/SKILL.md +39 -37
- package/skills/agent-worker/references/agent-worker-operator.md +60 -43
- package/skills/ai-engineering-context/SKILL.md +48 -48
- package/skills/analyze-product-dependencies/SKILL.md +67 -0
- package/skills/analyze-product-dependencies/agents/openai.yaml +4 -0
- package/skills/analyze-product-dependencies/references/api-documentation-schema.md +30 -0
- package/skills/analyze-product-dependencies/references/dependency-analysis-schema.md +28 -0
- package/skills/analyze-product-dependencies/references/example.md +76 -0
- package/skills/analyze-product-dependencies/references/forward-test-cases.md +35 -0
- package/skills/analyze-product-dependencies/references/input-contract.md +11 -0
- package/skills/analyze-product-dependencies/references/scouting-rules.md +61 -0
- package/skills/analyze-product-dependencies/scripts/test-validators.mjs +267 -0
- package/skills/analyze-product-dependencies/scripts/validate-api-documentation.mjs +101 -0
- package/skills/analyze-product-dependencies/scripts/validate-dependency-analysis.mjs +142 -0
- package/skills/analyze-product-dependencies/scripts/validate-product-requirement-input.mjs +76 -0
- package/skills/analyze-product-dependencies/scripts/validation-helpers.mjs +146 -0
- package/skills/analyze-product-requirements/SKILL.md +90 -0
- package/skills/analyze-product-requirements/agents/openai.yaml +4 -0
- package/skills/analyze-product-requirements/references/acceptance-criteria.md +91 -0
- package/skills/analyze-product-requirements/references/clarification-and-knowledge.md +56 -0
- package/skills/analyze-product-requirements/references/example.md +86 -0
- package/skills/analyze-product-requirements/references/forward-test-cases.md +66 -0
- package/skills/analyze-product-requirements/references/product-analysis-schema.md +32 -0
- package/skills/analyze-product-requirements/references/product-requirement-schema.md +33 -0
- package/skills/analyze-product-requirements/references/requirement-clarification-schema.md +35 -0
- package/skills/analyze-product-requirements/scripts/test-validators.mjs +193 -0
- package/skills/analyze-product-requirements/scripts/validate-product-analysis.mjs +69 -0
- package/skills/analyze-product-requirements/scripts/validate-product-requirement.mjs +97 -0
- package/skills/analyze-product-requirements/scripts/validate-requirement-clarification.mjs +98 -0
- package/skills/analyze-product-requirements/scripts/validation-helpers.mjs +156 -0
- package/skills/code-review-core/SKILL.md +20 -20
- package/skills/codebase-scout/SKILL.md +19 -19
- package/skills/frontend-design-review/SKILL.md +66 -59
- package/skills/frontend-design-review/references/review-checklist.md +58 -37
- package/skills/frontend-implementation/SKILL.md +47 -51
- package/skills/frontend-implementation/references/code-standards.md +32 -34
- package/skills/frontend-implementation/references/design-spec.md +46 -46
- package/skills/frontend-implementation/references/node-contracts.md +76 -32
- package/skills/frontend-review/SKILL.md +59 -53
- package/skills/frontend-review/references/review-findings.md +47 -42
- package/skills/frontend-verification/SKILL.md +53 -40
- package/skills/frontend-verification/references/verification-checklist.md +68 -56
- package/skills/grill-me/SKILL.md +10 -10
- package/skills/grill-with-docs/SKILL.md +88 -88
- package/skills/grill-with-docs/adr-format.md +47 -47
- package/skills/grill-with-docs/context-format.md +60 -60
- package/skills/init-capability-evolution/SKILL.md +70 -70
- package/skills/loop-agent/SKILL.md +151 -151
- package/skills/loop-agent/references/README.md +67 -67
- package/skills/loop-agent/references/command-reference.md +505 -452
- package/skills/loop-agent/references/docs-converge.md +126 -126
- package/skills/loop-agent/references/harness-policy.md +263 -263
- package/skills/loop-agent/references/hybrid-dag.md +238 -233
- package/skills/loop-agent/references/learned/README.md +21 -21
- package/skills/loop-agent/references/long-running-loop.md +57 -57
- package/skills/loop-agent/references/model-routing.md +36 -36
- package/skills/loop-agent/references/multi-worktree.md +54 -54
- package/skills/loop-agent/references/one-shot-runs.md +85 -85
- package/skills/loop-agent/references/orchestrator-and-interventions.md +169 -169
- package/skills/loop-agent/references/pi-prompt.md +23 -23
- package/skills/loop-agent/references/pi-subagent-assisted-mode.md +84 -84
- package/skills/loop-agent/references/post-implementation-and-patterns.md +44 -44
- package/skills/loop-agent/references/task-workflow.md +89 -89
- package/skills/loop-agent/references/verification-and-failure-handling.md +139 -139
- package/skills/playwright-cli/SKILL.md +420 -0
- package/skills/playwright-cli/references/element-attributes.md +23 -0
- package/skills/playwright-cli/references/playwright-tests.md +39 -0
- package/skills/playwright-cli/references/request-mocking.md +87 -0
- package/skills/playwright-cli/references/running-code.md +241 -0
- package/skills/playwright-cli/references/session-management.md +225 -0
- package/skills/playwright-cli/references/storage-state.md +275 -0
- package/skills/playwright-cli/references/test-generation.md +433 -0
- package/skills/playwright-cli/references/tracing.md +139 -0
- package/skills/playwright-cli/references/video-recording.md +143 -0
- package/skills/playwright-cli-case-generator/SKILL.md +74 -0
- package/skills/requesting-code-review/SKILL.md +101 -101
- package/skills/requesting-code-review/code-reviewer.md +168 -168
- package/skills/systematic-debugging/CREATION-LOG.md +119 -119
- package/skills/systematic-debugging/SKILL.md +296 -296
- package/skills/systematic-debugging/condition-based-waiting-example.ts +158 -158
- package/skills/systematic-debugging/condition-based-waiting.md +115 -115
- package/skills/systematic-debugging/defense-in-depth.md +122 -122
- package/skills/systematic-debugging/find-polluter.sh +63 -63
- package/skills/systematic-debugging/root-cause-tracing.md +169 -169
- package/skills/systematic-debugging/test-academic.md +14 -14
- package/skills/systematic-debugging/test-pressure-1.md +58 -58
- package/skills/systematic-debugging/test-pressure-2.md +68 -68
- package/skills/systematic-debugging/test-pressure-3.md +69 -69
- package/skills/test-driven-development/SKILL.md +20 -20
- package/skills/using-git-worktrees/SKILL.md +215 -215
- package/skills/verification-before-completion/SKILL.md +154 -154
- package/skills/webapp-testing/SKILL.md +19 -19
|
@@ -1,276 +1,311 @@
|
|
|
1
|
-
{
|
|
2
|
-
"$schema": "./agent-dag.schema.json",
|
|
3
|
-
"version": 3,
|
|
4
|
-
"title": "Backend test DAG template",
|
|
5
|
-
"runtimeContract": {
|
|
6
|
-
"schemaVersion": 1,
|
|
7
|
-
"agentRuntime": "pi-only",
|
|
8
|
-
"repairWriterProtocol": "explicit-node-v1"
|
|
9
|
-
},
|
|
10
|
-
"objective": "End-to-end backend functional testing pipeline: analyze requirements → generate functional test cases → review cases → generate pytest automation → execute pytest → retrospective with maturity rating. Covers the full chain from requirement analysis to test maturity assessment.",
|
|
11
|
-
"successCriteria": [
|
|
12
|
-
"analyze-inputs-pi returns
|
|
13
|
-
"
|
|
14
|
-
"
|
|
15
|
-
"review-backend-cases-
|
|
16
|
-
"
|
|
17
|
-
"
|
|
18
|
-
"
|
|
19
|
-
"
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
"
|
|
24
|
-
"
|
|
25
|
-
"
|
|
26
|
-
"
|
|
27
|
-
"
|
|
28
|
-
"
|
|
29
|
-
"
|
|
30
|
-
"
|
|
31
|
-
"
|
|
32
|
-
"
|
|
33
|
-
"
|
|
34
|
-
"
|
|
35
|
-
"
|
|
36
|
-
"
|
|
37
|
-
"
|
|
38
|
-
"
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
"
|
|
43
|
-
"
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
"
|
|
54
|
-
|
|
55
|
-
"
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
"code-review
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
"
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
"
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
"
|
|
74
|
-
"
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
"
|
|
81
|
-
"
|
|
82
|
-
"
|
|
83
|
-
"
|
|
84
|
-
"
|
|
85
|
-
"
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
"
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
"
|
|
94
|
-
|
|
95
|
-
"
|
|
96
|
-
"
|
|
97
|
-
"
|
|
98
|
-
"
|
|
99
|
-
|
|
100
|
-
"
|
|
101
|
-
"
|
|
102
|
-
"
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
"
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
"
|
|
114
|
-
"
|
|
115
|
-
"
|
|
116
|
-
"writePolicy": "
|
|
117
|
-
"
|
|
118
|
-
"
|
|
119
|
-
],
|
|
120
|
-
"
|
|
121
|
-
"
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
"
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
"
|
|
147
|
-
"
|
|
148
|
-
|
|
149
|
-
"
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
"
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
"
|
|
160
|
-
},
|
|
161
|
-
{
|
|
162
|
-
"id": "review-backend-cases-
|
|
163
|
-
"depends_on": [
|
|
164
|
-
"
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
"
|
|
168
|
-
"
|
|
169
|
-
"
|
|
170
|
-
"
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
"
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
"
|
|
179
|
-
"
|
|
180
|
-
"
|
|
181
|
-
"
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
"
|
|
187
|
-
"
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
},
|
|
193
|
-
{
|
|
194
|
-
"id": "
|
|
195
|
-
"depends_on": [
|
|
196
|
-
"review-backend-cases-
|
|
197
|
-
],
|
|
198
|
-
"complexity": "
|
|
199
|
-
"executor": "
|
|
200
|
-
"role": "
|
|
201
|
-
"
|
|
202
|
-
"
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
"
|
|
208
|
-
],
|
|
209
|
-
"
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
"
|
|
233
|
-
"
|
|
234
|
-
"
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
"
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
|
|
251
|
-
|
|
252
|
-
|
|
253
|
-
"
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
"
|
|
258
|
-
"
|
|
259
|
-
"
|
|
260
|
-
"
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
"
|
|
269
|
-
|
|
270
|
-
"
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
|
|
1
|
+
{
|
|
2
|
+
"$schema": "./agent-dag.schema.json",
|
|
3
|
+
"version": 3,
|
|
4
|
+
"title": "Backend test DAG template",
|
|
5
|
+
"runtimeContract": {
|
|
6
|
+
"schemaVersion": 1,
|
|
7
|
+
"agentRuntime": "pi-only",
|
|
8
|
+
"repairWriterProtocol": "explicit-node-v1"
|
|
9
|
+
},
|
|
10
|
+
"objective": "End-to-end backend functional testing pipeline: analyze requirements as Backend Test Analysis v1 JSON → validate analysis contract → generate functional test cases → review cases → generate pytest automation → execute pytest → retrospective with maturity rating. Covers the full chain from requirement analysis to test maturity assessment.",
|
|
11
|
+
"successCriteria": [
|
|
12
|
+
"analyze-inputs-pi returns pure Backend Test Analysis v1 JSON (schema docs/templates/backend-test-analysis.schema.json) with no Markdown prose",
|
|
13
|
+
"backend-test-analysis-contract-shell validates schemaId backend-test-analysis-v1 and materializes run-owned contracts/backend-test-analysis.json",
|
|
14
|
+
"generate-backend-functional-cases-pi consumes the validated analysis artifact and produces structured test cases with BE-<MODULE>-<NNN> IDs under testcase/md/",
|
|
15
|
+
"review-backend-cases-pi returns VERDICT: pass or request-revision with coverage assessment",
|
|
16
|
+
"review-backend-cases-gate-shell blocks pytest generation unless the review verdict is VERDICT: pass",
|
|
17
|
+
"generate-backend-pytest-pi converts reviewed cases into pytest code with 1:1 traceability under testcase/",
|
|
18
|
+
"execute-backend-pytest-shell runs pytest with a read-only worktree and writes JUnit XML under the current HARNESS_DAG_RUN_DIR/reports/** only",
|
|
19
|
+
"test-retrospect-pi generates retrospective report with A/B/C/D maturity rating under docs/test-reports/",
|
|
20
|
+
"Full traceability from acceptance criteria → functional test case ID → pytest function name"
|
|
21
|
+
],
|
|
22
|
+
"globalConstraints": [
|
|
23
|
+
"Replace every REPLACE/WITH/... placeholder with concrete repo paths before execution; do not leave template placeholders in production DAG JSON.",
|
|
24
|
+
"Do not commit runtime traces under .harness/dag-runs/.",
|
|
25
|
+
"Use only existing executors: pi, shell. Pi writer nodes must set toolProfile=write.",
|
|
26
|
+
"Read-only nodes must not write repository files, including root artifacts/**.",
|
|
27
|
+
"Exclusive writer nodes must stay within declared writeSet.",
|
|
28
|
+
"Every task must explicitly declare executor; defaults.executor is schema-only and not a runtime fallback.",
|
|
29
|
+
"Functional test case IDs must use BE-<MODULE>-<NNN> format.",
|
|
30
|
+
"pytest execution must keep the target worktree read-only and write machine-readable results only under the current HARNESS_DAG_RUN_DIR/reports/** (e.g. JUnit XML).",
|
|
31
|
+
"Backend-test-dag nodes must maintain traceability from requirements to functional cases to pytest automation.",
|
|
32
|
+
"pytest automation scripts must use test_ filename prefix for pytest discovery.",
|
|
33
|
+
"generate-backend-pytest-pi may create only new files under testcase/**/test_*.py, testcase/**/helpers/**, and testcase/**/factories/**; modifying conftest.py, pytest.ini, pyproject.toml, or production code is forbidden.",
|
|
34
|
+
"review-backend-cases-gate-shell must block pytest generation unless the review verdict is exactly VERDICT: pass.",
|
|
35
|
+
"If a target test filename already exists under testcase/, add a numeric suffix (_01, _02, ...); never overwrite or append to existing files.",
|
|
36
|
+
"execute-backend-pytest-shell must not modify test assertions or production code to make tests pass; test failures indicate potential implementation issues and must be reported honestly.",
|
|
37
|
+
"Prompt templates must not ask main session to write artifacts; node output is the artifact and runner archives it under .harness/dag-runs/.",
|
|
38
|
+
"Actual file operation paths must be macOS/Windows compatible; use / only for stable repo refs, JSON/Markdown evidence refs, and glob conventions.",
|
|
39
|
+
"Same-rank exclusive writeSet entries must be disjoint."
|
|
40
|
+
],
|
|
41
|
+
"defaults": {
|
|
42
|
+
"executor": "pi",
|
|
43
|
+
"contextProfile": "slim",
|
|
44
|
+
"skills": [
|
|
45
|
+
"ai-engineering-context"
|
|
46
|
+
],
|
|
47
|
+
"writePolicy": "read-only"
|
|
48
|
+
},
|
|
49
|
+
"skillsByRole": {
|
|
50
|
+
"planner": [
|
|
51
|
+
"loop-agent"
|
|
52
|
+
],
|
|
53
|
+
"scout": [],
|
|
54
|
+
"implementer": [
|
|
55
|
+
"test-driven-development",
|
|
56
|
+
"verification-before-completion"
|
|
57
|
+
],
|
|
58
|
+
"reviewer": [
|
|
59
|
+
"requesting-code-review",
|
|
60
|
+
"code-review-core"
|
|
61
|
+
],
|
|
62
|
+
"verifier": [
|
|
63
|
+
"verification-before-completion",
|
|
64
|
+
"systematic-debugging"
|
|
65
|
+
],
|
|
66
|
+
"closeout": [
|
|
67
|
+
"loop-agent",
|
|
68
|
+
"verification-before-completion"
|
|
69
|
+
]
|
|
70
|
+
},
|
|
71
|
+
"executorModels": {
|
|
72
|
+
"pi": {
|
|
73
|
+
"LOW": "gpt-5.3-codex-spark",
|
|
74
|
+
"MED": "glm-5.2",
|
|
75
|
+
"HIGH": "gpt-5.5"
|
|
76
|
+
}
|
|
77
|
+
},
|
|
78
|
+
"tasks": [
|
|
79
|
+
{
|
|
80
|
+
"id": "analyze-inputs-pi",
|
|
81
|
+
"depends_on": [],
|
|
82
|
+
"complexity": "MED",
|
|
83
|
+
"executor": "pi",
|
|
84
|
+
"role": "planner",
|
|
85
|
+
"writePolicy": "read-only",
|
|
86
|
+
"allowedPaths": [
|
|
87
|
+
"REPLACE/WITH/SOURCE/PATH/**"
|
|
88
|
+
],
|
|
89
|
+
"forbiddenPaths": [
|
|
90
|
+
".harness/**",
|
|
91
|
+
"artifacts/**"
|
|
92
|
+
],
|
|
93
|
+
"outputContract": "Pure Backend Test Analysis v1 JSON object matching docs/templates/backend-test-analysis.schema.json. No Markdown prose and no file writes.",
|
|
94
|
+
"retryPolicy": {
|
|
95
|
+
"maxAttempts": 3,
|
|
96
|
+
"backoff": "exponential",
|
|
97
|
+
"initialDelayMs": 2000,
|
|
98
|
+
"maxDelayMs": 30000,
|
|
99
|
+
"retryCategories": [
|
|
100
|
+
"timeout",
|
|
101
|
+
"network",
|
|
102
|
+
"rate-limit",
|
|
103
|
+
"unavailable"
|
|
104
|
+
]
|
|
105
|
+
},
|
|
106
|
+
"subtask_prompt": "Read the task source materials and return exactly one JSON object matching Backend Test Analysis v1.\n\nDo not wrap it in explanatory prose. A single fenced json block is tolerated, but pure JSON is preferred.\n\nCopy taskId, requirementPath, requirementSha256, referencePaths, and requirementIds exactly from the DAG source binding shown below.\n\nPreserve existing AC IDs. Do not invent endpoint methods, paths, fields, errors, boundaries, or business rules; record unknowns in evidenceGaps.\n\nUse empty arrays for categories not documented. Never include credentials, tokens, private keys, or secret values.\n\nRequired top-level keys: schemaVersion, sourceBinding, acceptanceCriteria, endpoints, dataModels, businessRules, stateTransitions, boundaryConstraints, externalDependencies, risks, evidenceGaps.\n\nRead-only: do not modify code, docs, artifacts, or repository files."
|
|
107
|
+
},
|
|
108
|
+
{
|
|
109
|
+
"id": "backend-test-analysis-contract-shell",
|
|
110
|
+
"depends_on": [
|
|
111
|
+
"analyze-inputs-pi"
|
|
112
|
+
],
|
|
113
|
+
"complexity": "LOW",
|
|
114
|
+
"executor": "shell",
|
|
115
|
+
"role": "verifier",
|
|
116
|
+
"writePolicy": "read-only",
|
|
117
|
+
"allowedPaths": [
|
|
118
|
+
"REPLACE/WITH/SOURCE/PATH/**"
|
|
119
|
+
],
|
|
120
|
+
"forbiddenPaths": [
|
|
121
|
+
".harness/**",
|
|
122
|
+
"artifacts/**"
|
|
123
|
+
],
|
|
124
|
+
"outputContract": "Validated run-owned Backend Test Analysis v1 artifact pointer, schema ID, and SHA-256.",
|
|
125
|
+
"subtask_prompt": "Materialize and validate the backend-test analysis contract under the current DAG run.",
|
|
126
|
+
"shell": {
|
|
127
|
+
"commands": [],
|
|
128
|
+
"jsonArtifactGate": {
|
|
129
|
+
"fromNodeId": "analyze-inputs-pi",
|
|
130
|
+
"schemaId": "backend-test-analysis-v1",
|
|
131
|
+
"artifactName": "backend-test-analysis.json",
|
|
132
|
+
"outputDir": "contracts"
|
|
133
|
+
},
|
|
134
|
+
"cwd": ".",
|
|
135
|
+
"timeoutMs": 60000
|
|
136
|
+
}
|
|
137
|
+
},
|
|
138
|
+
{
|
|
139
|
+
"id": "generate-backend-functional-cases-pi",
|
|
140
|
+
"depends_on": [
|
|
141
|
+
"backend-test-analysis-contract-shell"
|
|
142
|
+
],
|
|
143
|
+
"complexity": "MED",
|
|
144
|
+
"executor": "pi",
|
|
145
|
+
"role": "implementer",
|
|
146
|
+
"toolProfile": "write",
|
|
147
|
+
"writePolicy": "exclusive",
|
|
148
|
+
"writeSet": [
|
|
149
|
+
"testcase/md/**"
|
|
150
|
+
],
|
|
151
|
+
"allowedPaths": [
|
|
152
|
+
"testcase/md/**"
|
|
153
|
+
],
|
|
154
|
+
"forbiddenPaths": [
|
|
155
|
+
".harness/**",
|
|
156
|
+
"artifacts/**"
|
|
157
|
+
],
|
|
158
|
+
"outputContract": "Structured backend functional test cases in Markdown under testcase/md/. Each case uses BE-<MODULE>-<NNN> ID format.",
|
|
159
|
+
"subtask_prompt": "Read the validated structured artifact pointer from backend-test-analysis-contract-shell and generate cases only from that JSON contract.\n\n## Output Steps (do in order):\n1. First, output a brief summary: how many modules, how many cases planned per module\n2. Then write each test case file under testcase/md/\n\n## Format Rules:\n- Each test case ID: BE-<MODULE>-<NNN> (e.g. BE-ORDER-001)\n- Each file covers one module\n- Case structure: ID, Title, Precondition, Steps, Expected Result\n- Map each case to acceptance criteria (AC-xxx)\n\n## Coverage Requirements:\n- Positive paths: happy path for each acceptance criterion\n- Negative paths: error scenarios (invalid input, not found, state violations)\n\n## Conditional Coverage (include ONLY if mentioned in upstream analysis):\n- Boundary conditions: include ONLY if upstream analyze-inputs-pi mentions value ranges, length limits, numeric bounds, or format constraints\n- State transitions: include ONLY if upstream analyze-inputs-pi mentions state machine\n- Authentication scenarios: include ONLY if upstream analyze-inputs-pi mentions auth mechanism\n- Timeout scenarios: include ONLY if upstream analyze-inputs-pi mentions timeout handling\n- Concurrency scenarios: include ONLY if upstream analyze-inputs-pi mentions concurrency/idempotency rules\n- If not mentioned, do NOT generate these test cases\n\n## Constraints:\n- Stay within writeSet: testcase/md/**\n- Do NOT re-read source documents or fall back to free-form analysis — use the validated structured artifact only\n- Do not write root artifacts/**"
|
|
160
|
+
},
|
|
161
|
+
{
|
|
162
|
+
"id": "review-backend-cases-pi",
|
|
163
|
+
"depends_on": [
|
|
164
|
+
"generate-backend-functional-cases-pi",
|
|
165
|
+
"backend-test-analysis-contract-shell"
|
|
166
|
+
],
|
|
167
|
+
"complexity": "HIGH",
|
|
168
|
+
"executor": "pi",
|
|
169
|
+
"role": "reviewer",
|
|
170
|
+
"writePolicy": "read-only",
|
|
171
|
+
"allowedPaths": [
|
|
172
|
+
"**"
|
|
173
|
+
],
|
|
174
|
+
"forbiddenPaths": [
|
|
175
|
+
".harness/**",
|
|
176
|
+
"artifacts/**"
|
|
177
|
+
],
|
|
178
|
+
"outputContract": "Plain Markdown whose first non-empty line is VERDICT: pass or VERDICT: request-revision; followed by Findings and Coverage Assessment. No file writes.",
|
|
179
|
+
"retryPolicy": {
|
|
180
|
+
"maxAttempts": 3,
|
|
181
|
+
"backoff": "exponential",
|
|
182
|
+
"initialDelayMs": 2000,
|
|
183
|
+
"maxDelayMs": 30000,
|
|
184
|
+
"retryCategories": [
|
|
185
|
+
"timeout",
|
|
186
|
+
"network",
|
|
187
|
+
"rate-limit",
|
|
188
|
+
"unavailable"
|
|
189
|
+
]
|
|
190
|
+
},
|
|
191
|
+
"subtask_prompt_markdown": "./backend-test-dag.review-cases.prompt.md"
|
|
192
|
+
},
|
|
193
|
+
{
|
|
194
|
+
"id": "review-backend-cases-gate-shell",
|
|
195
|
+
"depends_on": [
|
|
196
|
+
"review-backend-cases-pi"
|
|
197
|
+
],
|
|
198
|
+
"complexity": "LOW",
|
|
199
|
+
"executor": "shell",
|
|
200
|
+
"role": "verifier",
|
|
201
|
+
"writePolicy": "read-only",
|
|
202
|
+
"allowedPaths": [
|
|
203
|
+
"**"
|
|
204
|
+
],
|
|
205
|
+
"forbiddenPaths": [
|
|
206
|
+
".harness/**",
|
|
207
|
+
"artifacts/**"
|
|
208
|
+
],
|
|
209
|
+
"outputContract": "Deterministic backend case review gate: exit 0 only when review-backend-cases-pi emits VERDICT: pass.",
|
|
210
|
+
"subtask_prompt": "Deterministic gate: block pytest generation unless backend case review emitted VERDICT: pass.",
|
|
211
|
+
"shell": {
|
|
212
|
+
"commands": [],
|
|
213
|
+
"verdictGate": {
|
|
214
|
+
"fromNodeId": "review-backend-cases-pi",
|
|
215
|
+
"accept": [
|
|
216
|
+
"VERDICT: pass"
|
|
217
|
+
],
|
|
218
|
+
"label": "backend case review",
|
|
219
|
+
"lineMode": "first-verdict-line"
|
|
220
|
+
},
|
|
221
|
+
"cwd": ".",
|
|
222
|
+
"timeoutMs": 60000
|
|
223
|
+
}
|
|
224
|
+
},
|
|
225
|
+
{
|
|
226
|
+
"id": "generate-backend-pytest-pi",
|
|
227
|
+
"depends_on": [
|
|
228
|
+
"review-backend-cases-gate-shell"
|
|
229
|
+
],
|
|
230
|
+
"complexity": "HIGH",
|
|
231
|
+
"executor": "pi",
|
|
232
|
+
"role": "implementer",
|
|
233
|
+
"toolProfile": "write",
|
|
234
|
+
"writePolicy": "exclusive",
|
|
235
|
+
"writeSet": [
|
|
236
|
+
"testcase/**/test_*.py",
|
|
237
|
+
"testcase/**/helpers/**",
|
|
238
|
+
"testcase/**/factories/**"
|
|
239
|
+
],
|
|
240
|
+
"allowedPaths": [
|
|
241
|
+
"testcase/**",
|
|
242
|
+
"**"
|
|
243
|
+
],
|
|
244
|
+
"forbiddenPaths": [
|
|
245
|
+
".harness/**",
|
|
246
|
+
"artifacts/**"
|
|
247
|
+
],
|
|
248
|
+
"outputContract": "Pytest test files under testcase/ with 1:1 mapping to functional test case IDs; optional helpers/factories under testcase/**/helpers|factories. Summary lists generated files, test function count, and any skipped cases with reasons.",
|
|
249
|
+
"subtask_prompt_markdown": "./backend-test-dag.generate-pytest.prompt.md"
|
|
250
|
+
},
|
|
251
|
+
{
|
|
252
|
+
"id": "execute-backend-pytest-shell",
|
|
253
|
+
"depends_on": [
|
|
254
|
+
"generate-backend-pytest-pi"
|
|
255
|
+
],
|
|
256
|
+
"complexity": "LOW",
|
|
257
|
+
"executor": "shell",
|
|
258
|
+
"role": "verifier",
|
|
259
|
+
"writePolicy": "read-only",
|
|
260
|
+
"allowedPaths": [
|
|
261
|
+
"**"
|
|
262
|
+
],
|
|
263
|
+
"forbiddenPaths": [
|
|
264
|
+
".harness/**",
|
|
265
|
+
"artifacts/**"
|
|
266
|
+
],
|
|
267
|
+
"outputContract": "Archived pytest stdout/stderr with exit codes; JUnit XML is runner-owned evidence at $HARNESS_DAG_RUN_DIR/reports/backend-test-junit.xml. Must not modify worktree files, testcase sources, production code, or assertions.",
|
|
268
|
+
"subtask_prompt": "Run pytest for the backend test suite; write JUnit evidence only under the current HARNESS_DAG_RUN_DIR/reports/.",
|
|
269
|
+
"shell": {
|
|
270
|
+
"commands": [
|
|
271
|
+
"test -n \"${HARNESS_DAG_RUN_DIR:-}\" || { echo \"missing HARNESS_DAG_RUN_DIR for backend pytest report\" >&2; exit 2; }; REPORT=\"${HARNESS_DAG_RUN_DIR}/reports/backend-test-junit.xml\"; mkdir -p \"$(dirname \"${REPORT}\")\"; PYTHONDONTWRITEBYTECODE=1 python -m pytest testcase/ -v -p no:cacheprovider --junitxml=\"${REPORT}\"; STATUS=$?; printf \"JUnit report: %s\\n\" \"${REPORT}\"; exit \"${STATUS}\""
|
|
272
|
+
],
|
|
273
|
+
"verifyEvidence": {
|
|
274
|
+
"phase": "final",
|
|
275
|
+
"quota": "full",
|
|
276
|
+
"commandSource": "inline",
|
|
277
|
+
"commandCount": 1,
|
|
278
|
+
"commandLabels": [
|
|
279
|
+
"backend pytest execution"
|
|
280
|
+
],
|
|
281
|
+
"finalFullRequired": true
|
|
282
|
+
},
|
|
283
|
+
"cwd": ".",
|
|
284
|
+
"timeoutMs": 300000
|
|
285
|
+
}
|
|
286
|
+
},
|
|
287
|
+
{
|
|
288
|
+
"id": "test-retrospect-pi",
|
|
289
|
+
"depends_on": [
|
|
290
|
+
"execute-backend-pytest-shell"
|
|
291
|
+
],
|
|
292
|
+
"complexity": "MED",
|
|
293
|
+
"executor": "pi",
|
|
294
|
+
"role": "closeout",
|
|
295
|
+
"toolProfile": "write",
|
|
296
|
+
"writePolicy": "exclusive",
|
|
297
|
+
"writeSet": [
|
|
298
|
+
"docs/test-reports/**"
|
|
299
|
+
],
|
|
300
|
+
"allowedPaths": [
|
|
301
|
+
"docs/test-reports/**"
|
|
302
|
+
],
|
|
303
|
+
"forbiddenPaths": [
|
|
304
|
+
".harness/**",
|
|
305
|
+
"artifacts/**"
|
|
306
|
+
],
|
|
307
|
+
"outputContract": "Markdown retrospective report under docs/test-reports/ with coverage summary, review findings, pytest results, and maturity rating (A/B/C/D).",
|
|
308
|
+
"subtask_prompt_markdown": "./backend-test-dag.retrospect.prompt.md"
|
|
309
|
+
}
|
|
310
|
+
]
|
|
311
|
+
}
|