@tea-agent/loop-agent 0.5.0 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (133) hide show
  1. package/AGENTS.md +142 -142
  2. package/CHANGELOG.md +116 -98
  3. package/README.md +195 -195
  4. package/bin/agent-worker.js +22 -22
  5. package/bin/loop-agent.js +21 -21
  6. package/dist/application/dag/args.js +9 -1
  7. package/dist/application/dag/run-dag.js +16 -2
  8. package/dist/cli/command-definitions.js +22 -4
  9. package/dist/cli/help.js +3 -2
  10. package/dist/cli/program.js +7 -5
  11. package/dist/commands/import-prd.js +76 -0
  12. package/dist/commands/init.js +467 -457
  13. package/dist/commands/instructions.js +90 -58
  14. package/dist/commands/loop-benchmark.js +11 -11
  15. package/dist/commands/pi-reuse-benchmark.js +16 -16
  16. package/dist/executors/cursor-executor.js +1 -1
  17. package/dist/executors/dag-pi-executor.js +1 -0
  18. package/dist/executors/pi-sdk-executor.js +63 -1
  19. package/dist/shared/preview.js +39 -0
  20. package/dist/task/runtime.js +27 -27
  21. package/dist/task/source-references.js +221 -0
  22. package/dist/worker/cli.js +62 -1
  23. package/dist/worker/loop-agent/loop-agent-client.js +97 -5
  24. package/dist/worker/materialize/harness-task-materializer.js +162 -5
  25. package/dist/worker/observability/event-store.js +82 -0
  26. package/dist/worker/observability/events.js +79 -0
  27. package/dist/worker/observability/progress-composite.js +33 -0
  28. package/dist/worker/observability/read-model.js +1013 -0
  29. package/dist/worker/observability/snapshot-store.js +43 -0
  30. package/dist/worker/observability/types.js +1 -0
  31. package/dist/worker/observe/paths.js +64 -0
  32. package/dist/worker/observe/routes.js +423 -0
  33. package/dist/worker/observe/server.js +61 -0
  34. package/dist/worker/observe/static/app.js +1419 -0
  35. package/dist/worker/observe/static/index.html +63 -0
  36. package/dist/worker/observe/static/styles.css +613 -0
  37. package/dist/worker/pool/failure-routing.js +41 -6
  38. package/dist/worker/pool/run-store.js +50 -0
  39. package/dist/worker/progress-reporter.js +0 -18
  40. package/dist/worker/run-task/run-task.js +327 -92
  41. package/dist/worker/runner/run-ready.js +112 -4
  42. package/dist/workflows/dag/canvas-observer.js +275 -275
  43. package/dist/workflows/dag/event-observer.js +132 -0
  44. package/dist/workflows/dag/init-hybrid.js +146 -13
  45. package/dist/workflows/dag/observer-compose.js +52 -0
  46. package/docs/README.md +74 -72
  47. package/docs/agent-dag-recovery-playbook.md +184 -184
  48. package/docs/agent-dag-runner.md +42 -42
  49. package/docs/architecture/runtime-boundaries.md +162 -147
  50. package/docs/cursor-executor-usage.md +25 -25
  51. package/docs/decisions/README.md +3 -3
  52. package/docs/design/README.md +49 -36
  53. package/docs/development-principles.md +73 -73
  54. package/docs/dynamic-workflow-dag-engine-roadmap.md +1749 -1749
  55. package/docs/exec-plans/README.md +6 -6
  56. package/docs/exec-plans/active/README.md +12 -7
  57. package/docs/exec-plans/completed/README.md +31 -19
  58. package/docs/feature-workflow.md +186 -186
  59. package/docs/harness-methodology-debugging.md +153 -153
  60. package/docs/harness-methodology-tdd.md +130 -130
  61. package/docs/harness-methodology-verification.md +27 -27
  62. package/docs/init-surface.manifest.json +205 -199
  63. package/docs/loop-agent-harness.md +55 -42
  64. package/docs/production-readiness.md +96 -96
  65. package/docs/progress/README.md +3 -3
  66. package/docs/reports/README.md +9 -5
  67. package/docs/skills/README.md +6 -6
  68. package/docs/skills/vetted-skill-registry.md +26 -26
  69. package/docs/templates/adr.md +60 -60
  70. package/docs/templates/agent-dag-authority-surface-audit.prompt.md +94 -94
  71. package/docs/templates/agent-dag-decision-envelope.schema.json +213 -213
  72. package/docs/templates/agent-dag-decision-gate-dogfood-report.md +117 -117
  73. package/docs/templates/agent-dag-decision-gate.prompt.md +246 -246
  74. package/docs/templates/agent-dag-process-supervisor.prompt.md +98 -98
  75. package/docs/templates/agent-dag-report.schema.json +454 -454
  76. package/docs/templates/agent-dag-review-verdict.prompt.md +68 -68
  77. package/docs/templates/agent-dag.base.json +195 -195
  78. package/docs/templates/agent-dag.final-verification.json +190 -190
  79. package/docs/templates/agent-dag.schema.json +316 -316
  80. package/docs/templates/agent-dag.supervised-implementation.json +500 -500
  81. package/docs/templates/exec-plan.md +64 -64
  82. package/docs/templates/feature-spec.md +53 -53
  83. package/docs/templates/hybrid-dag.json +193 -193
  84. package/docs/templates/init-evolution-review.md +33 -33
  85. package/docs/templates/production-readiness-checklist.md +57 -57
  86. package/docs/templates/progress-log.md +17 -17
  87. package/docs/templates/project-start-checklist.md +9 -9
  88. package/docs/templates/qa-report.md +48 -48
  89. package/docs/templates/sprint-contract.md +29 -29
  90. package/docs/templates/worker-dogfood-evidence.md +52 -0
  91. package/docs/templates/worker-dogfood-setup.md +48 -0
  92. package/docs/verification-matrix.md +41 -41
  93. package/examples/decision-gate-agent-dag.json +123 -123
  94. package/examples/example-dag.json +51 -51
  95. package/examples/hybrid-loop-agent-dag.json +194 -194
  96. package/harness.json +70 -69
  97. package/package.json +66 -66
  98. package/skills/ai-engineering-context/SKILL.md +48 -48
  99. package/skills/code-review-core/SKILL.md +20 -20
  100. package/skills/codebase-scout/SKILL.md +19 -19
  101. package/skills/init-capability-evolution/SKILL.md +69 -69
  102. package/skills/loop-agent/SKILL.md +149 -147
  103. package/skills/loop-agent/references/README.md +67 -67
  104. package/skills/loop-agent/references/command-reference.md +412 -403
  105. package/skills/loop-agent/references/harness-policy.md +263 -259
  106. package/skills/loop-agent/references/hybrid-dag.md +216 -216
  107. package/skills/loop-agent/references/learned/README.md +21 -21
  108. package/skills/loop-agent/references/long-running-loop.md +59 -59
  109. package/skills/loop-agent/references/model-routing.md +36 -36
  110. package/skills/loop-agent/references/multi-worktree.md +54 -54
  111. package/skills/loop-agent/references/one-shot-runs.md +85 -85
  112. package/skills/loop-agent/references/orchestrator-and-interventions.md +169 -169
  113. package/skills/loop-agent/references/pi-prompt.md +23 -23
  114. package/skills/loop-agent/references/pi-subagent-assisted-mode.md +81 -81
  115. package/skills/loop-agent/references/post-implementation-and-patterns.md +44 -44
  116. package/skills/loop-agent/references/task-workflow.md +89 -84
  117. package/skills/loop-agent/references/verification-and-failure-handling.md +128 -128
  118. package/skills/requesting-code-review/SKILL.md +101 -101
  119. package/skills/requesting-code-review/code-reviewer.md +168 -168
  120. package/skills/systematic-debugging/CREATION-LOG.md +119 -119
  121. package/skills/systematic-debugging/SKILL.md +296 -296
  122. package/skills/systematic-debugging/condition-based-waiting-example.ts +158 -158
  123. package/skills/systematic-debugging/condition-based-waiting.md +115 -115
  124. package/skills/systematic-debugging/defense-in-depth.md +122 -122
  125. package/skills/systematic-debugging/find-polluter.sh +63 -63
  126. package/skills/systematic-debugging/root-cause-tracing.md +169 -169
  127. package/skills/systematic-debugging/test-academic.md +14 -14
  128. package/skills/systematic-debugging/test-pressure-1.md +58 -58
  129. package/skills/systematic-debugging/test-pressure-2.md +68 -68
  130. package/skills/systematic-debugging/test-pressure-3.md +69 -69
  131. package/skills/test-driven-development/SKILL.md +20 -20
  132. package/skills/verification-before-completion/SKILL.md +154 -154
  133. package/skills/webapp-testing/SKILL.md +19 -19
@@ -1,213 +1,213 @@
1
- {
2
- "$schema": "https://json-schema.org/draft/2020-12/schema",
3
- "$id": "https://loop-agent.local/schemas/agent-dag-decision-envelope.schema.json",
4
- "title": "Agent DAG Decision Envelope",
5
- "description": "Structured output contract for advisory Agent DAG AI Secretary / Decision Gate nodes. M1 treats this as a documentation contract; M3 runtime parser extracts a ```DECISION_ENVELOPE_JSON fenced block and validates with zod plus semantic cross-field checks (recommendedOption must match options[].id; auto-approve must not set requiresHuman=true).",
6
- "type": "object",
7
- "additionalProperties": false,
8
- "required": [
9
- "schemaVersion",
10
- "gateType",
11
- "decisionScope",
12
- "decision",
13
- "confidence",
14
- "riskLevel",
15
- "requiresHuman",
16
- "nextAction",
17
- "policyVersion",
18
- "policyChecks",
19
- "rationale",
20
- "evidence",
21
- "blockingFindings",
22
- "requiredRevisions",
23
- "riskFlags",
24
- "humanEscalation",
25
- "audit"
26
- ],
27
- "properties": {
28
- "schemaVersion": {
29
- "type": "integer",
30
- "const": 1
31
- },
32
- "gateType": {
33
- "type": "string",
34
- "enum": ["plan-gate", "side-effect-gate", "acceptance-gate", "closeout-gate"]
35
- },
36
- "decisionScope": {
37
- "type": "string",
38
- "enum": ["node", "rank", "dag-run", "task", "repo"]
39
- },
40
- "decision": {
41
- "type": "string",
42
- "enum": [
43
- "auto-approve",
44
- "approve-with-constraints",
45
- "request-revision",
46
- "run-more-verification",
47
- "split-followup",
48
- "reject",
49
- "escalate-to-human",
50
- "pause-wait-external"
51
- ]
52
- },
53
- "confidence": {
54
- "type": "number",
55
- "minimum": 0,
56
- "maximum": 1
57
- },
58
- "riskLevel": {
59
- "type": "string",
60
- "enum": ["low", "medium", "high", "critical"]
61
- },
62
- "requiresHuman": {
63
- "type": "boolean"
64
- },
65
- "nextAction": {
66
- "type": "string",
67
- "enum": [
68
- "continue",
69
- "rerun-implement",
70
- "rerun-verify",
71
- "run-targeted-check",
72
- "split-followup",
73
- "pause-and-ask",
74
- "abort"
75
- ]
76
- },
77
- "policyVersion": {
78
- "type": "string",
79
- "minLength": 1
80
- },
81
- "policyChecks": {
82
- "type": "object",
83
- "additionalProperties": false,
84
- "required": ["mustEscalateFlags", "evidenceComplete", "allowedAutoApprove"],
85
- "properties": {
86
- "mustEscalateFlags": {
87
- "type": "array",
88
- "items": { "type": "string", "minLength": 1 }
89
- },
90
- "evidenceComplete": {
91
- "type": "boolean"
92
- },
93
- "allowedAutoApprove": {
94
- "type": "boolean"
95
- }
96
- }
97
- },
98
- "rationale": {
99
- "type": "array",
100
- "minItems": 1,
101
- "items": { "type": "string", "minLength": 1 }
102
- },
103
- "evidence": {
104
- "type": "array",
105
- "minItems": 1,
106
- "items": {
107
- "type": "object",
108
- "additionalProperties": false,
109
- "required": ["path", "kind", "status", "summary"],
110
- "properties": {
111
- "path": { "type": "string", "minLength": 1 },
112
- "kind": { "type": "string", "minLength": 1 },
113
- "status": {
114
- "type": "string",
115
- "enum": ["verified", "partial", "self-reported", "conflicting", "missing"]
116
- },
117
- "summary": { "type": "string", "minLength": 1 }
118
- }
119
- }
120
- },
121
- "blockingFindings": {
122
- "type": "array",
123
- "items": { "type": "string", "minLength": 1 }
124
- },
125
- "requiredRevisions": {
126
- "type": "array",
127
- "items": { "type": "string", "minLength": 1 }
128
- },
129
- "riskFlags": {
130
- "type": "array",
131
- "items": { "type": "string", "minLength": 1 }
132
- },
133
- "humanEscalation": {
134
- "oneOf": [
135
- { "type": "null" },
136
- {
137
- "type": "object",
138
- "additionalProperties": false,
139
- "required": ["question", "recommendedOption", "options"],
140
- "properties": {
141
- "question": { "type": "string", "minLength": 1 },
142
- "recommendedOption": { "type": "string", "minLength": 1 },
143
- "options": {
144
- "type": "array",
145
- "minItems": 1,
146
- "items": {
147
- "type": "object",
148
- "additionalProperties": false,
149
- "required": ["id", "label", "risk", "reason"],
150
- "properties": {
151
- "id": { "type": "string", "minLength": 1 },
152
- "label": { "type": "string", "minLength": 1 },
153
- "risk": { "type": "string", "minLength": 1 },
154
- "reason": { "type": "string", "minLength": 1 }
155
- }
156
- }
157
- }
158
- }
159
- }
160
- ]
161
- },
162
- "audit": {
163
- "type": "object",
164
- "additionalProperties": true,
165
- "required": ["runId", "nodeId", "model"],
166
- "properties": {
167
- "runId": { "type": "string", "minLength": 1 },
168
- "nodeId": { "type": "string", "minLength": 1 },
169
- "model": { "type": "string", "minLength": 1 },
170
- "sourceHash": { "type": "string" }
171
- }
172
- }
173
- },
174
- "allOf": [
175
- {
176
- "if": {
177
- "properties": { "requiresHuman": { "const": true } },
178
- "required": ["requiresHuman"]
179
- },
180
- "then": {
181
- "properties": {
182
- "humanEscalation": { "type": "object" },
183
- "nextAction": { "const": "pause-and-ask" }
184
- },
185
- "required": ["humanEscalation"]
186
- }
187
- },
188
- {
189
- "if": {
190
- "properties": { "requiresHuman": { "const": false } },
191
- "required": ["requiresHuman"]
192
- },
193
- "then": {
194
- "properties": {
195
- "humanEscalation": { "type": "null" }
196
- }
197
- }
198
- },
199
- {
200
- "if": {
201
- "properties": {
202
- "decision": { "enum": ["escalate-to-human", "pause-wait-external"] }
203
- },
204
- "required": ["decision"]
205
- },
206
- "then": {
207
- "properties": {
208
- "requiresHuman": { "const": true }
209
- }
210
- }
211
- }
212
- ]
213
- }
1
+ {
2
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
3
+ "$id": "https://loop-agent.local/schemas/agent-dag-decision-envelope.schema.json",
4
+ "title": "Agent DAG Decision Envelope",
5
+ "description": "Structured output contract for advisory Agent DAG AI Secretary / Decision Gate nodes. M1 treats this as a documentation contract; M3 runtime parser extracts a ```DECISION_ENVELOPE_JSON fenced block and validates with zod plus semantic cross-field checks (recommendedOption must match options[].id; auto-approve must not set requiresHuman=true).",
6
+ "type": "object",
7
+ "additionalProperties": false,
8
+ "required": [
9
+ "schemaVersion",
10
+ "gateType",
11
+ "decisionScope",
12
+ "decision",
13
+ "confidence",
14
+ "riskLevel",
15
+ "requiresHuman",
16
+ "nextAction",
17
+ "policyVersion",
18
+ "policyChecks",
19
+ "rationale",
20
+ "evidence",
21
+ "blockingFindings",
22
+ "requiredRevisions",
23
+ "riskFlags",
24
+ "humanEscalation",
25
+ "audit"
26
+ ],
27
+ "properties": {
28
+ "schemaVersion": {
29
+ "type": "integer",
30
+ "const": 1
31
+ },
32
+ "gateType": {
33
+ "type": "string",
34
+ "enum": ["plan-gate", "side-effect-gate", "acceptance-gate", "closeout-gate"]
35
+ },
36
+ "decisionScope": {
37
+ "type": "string",
38
+ "enum": ["node", "rank", "dag-run", "task", "repo"]
39
+ },
40
+ "decision": {
41
+ "type": "string",
42
+ "enum": [
43
+ "auto-approve",
44
+ "approve-with-constraints",
45
+ "request-revision",
46
+ "run-more-verification",
47
+ "split-followup",
48
+ "reject",
49
+ "escalate-to-human",
50
+ "pause-wait-external"
51
+ ]
52
+ },
53
+ "confidence": {
54
+ "type": "number",
55
+ "minimum": 0,
56
+ "maximum": 1
57
+ },
58
+ "riskLevel": {
59
+ "type": "string",
60
+ "enum": ["low", "medium", "high", "critical"]
61
+ },
62
+ "requiresHuman": {
63
+ "type": "boolean"
64
+ },
65
+ "nextAction": {
66
+ "type": "string",
67
+ "enum": [
68
+ "continue",
69
+ "rerun-implement",
70
+ "rerun-verify",
71
+ "run-targeted-check",
72
+ "split-followup",
73
+ "pause-and-ask",
74
+ "abort"
75
+ ]
76
+ },
77
+ "policyVersion": {
78
+ "type": "string",
79
+ "minLength": 1
80
+ },
81
+ "policyChecks": {
82
+ "type": "object",
83
+ "additionalProperties": false,
84
+ "required": ["mustEscalateFlags", "evidenceComplete", "allowedAutoApprove"],
85
+ "properties": {
86
+ "mustEscalateFlags": {
87
+ "type": "array",
88
+ "items": { "type": "string", "minLength": 1 }
89
+ },
90
+ "evidenceComplete": {
91
+ "type": "boolean"
92
+ },
93
+ "allowedAutoApprove": {
94
+ "type": "boolean"
95
+ }
96
+ }
97
+ },
98
+ "rationale": {
99
+ "type": "array",
100
+ "minItems": 1,
101
+ "items": { "type": "string", "minLength": 1 }
102
+ },
103
+ "evidence": {
104
+ "type": "array",
105
+ "minItems": 1,
106
+ "items": {
107
+ "type": "object",
108
+ "additionalProperties": false,
109
+ "required": ["path", "kind", "status", "summary"],
110
+ "properties": {
111
+ "path": { "type": "string", "minLength": 1 },
112
+ "kind": { "type": "string", "minLength": 1 },
113
+ "status": {
114
+ "type": "string",
115
+ "enum": ["verified", "partial", "self-reported", "conflicting", "missing"]
116
+ },
117
+ "summary": { "type": "string", "minLength": 1 }
118
+ }
119
+ }
120
+ },
121
+ "blockingFindings": {
122
+ "type": "array",
123
+ "items": { "type": "string", "minLength": 1 }
124
+ },
125
+ "requiredRevisions": {
126
+ "type": "array",
127
+ "items": { "type": "string", "minLength": 1 }
128
+ },
129
+ "riskFlags": {
130
+ "type": "array",
131
+ "items": { "type": "string", "minLength": 1 }
132
+ },
133
+ "humanEscalation": {
134
+ "oneOf": [
135
+ { "type": "null" },
136
+ {
137
+ "type": "object",
138
+ "additionalProperties": false,
139
+ "required": ["question", "recommendedOption", "options"],
140
+ "properties": {
141
+ "question": { "type": "string", "minLength": 1 },
142
+ "recommendedOption": { "type": "string", "minLength": 1 },
143
+ "options": {
144
+ "type": "array",
145
+ "minItems": 1,
146
+ "items": {
147
+ "type": "object",
148
+ "additionalProperties": false,
149
+ "required": ["id", "label", "risk", "reason"],
150
+ "properties": {
151
+ "id": { "type": "string", "minLength": 1 },
152
+ "label": { "type": "string", "minLength": 1 },
153
+ "risk": { "type": "string", "minLength": 1 },
154
+ "reason": { "type": "string", "minLength": 1 }
155
+ }
156
+ }
157
+ }
158
+ }
159
+ }
160
+ ]
161
+ },
162
+ "audit": {
163
+ "type": "object",
164
+ "additionalProperties": true,
165
+ "required": ["runId", "nodeId", "model"],
166
+ "properties": {
167
+ "runId": { "type": "string", "minLength": 1 },
168
+ "nodeId": { "type": "string", "minLength": 1 },
169
+ "model": { "type": "string", "minLength": 1 },
170
+ "sourceHash": { "type": "string" }
171
+ }
172
+ }
173
+ },
174
+ "allOf": [
175
+ {
176
+ "if": {
177
+ "properties": { "requiresHuman": { "const": true } },
178
+ "required": ["requiresHuman"]
179
+ },
180
+ "then": {
181
+ "properties": {
182
+ "humanEscalation": { "type": "object" },
183
+ "nextAction": { "const": "pause-and-ask" }
184
+ },
185
+ "required": ["humanEscalation"]
186
+ }
187
+ },
188
+ {
189
+ "if": {
190
+ "properties": { "requiresHuman": { "const": false } },
191
+ "required": ["requiresHuman"]
192
+ },
193
+ "then": {
194
+ "properties": {
195
+ "humanEscalation": { "type": "null" }
196
+ }
197
+ }
198
+ },
199
+ {
200
+ "if": {
201
+ "properties": {
202
+ "decision": { "enum": ["escalate-to-human", "pause-wait-external"] }
203
+ },
204
+ "required": ["decision"]
205
+ },
206
+ "then": {
207
+ "properties": {
208
+ "requiresHuman": { "const": true }
209
+ }
210
+ }
211
+ }
212
+ ]
213
+ }
@@ -1,117 +1,117 @@
1
- # Agent DAG Decision Gate Dogfood Report Template
2
-
3
- > Copy to `docs/reports/YYYY-MM-DD-agent-dag-decision-gate-dogfood.md` after each dogfood run.
4
-
5
- ## Run Metadata
6
-
7
- | Field | Value |
8
- |---|---|
9
- | Date | YYYY-MM-DD |
10
- | Topic | |
11
- | DAG input | `<temp-dir>/<topic>-decision-gate-dag.json` |
12
- | Run ID | |
13
- | Run facts | `.harness/dag-runs/completed/<run-id>/` |
14
- | Gate type | `acceptance-gate` |
15
- | Decision node | `decision-pi` |
16
- | Model | `gpt-5.5` via `executorModels.pi.HIGH` |
17
-
18
- ## Work Type
19
-
20
- Choose one:
21
-
22
- - [ ] Low-risk docs/template change — expected `auto-approve` or `approve-with-constraints`
23
- - [ ] Runtime/loop-agent change — expected `request-revision`, `run-more-verification`, or `auto-approve`
24
- - [ ] Contract/security/high-risk simulation — expected `escalate-to-human`
25
-
26
- ## Deterministic Evidence
27
-
28
- | Evidence | Status | Notes |
29
- |---|---|---|
30
- | `verify-shell/result.summary.md` | verified / partial / missing | |
31
- | `state.json` | verified / partial / missing | |
32
- | git diff / changed file list | verified / partial / missing | |
33
- | progress/report/artifacts | verified / partial / missing | |
34
-
35
- ## Decision Envelope Summary
36
-
37
- Paste the **exact** ` ```DECISION_ENVELOPE_JSON ` fenced block from `decision-pi/result.summary.md` (info string must be `DECISION_ENVELOPE_JSON`, not `json`). Also record `decision`, `requiresHuman`, `nextAction`, and verified evidence count.
38
-
39
- ```DECISION_ENVELOPE_JSON
40
- {
41
- "schemaVersion": 1,
42
- "gateType": "acceptance-gate",
43
- "decisionScope": "dag-run",
44
- "decision": "auto-approve",
45
- "confidence": 0.88,
46
- "riskLevel": "low",
47
- "requiresHuman": false,
48
- "nextAction": "continue",
49
- "policyVersion": "agent-dag-decision-gate-v1",
50
- "policyChecks": {
51
- "mustEscalateFlags": [],
52
- "evidenceComplete": true,
53
- "allowedAutoApprove": true
54
- },
55
- "rationale": ["..."],
56
- "evidence": [
57
- {
58
- "path": ".harness/dag-runs/completed/<run-id>/verify-shell/result.summary.md",
59
- "kind": "shell-output",
60
- "status": "verified",
61
- "summary": "verification commands passed"
62
- }
63
- ],
64
- "blockingFindings": [],
65
- "requiredRevisions": [],
66
- "riskFlags": [],
67
- "humanEscalation": null,
68
- "audit": {
69
- "runId": "<run-id>",
70
- "nodeId": "decision-pi",
71
- "model": "gpt-5.5"
72
- }
73
- }
74
- ```
75
-
76
- ## Quality Metrics
77
-
78
- | Metric | Value | Notes |
79
- |---|---:|---|
80
- | `decisionLatencyMs` | | From node duration |
81
- | `tokenCostEstimate` | | If available |
82
- | `humanInterruptCount` | | Should be 0 for low-risk auto decisions |
83
- | `falseEscalation` | yes / no | Escalated when policy allowed auto decision |
84
- | `missedEscalation` | yes / no | Auto-approved when policy required human escalation |
85
- | `revisionCaughtBeforeHuman` | yes / no | Gate requested revision before human involvement |
86
- | `verifyPassAfterGate` | yes / no | Later verification passed after gate action |
87
- | `evidenceCompleteness` | complete / partial / missing | Based on verified evidence count |
88
-
89
- ## Evaluation
90
-
91
- ### What the gate got right
92
-
93
- -
94
-
95
- ### What the gate got wrong or over/under-weighted
96
-
97
- -
98
-
99
- ### Prompt / policy adjustments needed
100
-
101
- -
102
-
103
- ## Recommendation
104
-
105
- Choose one:
106
-
107
- - [ ] `proceed` — enough evidence to move toward parser/runtime work
108
- - [ ] `needs-more-dogfood` — continue M1/M2 advisory runs
109
- - [ ] `defer` — gate quality not yet good enough
110
-
111
- Rationale:
112
-
113
- -
114
-
115
- ## Follow-ups
116
-
117
- -
1
+ # Agent DAG Decision Gate Dogfood Report Template
2
+
3
+ > Copy to `docs/reports/YYYY-MM-DD-agent-dag-decision-gate-dogfood.md` after each dogfood run.
4
+
5
+ ## Run Metadata
6
+
7
+ | Field | Value |
8
+ |---|---|
9
+ | Date | YYYY-MM-DD |
10
+ | Topic | |
11
+ | DAG input | `<temp-dir>/<topic>-decision-gate-dag.json` |
12
+ | Run ID | |
13
+ | Run facts | `.harness/dag-runs/completed/<run-id>/` |
14
+ | Gate type | `acceptance-gate` |
15
+ | Decision node | `decision-pi` |
16
+ | Model | `gpt-5.5` via `executorModels.pi.HIGH` |
17
+
18
+ ## Work Type
19
+
20
+ Choose one:
21
+
22
+ - [ ] Low-risk docs/template change — expected `auto-approve` or `approve-with-constraints`
23
+ - [ ] Runtime/loop-agent change — expected `request-revision`, `run-more-verification`, or `auto-approve`
24
+ - [ ] Contract/security/high-risk simulation — expected `escalate-to-human`
25
+
26
+ ## Deterministic Evidence
27
+
28
+ | Evidence | Status | Notes |
29
+ |---|---|---|
30
+ | `verify-shell/result.summary.md` | verified / partial / missing | |
31
+ | `state.json` | verified / partial / missing | |
32
+ | git diff / changed file list | verified / partial / missing | |
33
+ | progress/report/artifacts | verified / partial / missing | |
34
+
35
+ ## Decision Envelope Summary
36
+
37
+ Paste the **exact** ` ```DECISION_ENVELOPE_JSON ` fenced block from `decision-pi/result.summary.md` (info string must be `DECISION_ENVELOPE_JSON`, not `json`). Also record `decision`, `requiresHuman`, `nextAction`, and verified evidence count.
38
+
39
+ ```DECISION_ENVELOPE_JSON
40
+ {
41
+ "schemaVersion": 1,
42
+ "gateType": "acceptance-gate",
43
+ "decisionScope": "dag-run",
44
+ "decision": "auto-approve",
45
+ "confidence": 0.88,
46
+ "riskLevel": "low",
47
+ "requiresHuman": false,
48
+ "nextAction": "continue",
49
+ "policyVersion": "agent-dag-decision-gate-v1",
50
+ "policyChecks": {
51
+ "mustEscalateFlags": [],
52
+ "evidenceComplete": true,
53
+ "allowedAutoApprove": true
54
+ },
55
+ "rationale": ["..."],
56
+ "evidence": [
57
+ {
58
+ "path": ".harness/dag-runs/completed/<run-id>/verify-shell/result.summary.md",
59
+ "kind": "shell-output",
60
+ "status": "verified",
61
+ "summary": "verification commands passed"
62
+ }
63
+ ],
64
+ "blockingFindings": [],
65
+ "requiredRevisions": [],
66
+ "riskFlags": [],
67
+ "humanEscalation": null,
68
+ "audit": {
69
+ "runId": "<run-id>",
70
+ "nodeId": "decision-pi",
71
+ "model": "gpt-5.5"
72
+ }
73
+ }
74
+ ```
75
+
76
+ ## Quality Metrics
77
+
78
+ | Metric | Value | Notes |
79
+ |---|---:|---|
80
+ | `decisionLatencyMs` | | From node duration |
81
+ | `tokenCostEstimate` | | If available |
82
+ | `humanInterruptCount` | | Should be 0 for low-risk auto decisions |
83
+ | `falseEscalation` | yes / no | Escalated when policy allowed auto decision |
84
+ | `missedEscalation` | yes / no | Auto-approved when policy required human escalation |
85
+ | `revisionCaughtBeforeHuman` | yes / no | Gate requested revision before human involvement |
86
+ | `verifyPassAfterGate` | yes / no | Later verification passed after gate action |
87
+ | `evidenceCompleteness` | complete / partial / missing | Based on verified evidence count |
88
+
89
+ ## Evaluation
90
+
91
+ ### What the gate got right
92
+
93
+ -
94
+
95
+ ### What the gate got wrong or over/under-weighted
96
+
97
+ -
98
+
99
+ ### Prompt / policy adjustments needed
100
+
101
+ -
102
+
103
+ ## Recommendation
104
+
105
+ Choose one:
106
+
107
+ - [ ] `proceed` — enough evidence to move toward parser/runtime work
108
+ - [ ] `needs-more-dogfood` — continue M1/M2 advisory runs
109
+ - [ ] `defer` — gate quality not yet good enough
110
+
111
+ Rationale:
112
+
113
+ -
114
+
115
+ ## Follow-ups
116
+
117
+ -