@tea-agent/loop-agent 0.13.0 → 0.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (270) hide show
  1. package/AGENTS.md +157 -157
  2. package/CHANGELOG.md +73 -301
  3. package/README.md +338 -334
  4. package/bin/agent-worker.js +22 -22
  5. package/bin/loop-agent.js +21 -21
  6. package/dist/commands/cursor-prompt.js +6 -6
  7. package/dist/commands/init.js +505 -505
  8. package/dist/commands/loop-benchmark.js +11 -11
  9. package/dist/commands/pi-reuse-benchmark.js +16 -16
  10. package/dist/executors/pi-event-serializer.js +33 -11
  11. package/dist/sidecars/cursor-prompt/executor.js +1 -1
  12. package/dist/task/runtime.js +27 -27
  13. package/dist/worker/observe/spec-evidence.js +19 -10
  14. package/dist/worker/observe/static/api.js +46 -46
  15. package/dist/worker/observe/static/app.js +151 -150
  16. package/dist/worker/observe/static/constants.js +156 -148
  17. package/dist/worker/observe/static/copy.js +67 -67
  18. package/dist/worker/observe/static/dag-helpers.js +201 -172
  19. package/dist/worker/observe/static/dag-layout.d.ts +31 -31
  20. package/dist/worker/observe/static/dag-layout.js +83 -83
  21. package/dist/worker/observe/static/dag-model.js +72 -72
  22. package/dist/worker/observe/static/dom.js +122 -122
  23. package/dist/worker/observe/static/format-pool.d.ts +71 -0
  24. package/dist/worker/observe/static/format-pool.js +134 -67
  25. package/dist/worker/observe/static/format.js +317 -292
  26. package/dist/worker/observe/static/index.html +350 -308
  27. package/dist/worker/observe/static/kpi.js +100 -94
  28. package/dist/worker/observe/static/markdown-render.js +124 -0
  29. package/dist/worker/observe/static/relations.js +133 -133
  30. package/dist/worker/observe/static/router.js +93 -93
  31. package/dist/worker/observe/static/run-processing.js +148 -148
  32. package/dist/worker/observe/static/shell-chrome.js +74 -68
  33. package/dist/worker/observe/static/state.js +273 -267
  34. package/dist/worker/observe/static/styles.css +2504 -1902
  35. package/dist/worker/observe/static/views/batch.js +227 -227
  36. package/dist/worker/observe/static/views/dag-graph.js +172 -172
  37. package/dist/worker/observe/static/views/dag-inspector.js +530 -627
  38. package/dist/worker/observe/static/views/dag.js +371 -371
  39. package/dist/worker/observe/static/views/dashboard.js +86 -100
  40. package/dist/worker/observe/static/views/failures.js +143 -143
  41. package/dist/worker/observe/static/views/feature.js +492 -492
  42. package/dist/worker/observe/static/views/pool.js +708 -350
  43. package/dist/worker/observe/static/views/run.js +453 -453
  44. package/dist/worker/observe/static/views/session-timeline.js +771 -219
  45. package/dist/worker/observe/static/views/shell.js +7 -7
  46. package/dist/worker/observe/static/views/task.js +314 -314
  47. package/dist/worker/observe/static/views/timeline.js +163 -163
  48. package/dist/workflows/dag/canvas-observer.js +275 -275
  49. package/docs/README.md +105 -104
  50. package/docs/agent-dag-recovery-playbook.md +195 -195
  51. package/docs/agent-dag-runner.md +67 -67
  52. package/docs/architecture/README.md +26 -26
  53. package/docs/architecture/dag-execution.md +140 -140
  54. package/docs/architecture/evolution.md +54 -54
  55. package/docs/architecture/facts-and-state.md +71 -71
  56. package/docs/architecture/runtime-boundaries.md +191 -191
  57. package/docs/architecture/system-overview.md +93 -93
  58. package/docs/architecture/worker-and-feature.md +85 -85
  59. package/docs/cursor-prompt-sidecar.md +36 -36
  60. package/docs/decisions/README.md +18 -18
  61. package/docs/design/README.md +167 -167
  62. package/docs/development-principles.md +73 -73
  63. package/docs/exec-plans/README.md +6 -6
  64. package/docs/exec-plans/active/README.md +2 -1
  65. package/docs/exec-plans/completed/README.md +105 -104
  66. package/docs/feature-workflow.md +414 -414
  67. package/docs/harness-methodology-debugging.md +153 -153
  68. package/docs/harness-methodology-tdd.md +130 -130
  69. package/docs/harness-methodology-verification.md +27 -27
  70. package/docs/init-surface.manifest.json +307 -307
  71. package/docs/loop-agent-harness.md +142 -142
  72. package/docs/production-readiness.md +96 -96
  73. package/docs/progress/README.md +59 -58
  74. package/docs/reports/README.md +123 -119
  75. package/docs/skills/README.md +7 -7
  76. package/docs/skills/vetted-skill-registry.md +29 -29
  77. package/docs/templates/adr.md +60 -60
  78. package/docs/templates/agent-dag-authority-surface-audit.prompt.md +94 -94
  79. package/docs/templates/agent-dag-decision-envelope.schema.json +213 -213
  80. package/docs/templates/agent-dag-decision-gate-dogfood-report.md +117 -117
  81. package/docs/templates/agent-dag-decision-gate.prompt.md +246 -246
  82. package/docs/templates/agent-dag-process-supervisor.prompt.md +98 -98
  83. package/docs/templates/agent-dag-report.schema.json +473 -473
  84. package/docs/templates/agent-dag-review-verdict.prompt.md +68 -68
  85. package/docs/templates/agent-dag.base.json +190 -190
  86. package/docs/templates/agent-dag.final-verification.json +185 -185
  87. package/docs/templates/agent-dag.schema.json +411 -411
  88. package/docs/templates/agent-dag.supervised-implementation.json +620 -620
  89. package/docs/templates/backend-test-analysis.schema.json +44 -44
  90. package/docs/templates/backend-test-case-manifest.schema.json +190 -190
  91. package/docs/templates/backend-test-dag.classify.prompt.md +75 -75
  92. package/docs/templates/backend-test-dag.generate-pytest.prompt.md +204 -204
  93. package/docs/templates/backend-test-dag.json +559 -559
  94. package/docs/templates/backend-test-dag.retrospect.prompt.md +139 -139
  95. package/docs/templates/backend-test-dag.review-cases.prompt.md +83 -83
  96. package/docs/templates/backend-test-execution.schema.json +133 -133
  97. package/docs/templates/backend-test-result.schema.json +99 -99
  98. package/docs/templates/exec-plan.md +64 -64
  99. package/docs/templates/feature-spec.md +53 -53
  100. package/docs/templates/frontend-design-contract.md +42 -42
  101. package/docs/templates/frontend-eval/fixtures/failures/01-type-build-error.md +17 -17
  102. package/docs/templates/frontend-eval/fixtures/failures/02-unit-component-test-fail.md +16 -16
  103. package/docs/templates/frontend-eval/fixtures/failures/03-fixture-schema-drift.md +16 -16
  104. package/docs/templates/frontend-eval/fixtures/failures/04-missing-loading-empty-error-state.md +16 -16
  105. package/docs/templates/frontend-eval/fixtures/failures/05-forbidden-write-writeset-expansion.md +16 -16
  106. package/docs/templates/frontend-eval/fixtures/failures/06-unapproved-dependency-add.md +16 -16
  107. package/docs/templates/frontend-eval/fixtures/failures/07-mock-production-on.md +21 -21
  108. package/docs/templates/frontend-eval/fixtures/functional/01-simple-component-style.md +29 -29
  109. package/docs/templates/frontend-eval/fixtures/functional/02-form-validation.md +28 -28
  110. package/docs/templates/frontend-eval/fixtures/functional/03-list-detail-page.md +28 -28
  111. package/docs/templates/frontend-eval/fixtures/functional/04-api-mock.md +29 -29
  112. package/docs/templates/frontend-eval/fixtures/functional/05-permission-auth-gated-ui.md +27 -27
  113. package/docs/templates/frontend-eval/fixtures/functional/06-ssr-server-client-boundary.md +28 -28
  114. package/docs/templates/frontend-eval/fixtures/functional/07-shared-public-component-api.md +28 -28
  115. package/docs/templates/frontend-eval/fixtures/functional/08-pure-local-no-remote.md +27 -27
  116. package/docs/templates/frontend-eval/metrics.md +138 -138
  117. package/docs/templates/frontend-eval/smoke-targets.md +53 -53
  118. package/docs/templates/frontend-implementation-contract.schema.json +27 -27
  119. package/docs/templates/frontend-task-constraints.md +35 -35
  120. package/docs/templates/frontend-task-requirement.md +70 -70
  121. package/docs/templates/frontend-test-dag.generate-cases.prompt.md +5 -5
  122. package/docs/templates/frontend-test-dag.json +23 -23
  123. package/docs/templates/frontend-test-dag.retrieve-context.prompt.md +3 -3
  124. package/docs/templates/frontend-test-dag.retrospect.prompt.md +3 -3
  125. package/docs/templates/frontend-test-dag.review-cases.prompt.md +3 -3
  126. package/docs/templates/frontend-test-dag.review-execution.prompt.md +3 -3
  127. package/docs/templates/harness.schema.json +221 -221
  128. package/docs/templates/hybrid-dag.json +188 -188
  129. package/docs/templates/init-evolution-review.md +35 -35
  130. package/docs/templates/interactive-ui-round2-experiment.md +66 -66
  131. package/docs/templates/knowledge-graph-bootstrap-dag.json +118 -118
  132. package/docs/templates/knowledge-sync-dag.json +178 -178
  133. package/docs/templates/knowledge-sync-draft.schema.json +71 -71
  134. package/docs/templates/product-line/AGENTS.md +8 -8
  135. package/docs/templates/product-line/README.md +9 -9
  136. package/docs/templates/product-line/acceptance.yaml +14 -14
  137. package/docs/templates/product-line/closeout.yaml +9 -9
  138. package/docs/templates/product-line/design.md +13 -13
  139. package/docs/templates/product-line/links.md +10 -10
  140. package/docs/templates/product-line/requirement.md +17 -17
  141. package/docs/templates/product-line/task-graph.yaml +15 -15
  142. package/docs/templates/product-line/task.yaml +64 -64
  143. package/docs/templates/product-line/test-plan.md +7 -7
  144. package/docs/templates/production-readiness-checklist.md +57 -57
  145. package/docs/templates/progress-log.md +17 -17
  146. package/docs/templates/project-start-checklist.md +9 -9
  147. package/docs/templates/qa-report.md +48 -48
  148. package/docs/templates/sprint-contract.md +29 -29
  149. package/docs/templates/worker-dogfood-evidence.md +80 -80
  150. package/docs/templates/worker-dogfood-setup.md +68 -68
  151. package/docs/verification-matrix.md +70 -70
  152. package/examples/decision-gate-agent-dag.json +173 -173
  153. package/examples/example-dag.json +46 -46
  154. package/examples/hybrid-loop-agent-dag.json +188 -188
  155. package/harness.json +66 -66
  156. package/package.json +88 -52
  157. package/scripts/check-product-line-docs.sh +29 -29
  158. package/scripts/check-task-pool-root.sh +32 -32
  159. package/scripts/kb-bootstrap-init-skeleton.sh +240 -240
  160. package/scripts/kb-graph-incremental-prepare.mjs +386 -386
  161. package/scripts/kb-graph-incremental-prepare.sh +5 -5
  162. package/scripts/kb-graph-materialize.mjs +105 -105
  163. package/scripts/kb-graph-materialize.sh +4 -4
  164. package/scripts/kb-graph-promote.mjs +164 -164
  165. package/scripts/kb-graph-promote.sh +4 -4
  166. package/scripts/kb-query.mjs +554 -554
  167. package/scripts/kb-query.sh +5 -5
  168. package/skills/agent-worker/SKILL.md +39 -39
  169. package/skills/agent-worker/references/agent-worker-operator.md +60 -60
  170. package/skills/ai-engineering-context/SKILL.md +48 -48
  171. package/skills/analyze-product-dependencies/SKILL.md +67 -67
  172. package/skills/analyze-product-dependencies/agents/openai.yaml +4 -4
  173. package/skills/analyze-product-dependencies/references/api-documentation-schema.md +30 -30
  174. package/skills/analyze-product-dependencies/references/dependency-analysis-schema.md +28 -28
  175. package/skills/analyze-product-dependencies/references/example.md +76 -76
  176. package/skills/analyze-product-dependencies/references/forward-test-cases.md +35 -35
  177. package/skills/analyze-product-dependencies/references/input-contract.md +11 -11
  178. package/skills/analyze-product-dependencies/references/scouting-rules.md +61 -61
  179. package/skills/analyze-product-dependencies/scripts/test-validators.mjs +267 -267
  180. package/skills/analyze-product-dependencies/scripts/validate-api-documentation.mjs +101 -101
  181. package/skills/analyze-product-dependencies/scripts/validate-dependency-analysis.mjs +142 -142
  182. package/skills/analyze-product-dependencies/scripts/validate-product-requirement-input.mjs +76 -76
  183. package/skills/analyze-product-dependencies/scripts/validation-helpers.mjs +146 -146
  184. package/skills/analyze-product-requirements/SKILL.md +90 -90
  185. package/skills/analyze-product-requirements/agents/openai.yaml +4 -4
  186. package/skills/analyze-product-requirements/references/acceptance-criteria.md +91 -91
  187. package/skills/analyze-product-requirements/references/clarification-and-knowledge.md +56 -56
  188. package/skills/analyze-product-requirements/references/example.md +86 -86
  189. package/skills/analyze-product-requirements/references/forward-test-cases.md +66 -66
  190. package/skills/analyze-product-requirements/references/product-analysis-schema.md +32 -32
  191. package/skills/analyze-product-requirements/references/product-requirement-schema.md +33 -33
  192. package/skills/analyze-product-requirements/references/requirement-clarification-schema.md +35 -35
  193. package/skills/analyze-product-requirements/scripts/test-validators.mjs +193 -193
  194. package/skills/analyze-product-requirements/scripts/validate-product-analysis.mjs +69 -69
  195. package/skills/analyze-product-requirements/scripts/validate-product-requirement.mjs +97 -97
  196. package/skills/analyze-product-requirements/scripts/validate-requirement-clarification.mjs +98 -98
  197. package/skills/analyze-product-requirements/scripts/validation-helpers.mjs +156 -156
  198. package/skills/browser-tools/SKILL.md +196 -196
  199. package/skills/browser-tools/browser-content.js +103 -103
  200. package/skills/browser-tools/browser-cookies.js +35 -35
  201. package/skills/browser-tools/browser-eval.js +53 -53
  202. package/skills/browser-tools/browser-hn-scraper.js +108 -108
  203. package/skills/browser-tools/browser-nav.js +44 -44
  204. package/skills/browser-tools/browser-pick.js +162 -162
  205. package/skills/browser-tools/browser-screenshot.js +34 -34
  206. package/skills/browser-tools/browser-start.js +86 -86
  207. package/skills/browser-tools/package-lock.json +2556 -2556
  208. package/skills/browser-tools/package.json +19 -19
  209. package/skills/code-review-core/SKILL.md +20 -20
  210. package/skills/codebase-scout/SKILL.md +19 -19
  211. package/skills/frontend-design-review/SKILL.md +66 -66
  212. package/skills/frontend-design-review/references/review-checklist.md +58 -58
  213. package/skills/frontend-implementation/SKILL.md +49 -49
  214. package/skills/frontend-implementation/references/code-standards.md +32 -32
  215. package/skills/frontend-implementation/references/design-spec.md +46 -46
  216. package/skills/frontend-implementation/references/node-contracts.md +27 -27
  217. package/skills/frontend-review/SKILL.md +59 -59
  218. package/skills/frontend-review/references/review-findings.md +47 -47
  219. package/skills/frontend-verification/SKILL.md +53 -53
  220. package/skills/frontend-verification/references/verification-checklist.md +68 -68
  221. package/skills/grill-me/SKILL.md +10 -10
  222. package/skills/grill-with-docs/SKILL.md +88 -88
  223. package/skills/grill-with-docs/adr-format.md +47 -47
  224. package/skills/grill-with-docs/context-format.md +60 -60
  225. package/skills/init-capability-evolution/SKILL.md +70 -70
  226. package/skills/loop-agent/SKILL.md +151 -151
  227. package/skills/loop-agent/references/README.md +67 -67
  228. package/skills/loop-agent/references/command-reference.md +527 -527
  229. package/skills/loop-agent/references/docs-converge.md +126 -126
  230. package/skills/loop-agent/references/harness-policy.md +263 -263
  231. package/skills/loop-agent/references/hybrid-dag.md +243 -243
  232. package/skills/loop-agent/references/learned/README.md +21 -21
  233. package/skills/loop-agent/references/long-running-loop.md +57 -57
  234. package/skills/loop-agent/references/model-routing.md +36 -36
  235. package/skills/loop-agent/references/multi-worktree.md +54 -54
  236. package/skills/loop-agent/references/one-shot-runs.md +85 -85
  237. package/skills/loop-agent/references/orchestrator-and-interventions.md +169 -169
  238. package/skills/loop-agent/references/pi-prompt.md +23 -23
  239. package/skills/loop-agent/references/pi-subagent-assisted-mode.md +84 -84
  240. package/skills/loop-agent/references/post-implementation-and-patterns.md +44 -44
  241. package/skills/loop-agent/references/task-workflow.md +89 -89
  242. package/skills/loop-agent/references/verification-and-failure-handling.md +141 -141
  243. package/skills/playwright-cli/SKILL.md +420 -420
  244. package/skills/playwright-cli/references/element-attributes.md +23 -23
  245. package/skills/playwright-cli/references/playwright-tests.md +39 -39
  246. package/skills/playwright-cli/references/request-mocking.md +87 -87
  247. package/skills/playwright-cli/references/running-code.md +241 -241
  248. package/skills/playwright-cli/references/session-management.md +225 -225
  249. package/skills/playwright-cli/references/storage-state.md +275 -275
  250. package/skills/playwright-cli/references/test-generation.md +433 -433
  251. package/skills/playwright-cli/references/tracing.md +139 -139
  252. package/skills/playwright-cli/references/video-recording.md +143 -143
  253. package/skills/playwright-cli-case-generator/SKILL.md +74 -74
  254. package/skills/requesting-code-review/SKILL.md +101 -101
  255. package/skills/requesting-code-review/code-reviewer.md +168 -168
  256. package/skills/systematic-debugging/CREATION-LOG.md +119 -119
  257. package/skills/systematic-debugging/SKILL.md +296 -296
  258. package/skills/systematic-debugging/condition-based-waiting-example.ts +158 -158
  259. package/skills/systematic-debugging/condition-based-waiting.md +115 -115
  260. package/skills/systematic-debugging/defense-in-depth.md +122 -122
  261. package/skills/systematic-debugging/find-polluter.sh +63 -63
  262. package/skills/systematic-debugging/root-cause-tracing.md +169 -169
  263. package/skills/systematic-debugging/test-academic.md +14 -14
  264. package/skills/systematic-debugging/test-pressure-1.md +58 -58
  265. package/skills/systematic-debugging/test-pressure-2.md +68 -68
  266. package/skills/systematic-debugging/test-pressure-3.md +69 -69
  267. package/skills/test-driven-development/SKILL.md +20 -20
  268. package/skills/using-git-worktrees/SKILL.md +215 -215
  269. package/skills/verification-before-completion/SKILL.md +154 -154
  270. package/skills/webapp-testing/SKILL.md +19 -19
@@ -1,213 +1,213 @@
1
- {
2
- "$schema": "https://json-schema.org/draft/2020-12/schema",
3
- "$id": "https://loop-agent.local/schemas/agent-dag-decision-envelope.schema.json",
4
- "title": "Agent DAG Decision Envelope",
5
- "description": "Structured output contract for advisory Agent DAG AI Secretary / Decision Gate nodes. M1 treats this as a documentation contract; M3 runtime parser extracts a ```DECISION_ENVELOPE_JSON fenced block and validates with zod plus semantic cross-field checks (recommendedOption must match options[].id; auto-approve must not set requiresHuman=true).",
6
- "type": "object",
7
- "additionalProperties": false,
8
- "required": [
9
- "schemaVersion",
10
- "gateType",
11
- "decisionScope",
12
- "decision",
13
- "confidence",
14
- "riskLevel",
15
- "requiresHuman",
16
- "nextAction",
17
- "policyVersion",
18
- "policyChecks",
19
- "rationale",
20
- "evidence",
21
- "blockingFindings",
22
- "requiredRevisions",
23
- "riskFlags",
24
- "humanEscalation",
25
- "audit"
26
- ],
27
- "properties": {
28
- "schemaVersion": {
29
- "type": "integer",
30
- "const": 1
31
- },
32
- "gateType": {
33
- "type": "string",
34
- "enum": ["plan-gate", "side-effect-gate", "acceptance-gate", "closeout-gate"]
35
- },
36
- "decisionScope": {
37
- "type": "string",
38
- "enum": ["node", "rank", "dag-run", "task", "repo"]
39
- },
40
- "decision": {
41
- "type": "string",
42
- "enum": [
43
- "auto-approve",
44
- "approve-with-constraints",
45
- "request-revision",
46
- "run-more-verification",
47
- "split-followup",
48
- "reject",
49
- "escalate-to-human",
50
- "pause-wait-external"
51
- ]
52
- },
53
- "confidence": {
54
- "type": "number",
55
- "minimum": 0,
56
- "maximum": 1
57
- },
58
- "riskLevel": {
59
- "type": "string",
60
- "enum": ["low", "medium", "high", "critical"]
61
- },
62
- "requiresHuman": {
63
- "type": "boolean"
64
- },
65
- "nextAction": {
66
- "type": "string",
67
- "enum": [
68
- "continue",
69
- "rerun-implement",
70
- "rerun-verify",
71
- "run-targeted-check",
72
- "split-followup",
73
- "pause-and-ask",
74
- "abort"
75
- ]
76
- },
77
- "policyVersion": {
78
- "type": "string",
79
- "minLength": 1
80
- },
81
- "policyChecks": {
82
- "type": "object",
83
- "additionalProperties": false,
84
- "required": ["mustEscalateFlags", "evidenceComplete", "allowedAutoApprove"],
85
- "properties": {
86
- "mustEscalateFlags": {
87
- "type": "array",
88
- "items": { "type": "string", "minLength": 1 }
89
- },
90
- "evidenceComplete": {
91
- "type": "boolean"
92
- },
93
- "allowedAutoApprove": {
94
- "type": "boolean"
95
- }
96
- }
97
- },
98
- "rationale": {
99
- "type": "array",
100
- "minItems": 1,
101
- "items": { "type": "string", "minLength": 1 }
102
- },
103
- "evidence": {
104
- "type": "array",
105
- "minItems": 1,
106
- "items": {
107
- "type": "object",
108
- "additionalProperties": false,
109
- "required": ["path", "kind", "status", "summary"],
110
- "properties": {
111
- "path": { "type": "string", "minLength": 1 },
112
- "kind": { "type": "string", "minLength": 1 },
113
- "status": {
114
- "type": "string",
115
- "enum": ["verified", "partial", "self-reported", "conflicting", "missing"]
116
- },
117
- "summary": { "type": "string", "minLength": 1 }
118
- }
119
- }
120
- },
121
- "blockingFindings": {
122
- "type": "array",
123
- "items": { "type": "string", "minLength": 1 }
124
- },
125
- "requiredRevisions": {
126
- "type": "array",
127
- "items": { "type": "string", "minLength": 1 }
128
- },
129
- "riskFlags": {
130
- "type": "array",
131
- "items": { "type": "string", "minLength": 1 }
132
- },
133
- "humanEscalation": {
134
- "oneOf": [
135
- { "type": "null" },
136
- {
137
- "type": "object",
138
- "additionalProperties": false,
139
- "required": ["question", "recommendedOption", "options"],
140
- "properties": {
141
- "question": { "type": "string", "minLength": 1 },
142
- "recommendedOption": { "type": "string", "minLength": 1 },
143
- "options": {
144
- "type": "array",
145
- "minItems": 1,
146
- "items": {
147
- "type": "object",
148
- "additionalProperties": false,
149
- "required": ["id", "label", "risk", "reason"],
150
- "properties": {
151
- "id": { "type": "string", "minLength": 1 },
152
- "label": { "type": "string", "minLength": 1 },
153
- "risk": { "type": "string", "minLength": 1 },
154
- "reason": { "type": "string", "minLength": 1 }
155
- }
156
- }
157
- }
158
- }
159
- }
160
- ]
161
- },
162
- "audit": {
163
- "type": "object",
164
- "additionalProperties": true,
165
- "required": ["runId", "nodeId", "model"],
166
- "properties": {
167
- "runId": { "type": "string", "minLength": 1 },
168
- "nodeId": { "type": "string", "minLength": 1 },
169
- "model": { "type": "string", "minLength": 1 },
170
- "sourceHash": { "type": "string" }
171
- }
172
- }
173
- },
174
- "allOf": [
175
- {
176
- "if": {
177
- "properties": { "requiresHuman": { "const": true } },
178
- "required": ["requiresHuman"]
179
- },
180
- "then": {
181
- "properties": {
182
- "humanEscalation": { "type": "object" },
183
- "nextAction": { "const": "pause-and-ask" }
184
- },
185
- "required": ["humanEscalation"]
186
- }
187
- },
188
- {
189
- "if": {
190
- "properties": { "requiresHuman": { "const": false } },
191
- "required": ["requiresHuman"]
192
- },
193
- "then": {
194
- "properties": {
195
- "humanEscalation": { "type": "null" }
196
- }
197
- }
198
- },
199
- {
200
- "if": {
201
- "properties": {
202
- "decision": { "enum": ["escalate-to-human", "pause-wait-external"] }
203
- },
204
- "required": ["decision"]
205
- },
206
- "then": {
207
- "properties": {
208
- "requiresHuman": { "const": true }
209
- }
210
- }
211
- }
212
- ]
213
- }
1
+ {
2
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
3
+ "$id": "https://loop-agent.local/schemas/agent-dag-decision-envelope.schema.json",
4
+ "title": "Agent DAG Decision Envelope",
5
+ "description": "Structured output contract for advisory Agent DAG AI Secretary / Decision Gate nodes. M1 treats this as a documentation contract; M3 runtime parser extracts a ```DECISION_ENVELOPE_JSON fenced block and validates with zod plus semantic cross-field checks (recommendedOption must match options[].id; auto-approve must not set requiresHuman=true).",
6
+ "type": "object",
7
+ "additionalProperties": false,
8
+ "required": [
9
+ "schemaVersion",
10
+ "gateType",
11
+ "decisionScope",
12
+ "decision",
13
+ "confidence",
14
+ "riskLevel",
15
+ "requiresHuman",
16
+ "nextAction",
17
+ "policyVersion",
18
+ "policyChecks",
19
+ "rationale",
20
+ "evidence",
21
+ "blockingFindings",
22
+ "requiredRevisions",
23
+ "riskFlags",
24
+ "humanEscalation",
25
+ "audit"
26
+ ],
27
+ "properties": {
28
+ "schemaVersion": {
29
+ "type": "integer",
30
+ "const": 1
31
+ },
32
+ "gateType": {
33
+ "type": "string",
34
+ "enum": ["plan-gate", "side-effect-gate", "acceptance-gate", "closeout-gate"]
35
+ },
36
+ "decisionScope": {
37
+ "type": "string",
38
+ "enum": ["node", "rank", "dag-run", "task", "repo"]
39
+ },
40
+ "decision": {
41
+ "type": "string",
42
+ "enum": [
43
+ "auto-approve",
44
+ "approve-with-constraints",
45
+ "request-revision",
46
+ "run-more-verification",
47
+ "split-followup",
48
+ "reject",
49
+ "escalate-to-human",
50
+ "pause-wait-external"
51
+ ]
52
+ },
53
+ "confidence": {
54
+ "type": "number",
55
+ "minimum": 0,
56
+ "maximum": 1
57
+ },
58
+ "riskLevel": {
59
+ "type": "string",
60
+ "enum": ["low", "medium", "high", "critical"]
61
+ },
62
+ "requiresHuman": {
63
+ "type": "boolean"
64
+ },
65
+ "nextAction": {
66
+ "type": "string",
67
+ "enum": [
68
+ "continue",
69
+ "rerun-implement",
70
+ "rerun-verify",
71
+ "run-targeted-check",
72
+ "split-followup",
73
+ "pause-and-ask",
74
+ "abort"
75
+ ]
76
+ },
77
+ "policyVersion": {
78
+ "type": "string",
79
+ "minLength": 1
80
+ },
81
+ "policyChecks": {
82
+ "type": "object",
83
+ "additionalProperties": false,
84
+ "required": ["mustEscalateFlags", "evidenceComplete", "allowedAutoApprove"],
85
+ "properties": {
86
+ "mustEscalateFlags": {
87
+ "type": "array",
88
+ "items": { "type": "string", "minLength": 1 }
89
+ },
90
+ "evidenceComplete": {
91
+ "type": "boolean"
92
+ },
93
+ "allowedAutoApprove": {
94
+ "type": "boolean"
95
+ }
96
+ }
97
+ },
98
+ "rationale": {
99
+ "type": "array",
100
+ "minItems": 1,
101
+ "items": { "type": "string", "minLength": 1 }
102
+ },
103
+ "evidence": {
104
+ "type": "array",
105
+ "minItems": 1,
106
+ "items": {
107
+ "type": "object",
108
+ "additionalProperties": false,
109
+ "required": ["path", "kind", "status", "summary"],
110
+ "properties": {
111
+ "path": { "type": "string", "minLength": 1 },
112
+ "kind": { "type": "string", "minLength": 1 },
113
+ "status": {
114
+ "type": "string",
115
+ "enum": ["verified", "partial", "self-reported", "conflicting", "missing"]
116
+ },
117
+ "summary": { "type": "string", "minLength": 1 }
118
+ }
119
+ }
120
+ },
121
+ "blockingFindings": {
122
+ "type": "array",
123
+ "items": { "type": "string", "minLength": 1 }
124
+ },
125
+ "requiredRevisions": {
126
+ "type": "array",
127
+ "items": { "type": "string", "minLength": 1 }
128
+ },
129
+ "riskFlags": {
130
+ "type": "array",
131
+ "items": { "type": "string", "minLength": 1 }
132
+ },
133
+ "humanEscalation": {
134
+ "oneOf": [
135
+ { "type": "null" },
136
+ {
137
+ "type": "object",
138
+ "additionalProperties": false,
139
+ "required": ["question", "recommendedOption", "options"],
140
+ "properties": {
141
+ "question": { "type": "string", "minLength": 1 },
142
+ "recommendedOption": { "type": "string", "minLength": 1 },
143
+ "options": {
144
+ "type": "array",
145
+ "minItems": 1,
146
+ "items": {
147
+ "type": "object",
148
+ "additionalProperties": false,
149
+ "required": ["id", "label", "risk", "reason"],
150
+ "properties": {
151
+ "id": { "type": "string", "minLength": 1 },
152
+ "label": { "type": "string", "minLength": 1 },
153
+ "risk": { "type": "string", "minLength": 1 },
154
+ "reason": { "type": "string", "minLength": 1 }
155
+ }
156
+ }
157
+ }
158
+ }
159
+ }
160
+ ]
161
+ },
162
+ "audit": {
163
+ "type": "object",
164
+ "additionalProperties": true,
165
+ "required": ["runId", "nodeId", "model"],
166
+ "properties": {
167
+ "runId": { "type": "string", "minLength": 1 },
168
+ "nodeId": { "type": "string", "minLength": 1 },
169
+ "model": { "type": "string", "minLength": 1 },
170
+ "sourceHash": { "type": "string" }
171
+ }
172
+ }
173
+ },
174
+ "allOf": [
175
+ {
176
+ "if": {
177
+ "properties": { "requiresHuman": { "const": true } },
178
+ "required": ["requiresHuman"]
179
+ },
180
+ "then": {
181
+ "properties": {
182
+ "humanEscalation": { "type": "object" },
183
+ "nextAction": { "const": "pause-and-ask" }
184
+ },
185
+ "required": ["humanEscalation"]
186
+ }
187
+ },
188
+ {
189
+ "if": {
190
+ "properties": { "requiresHuman": { "const": false } },
191
+ "required": ["requiresHuman"]
192
+ },
193
+ "then": {
194
+ "properties": {
195
+ "humanEscalation": { "type": "null" }
196
+ }
197
+ }
198
+ },
199
+ {
200
+ "if": {
201
+ "properties": {
202
+ "decision": { "enum": ["escalate-to-human", "pause-wait-external"] }
203
+ },
204
+ "required": ["decision"]
205
+ },
206
+ "then": {
207
+ "properties": {
208
+ "requiresHuman": { "const": true }
209
+ }
210
+ }
211
+ }
212
+ ]
213
+ }
@@ -1,117 +1,117 @@
1
- # Agent DAG Decision Gate Dogfood Report Template
2
-
3
- > Copy to `docs/reports/YYYY-MM-DD-agent-dag-decision-gate-dogfood.md` after each dogfood run.
4
-
5
- ## Run Metadata
6
-
7
- | Field | Value |
8
- |---|---|
9
- | Date | YYYY-MM-DD |
10
- | Topic | |
11
- | DAG input | `<temp-dir>/<topic>-decision-gate-dag.json` |
12
- | Run ID | |
13
- | Run facts | `.harness/dag-runs/completed/<run-id>/` |
14
- | Gate type | `acceptance-gate` |
15
- | Decision node | `decision-pi` |
16
- | Model | `gpt-5.5` via `executorModels.pi.HIGH` |
17
-
18
- ## Work Type
19
-
20
- Choose one:
21
-
22
- - [ ] Low-risk docs/template change — expected `auto-approve` or `approve-with-constraints`
23
- - [ ] Runtime/loop-agent change — expected `request-revision`, `run-more-verification`, or `auto-approve`
24
- - [ ] Contract/security/high-risk simulation — expected `escalate-to-human`
25
-
26
- ## Deterministic Evidence
27
-
28
- | Evidence | Status | Notes |
29
- |---|---|---|
30
- | `verify-shell/result.summary.md` | verified / partial / missing | |
31
- | `state.json` | verified / partial / missing | |
32
- | git diff / changed file list | verified / partial / missing | |
33
- | progress/report/artifacts | verified / partial / missing | |
34
-
35
- ## Decision Envelope Summary
36
-
37
- Paste the **exact** ` ```DECISION_ENVELOPE_JSON ` fenced block from `decision-pi/result.summary.md` (info string must be `DECISION_ENVELOPE_JSON`, not `json`). Also record `decision`, `requiresHuman`, `nextAction`, and verified evidence count.
38
-
39
- ```DECISION_ENVELOPE_JSON
40
- {
41
- "schemaVersion": 1,
42
- "gateType": "acceptance-gate",
43
- "decisionScope": "dag-run",
44
- "decision": "auto-approve",
45
- "confidence": 0.88,
46
- "riskLevel": "low",
47
- "requiresHuman": false,
48
- "nextAction": "continue",
49
- "policyVersion": "agent-dag-decision-gate-v1",
50
- "policyChecks": {
51
- "mustEscalateFlags": [],
52
- "evidenceComplete": true,
53
- "allowedAutoApprove": true
54
- },
55
- "rationale": ["..."],
56
- "evidence": [
57
- {
58
- "path": ".harness/dag-runs/completed/<run-id>/verify-shell/result.summary.md",
59
- "kind": "shell-output",
60
- "status": "verified",
61
- "summary": "verification commands passed"
62
- }
63
- ],
64
- "blockingFindings": [],
65
- "requiredRevisions": [],
66
- "riskFlags": [],
67
- "humanEscalation": null,
68
- "audit": {
69
- "runId": "<run-id>",
70
- "nodeId": "decision-pi",
71
- "model": "gpt-5.5"
72
- }
73
- }
74
- ```
75
-
76
- ## Quality Metrics
77
-
78
- | Metric | Value | Notes |
79
- |---|---:|---|
80
- | `decisionLatencyMs` | | From node duration |
81
- | `tokenCostEstimate` | | If available |
82
- | `humanInterruptCount` | | Should be 0 for low-risk auto decisions |
83
- | `falseEscalation` | yes / no | Escalated when policy allowed auto decision |
84
- | `missedEscalation` | yes / no | Auto-approved when policy required human escalation |
85
- | `revisionCaughtBeforeHuman` | yes / no | Gate requested revision before human involvement |
86
- | `verifyPassAfterGate` | yes / no | Later verification passed after gate action |
87
- | `evidenceCompleteness` | complete / partial / missing | Based on verified evidence count |
88
-
89
- ## Evaluation
90
-
91
- ### What the gate got right
92
-
93
- -
94
-
95
- ### What the gate got wrong or over/under-weighted
96
-
97
- -
98
-
99
- ### Prompt / policy adjustments needed
100
-
101
- -
102
-
103
- ## Recommendation
104
-
105
- Choose one:
106
-
107
- - [ ] `proceed` — enough evidence to move toward parser/runtime work
108
- - [ ] `needs-more-dogfood` — continue M1/M2 advisory runs
109
- - [ ] `defer` — gate quality not yet good enough
110
-
111
- Rationale:
112
-
113
- -
114
-
115
- ## Follow-ups
116
-
117
- -
1
+ # Agent DAG Decision Gate Dogfood Report Template
2
+
3
+ > Copy to `docs/reports/YYYY-MM-DD-agent-dag-decision-gate-dogfood.md` after each dogfood run.
4
+
5
+ ## Run Metadata
6
+
7
+ | Field | Value |
8
+ |---|---|
9
+ | Date | YYYY-MM-DD |
10
+ | Topic | |
11
+ | DAG input | `<temp-dir>/<topic>-decision-gate-dag.json` |
12
+ | Run ID | |
13
+ | Run facts | `.harness/dag-runs/completed/<run-id>/` |
14
+ | Gate type | `acceptance-gate` |
15
+ | Decision node | `decision-pi` |
16
+ | Model | `gpt-5.5` via `executorModels.pi.HIGH` |
17
+
18
+ ## Work Type
19
+
20
+ Choose one:
21
+
22
+ - [ ] Low-risk docs/template change — expected `auto-approve` or `approve-with-constraints`
23
+ - [ ] Runtime/loop-agent change — expected `request-revision`, `run-more-verification`, or `auto-approve`
24
+ - [ ] Contract/security/high-risk simulation — expected `escalate-to-human`
25
+
26
+ ## Deterministic Evidence
27
+
28
+ | Evidence | Status | Notes |
29
+ |---|---|---|
30
+ | `verify-shell/result.summary.md` | verified / partial / missing | |
31
+ | `state.json` | verified / partial / missing | |
32
+ | git diff / changed file list | verified / partial / missing | |
33
+ | progress/report/artifacts | verified / partial / missing | |
34
+
35
+ ## Decision Envelope Summary
36
+
37
+ Paste the **exact** ` ```DECISION_ENVELOPE_JSON ` fenced block from `decision-pi/result.summary.md` (info string must be `DECISION_ENVELOPE_JSON`, not `json`). Also record `decision`, `requiresHuman`, `nextAction`, and verified evidence count.
38
+
39
+ ```DECISION_ENVELOPE_JSON
40
+ {
41
+ "schemaVersion": 1,
42
+ "gateType": "acceptance-gate",
43
+ "decisionScope": "dag-run",
44
+ "decision": "auto-approve",
45
+ "confidence": 0.88,
46
+ "riskLevel": "low",
47
+ "requiresHuman": false,
48
+ "nextAction": "continue",
49
+ "policyVersion": "agent-dag-decision-gate-v1",
50
+ "policyChecks": {
51
+ "mustEscalateFlags": [],
52
+ "evidenceComplete": true,
53
+ "allowedAutoApprove": true
54
+ },
55
+ "rationale": ["..."],
56
+ "evidence": [
57
+ {
58
+ "path": ".harness/dag-runs/completed/<run-id>/verify-shell/result.summary.md",
59
+ "kind": "shell-output",
60
+ "status": "verified",
61
+ "summary": "verification commands passed"
62
+ }
63
+ ],
64
+ "blockingFindings": [],
65
+ "requiredRevisions": [],
66
+ "riskFlags": [],
67
+ "humanEscalation": null,
68
+ "audit": {
69
+ "runId": "<run-id>",
70
+ "nodeId": "decision-pi",
71
+ "model": "gpt-5.5"
72
+ }
73
+ }
74
+ ```
75
+
76
+ ## Quality Metrics
77
+
78
+ | Metric | Value | Notes |
79
+ |---|---:|---|
80
+ | `decisionLatencyMs` | | From node duration |
81
+ | `tokenCostEstimate` | | If available |
82
+ | `humanInterruptCount` | | Should be 0 for low-risk auto decisions |
83
+ | `falseEscalation` | yes / no | Escalated when policy allowed auto decision |
84
+ | `missedEscalation` | yes / no | Auto-approved when policy required human escalation |
85
+ | `revisionCaughtBeforeHuman` | yes / no | Gate requested revision before human involvement |
86
+ | `verifyPassAfterGate` | yes / no | Later verification passed after gate action |
87
+ | `evidenceCompleteness` | complete / partial / missing | Based on verified evidence count |
88
+
89
+ ## Evaluation
90
+
91
+ ### What the gate got right
92
+
93
+ -
94
+
95
+ ### What the gate got wrong or over/under-weighted
96
+
97
+ -
98
+
99
+ ### Prompt / policy adjustments needed
100
+
101
+ -
102
+
103
+ ## Recommendation
104
+
105
+ Choose one:
106
+
107
+ - [ ] `proceed` — enough evidence to move toward parser/runtime work
108
+ - [ ] `needs-more-dogfood` — continue M1/M2 advisory runs
109
+ - [ ] `defer` — gate quality not yet good enough
110
+
111
+ Rationale:
112
+
113
+ -
114
+
115
+ ## Follow-ups
116
+
117
+ -