@tea-agent/loop-agent 0.13.0 → 0.15.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (272) hide show
  1. package/AGENTS.md +157 -157
  2. package/CHANGELOG.md +116 -305
  3. package/README.md +357 -334
  4. package/bin/agent-worker.js +22 -22
  5. package/bin/loop-agent.js +21 -21
  6. package/dist/commands/cursor-prompt.js +6 -6
  7. package/dist/commands/init.js +505 -505
  8. package/dist/commands/loop-benchmark.js +11 -11
  9. package/dist/commands/pi-reuse-benchmark.js +16 -16
  10. package/dist/executors/pi-event-serializer.js +33 -11
  11. package/dist/sidecars/cursor-prompt/executor.js +1 -1
  12. package/dist/task/runtime.js +27 -27
  13. package/dist/worker/observe/spec-evidence.js +19 -10
  14. package/dist/worker/observe/static/api.js +46 -46
  15. package/dist/worker/observe/static/app.js +151 -150
  16. package/dist/worker/observe/static/constants.js +156 -148
  17. package/dist/worker/observe/static/copy.js +67 -67
  18. package/dist/worker/observe/static/dag-helpers.js +201 -172
  19. package/dist/worker/observe/static/dag-layout.d.ts +31 -31
  20. package/dist/worker/observe/static/dag-layout.js +83 -83
  21. package/dist/worker/observe/static/dag-model.js +72 -72
  22. package/dist/worker/observe/static/dom.js +122 -122
  23. package/dist/worker/observe/static/format-pool.d.ts +71 -0
  24. package/dist/worker/observe/static/format-pool.js +134 -67
  25. package/dist/worker/observe/static/format.js +317 -292
  26. package/dist/worker/observe/static/index.html +350 -308
  27. package/dist/worker/observe/static/kpi.js +100 -94
  28. package/dist/worker/observe/static/markdown-render.js +124 -0
  29. package/dist/worker/observe/static/relations.js +133 -133
  30. package/dist/worker/observe/static/router.js +93 -93
  31. package/dist/worker/observe/static/run-processing.js +148 -148
  32. package/dist/worker/observe/static/shell-chrome.js +74 -68
  33. package/dist/worker/observe/static/state.js +273 -267
  34. package/dist/worker/observe/static/styles.css +2504 -1902
  35. package/dist/worker/observe/static/views/batch.js +227 -227
  36. package/dist/worker/observe/static/views/dag-graph.js +172 -172
  37. package/dist/worker/observe/static/views/dag-inspector.js +530 -627
  38. package/dist/worker/observe/static/views/dag.js +371 -371
  39. package/dist/worker/observe/static/views/dashboard.js +86 -100
  40. package/dist/worker/observe/static/views/failures.js +143 -143
  41. package/dist/worker/observe/static/views/feature.js +492 -492
  42. package/dist/worker/observe/static/views/pool.js +708 -350
  43. package/dist/worker/observe/static/views/run.js +453 -453
  44. package/dist/worker/observe/static/views/session-timeline.js +771 -219
  45. package/dist/worker/observe/static/views/shell.js +7 -7
  46. package/dist/worker/observe/static/views/task.js +314 -314
  47. package/dist/worker/observe/static/views/timeline.js +163 -163
  48. package/dist/workflows/dag/canvas-observer.js +275 -275
  49. package/dist/workflows/dag/init-hybrid.js +27 -11
  50. package/docs/README.md +106 -104
  51. package/docs/architecture/README.md +26 -26
  52. package/docs/architecture/dag-execution.md +140 -140
  53. package/docs/architecture/evolution.md +54 -54
  54. package/docs/architecture/facts-and-state.md +71 -71
  55. package/docs/architecture/runtime-boundaries.md +191 -191
  56. package/docs/architecture/system-overview.md +93 -93
  57. package/docs/architecture/worker-and-feature.md +85 -85
  58. package/docs/harness-methodology-debugging.md +153 -153
  59. package/docs/harness-methodology-tdd.md +130 -130
  60. package/docs/harness-methodology-verification.md +27 -27
  61. package/docs/init-surface.manifest.json +304 -307
  62. package/docs/skills/README.md +7 -7
  63. package/docs/skills/vetted-skill-registry.md +29 -29
  64. package/docs/templates/adr.md +60 -60
  65. package/docs/templates/agent-dag-authority-surface-audit.prompt.md +94 -94
  66. package/docs/templates/agent-dag-decision-envelope.schema.json +213 -213
  67. package/docs/templates/agent-dag-decision-gate-dogfood-report.md +117 -117
  68. package/docs/templates/agent-dag-decision-gate.prompt.md +246 -246
  69. package/docs/templates/agent-dag-process-supervisor.prompt.md +98 -98
  70. package/docs/templates/agent-dag-report.schema.json +473 -473
  71. package/docs/templates/agent-dag-review-verdict.prompt.md +68 -68
  72. package/docs/templates/agent-dag.base.json +190 -190
  73. package/docs/templates/agent-dag.final-verification.json +185 -185
  74. package/docs/templates/agent-dag.schema.json +411 -411
  75. package/docs/templates/agent-dag.supervised-implementation.json +620 -620
  76. package/docs/templates/backend-test-analysis.schema.json +44 -44
  77. package/docs/templates/backend-test-case-manifest.schema.json +190 -190
  78. package/docs/templates/backend-test-dag.classify.prompt.md +75 -75
  79. package/docs/templates/backend-test-dag.generate-pytest.prompt.md +204 -204
  80. package/docs/templates/backend-test-dag.json +559 -559
  81. package/docs/templates/backend-test-dag.retrospect.prompt.md +139 -139
  82. package/docs/templates/backend-test-dag.review-cases.prompt.md +83 -83
  83. package/docs/templates/backend-test-execution.schema.json +133 -133
  84. package/docs/templates/backend-test-result.schema.json +99 -99
  85. package/docs/templates/branch-merge-report.md +0 -1
  86. package/docs/templates/exec-plan.md +64 -64
  87. package/docs/templates/feature-spec.md +53 -53
  88. package/docs/templates/frontend-design-contract.md +42 -42
  89. package/docs/templates/frontend-eval/fixtures/failures/01-type-build-error.md +17 -17
  90. package/docs/templates/frontend-eval/fixtures/failures/02-unit-component-test-fail.md +16 -16
  91. package/docs/templates/frontend-eval/fixtures/failures/03-fixture-schema-drift.md +16 -16
  92. package/docs/templates/frontend-eval/fixtures/failures/04-missing-loading-empty-error-state.md +16 -16
  93. package/docs/templates/frontend-eval/fixtures/failures/05-forbidden-write-writeset-expansion.md +16 -16
  94. package/docs/templates/frontend-eval/fixtures/failures/06-unapproved-dependency-add.md +16 -16
  95. package/docs/templates/frontend-eval/fixtures/failures/07-mock-production-on.md +21 -21
  96. package/docs/templates/frontend-eval/fixtures/functional/01-simple-component-style.md +29 -29
  97. package/docs/templates/frontend-eval/fixtures/functional/02-form-validation.md +28 -28
  98. package/docs/templates/frontend-eval/fixtures/functional/03-list-detail-page.md +28 -28
  99. package/docs/templates/frontend-eval/fixtures/functional/04-api-mock.md +29 -29
  100. package/docs/templates/frontend-eval/fixtures/functional/05-permission-auth-gated-ui.md +27 -27
  101. package/docs/templates/frontend-eval/fixtures/functional/06-ssr-server-client-boundary.md +28 -28
  102. package/docs/templates/frontend-eval/fixtures/functional/07-shared-public-component-api.md +28 -28
  103. package/docs/templates/frontend-eval/fixtures/functional/08-pure-local-no-remote.md +27 -27
  104. package/docs/templates/frontend-eval/metrics.md +138 -138
  105. package/docs/templates/frontend-eval/smoke-targets.md +53 -53
  106. package/docs/templates/frontend-implementation-contract.schema.json +27 -27
  107. package/docs/templates/frontend-task-constraints.md +35 -35
  108. package/docs/templates/frontend-task-requirement.md +70 -70
  109. package/docs/templates/frontend-test-dag.generate-cases.prompt.md +5 -5
  110. package/docs/templates/frontend-test-dag.json +23 -23
  111. package/docs/templates/frontend-test-dag.retrieve-context.prompt.md +3 -3
  112. package/docs/templates/frontend-test-dag.retrospect.prompt.md +3 -3
  113. package/docs/templates/frontend-test-dag.review-cases.prompt.md +3 -3
  114. package/docs/templates/frontend-test-dag.review-execution.prompt.md +3 -3
  115. package/docs/templates/harness.schema.json +221 -221
  116. package/docs/templates/hybrid-dag.json +188 -188
  117. package/docs/templates/init-evolution-review.md +35 -35
  118. package/docs/templates/interactive-ui-round2-experiment.md +66 -66
  119. package/docs/templates/knowledge-graph-bootstrap-dag.json +118 -118
  120. package/docs/templates/knowledge-sync-dag.json +178 -178
  121. package/docs/templates/knowledge-sync-draft.schema.json +71 -71
  122. package/docs/templates/product-line/AGENTS.md +8 -8
  123. package/docs/templates/product-line/README.md +9 -9
  124. package/docs/templates/product-line/acceptance.yaml +14 -14
  125. package/docs/templates/product-line/closeout.yaml +9 -9
  126. package/docs/templates/product-line/design.md +13 -13
  127. package/docs/templates/product-line/links.md +10 -10
  128. package/docs/templates/product-line/requirement.md +17 -17
  129. package/docs/templates/product-line/task-graph.yaml +15 -15
  130. package/docs/templates/product-line/task.yaml +64 -64
  131. package/docs/templates/product-line/test-plan.md +7 -7
  132. package/docs/templates/production-readiness-checklist.md +57 -57
  133. package/docs/templates/progress-log.md +17 -17
  134. package/docs/templates/project-start-checklist.md +9 -9
  135. package/docs/templates/qa-report.md +48 -48
  136. package/docs/templates/sprint-contract.md +29 -29
  137. package/docs/templates/worker-dogfood-evidence.md +80 -80
  138. package/docs/templates/worker-dogfood-setup.md +68 -68
  139. package/examples/decision-gate-agent-dag.json +173 -173
  140. package/examples/example-dag.json +46 -46
  141. package/examples/hybrid-loop-agent-dag.json +188 -188
  142. package/harness.json +66 -66
  143. package/package.json +78 -52
  144. package/scripts/kb-bootstrap-init-skeleton.sh +240 -240
  145. package/scripts/kb-graph-incremental-prepare.mjs +386 -386
  146. package/scripts/kb-graph-materialize.mjs +105 -105
  147. package/scripts/kb-graph-promote.mjs +164 -164
  148. package/scripts/kb-query.mjs +554 -554
  149. package/skills/agent-worker/SKILL.md +39 -39
  150. package/skills/agent-worker/references/agent-worker-operator.md +60 -60
  151. package/skills/ai-engineering-context/SKILL.md +48 -48
  152. package/skills/analyze-product-dependencies/SKILL.md +67 -67
  153. package/skills/analyze-product-dependencies/agents/openai.yaml +4 -4
  154. package/skills/analyze-product-dependencies/references/api-documentation-schema.md +30 -30
  155. package/skills/analyze-product-dependencies/references/dependency-analysis-schema.md +28 -28
  156. package/skills/analyze-product-dependencies/references/example.md +76 -76
  157. package/skills/analyze-product-dependencies/references/forward-test-cases.md +35 -35
  158. package/skills/analyze-product-dependencies/references/input-contract.md +11 -11
  159. package/skills/analyze-product-dependencies/references/scouting-rules.md +61 -61
  160. package/skills/analyze-product-dependencies/scripts/test-validators.mjs +267 -267
  161. package/skills/analyze-product-dependencies/scripts/validate-api-documentation.mjs +101 -101
  162. package/skills/analyze-product-dependencies/scripts/validate-dependency-analysis.mjs +142 -142
  163. package/skills/analyze-product-dependencies/scripts/validate-product-requirement-input.mjs +76 -76
  164. package/skills/analyze-product-dependencies/scripts/validation-helpers.mjs +146 -146
  165. package/skills/analyze-product-requirements/SKILL.md +90 -90
  166. package/skills/analyze-product-requirements/agents/openai.yaml +4 -4
  167. package/skills/analyze-product-requirements/references/acceptance-criteria.md +91 -91
  168. package/skills/analyze-product-requirements/references/clarification-and-knowledge.md +56 -56
  169. package/skills/analyze-product-requirements/references/example.md +86 -86
  170. package/skills/analyze-product-requirements/references/forward-test-cases.md +66 -66
  171. package/skills/analyze-product-requirements/references/product-analysis-schema.md +32 -32
  172. package/skills/analyze-product-requirements/references/product-requirement-schema.md +33 -33
  173. package/skills/analyze-product-requirements/references/requirement-clarification-schema.md +35 -35
  174. package/skills/analyze-product-requirements/scripts/test-validators.mjs +193 -193
  175. package/skills/analyze-product-requirements/scripts/validate-product-analysis.mjs +69 -69
  176. package/skills/analyze-product-requirements/scripts/validate-product-requirement.mjs +97 -97
  177. package/skills/analyze-product-requirements/scripts/validate-requirement-clarification.mjs +98 -98
  178. package/skills/analyze-product-requirements/scripts/validation-helpers.mjs +156 -156
  179. package/skills/browser-tools/SKILL.md +196 -196
  180. package/skills/browser-tools/browser-content.js +103 -103
  181. package/skills/browser-tools/browser-cookies.js +35 -35
  182. package/skills/browser-tools/browser-eval.js +53 -53
  183. package/skills/browser-tools/browser-hn-scraper.js +108 -108
  184. package/skills/browser-tools/browser-nav.js +44 -44
  185. package/skills/browser-tools/browser-pick.js +162 -162
  186. package/skills/browser-tools/browser-screenshot.js +34 -34
  187. package/skills/browser-tools/browser-start.js +86 -86
  188. package/skills/browser-tools/package-lock.json +2556 -2556
  189. package/skills/browser-tools/package.json +19 -19
  190. package/skills/code-review-core/SKILL.md +20 -20
  191. package/skills/codebase-scout/SKILL.md +19 -19
  192. package/skills/frontend-design-review/SKILL.md +66 -66
  193. package/skills/frontend-design-review/references/review-checklist.md +40 -58
  194. package/skills/frontend-implementation/SKILL.md +49 -49
  195. package/skills/frontend-implementation/references/code-standards.md +32 -32
  196. package/skills/frontend-implementation/references/design-spec.md +46 -46
  197. package/skills/frontend-implementation/references/node-contracts.md +27 -27
  198. package/skills/frontend-review/SKILL.md +61 -59
  199. package/skills/frontend-review/references/review-findings.md +48 -47
  200. package/skills/frontend-verification/SKILL.md +55 -53
  201. package/skills/frontend-verification/references/verification-checklist.md +59 -68
  202. package/skills/grill-me/SKILL.md +10 -10
  203. package/skills/grill-with-docs/SKILL.md +88 -88
  204. package/skills/grill-with-docs/adr-format.md +47 -47
  205. package/skills/grill-with-docs/context-format.md +60 -60
  206. package/skills/init-capability-evolution/SKILL.md +70 -70
  207. package/skills/loop-agent/SKILL.md +151 -151
  208. package/skills/loop-agent/references/README.md +67 -67
  209. package/skills/loop-agent/references/command-reference.md +527 -527
  210. package/skills/loop-agent/references/docs-converge.md +126 -126
  211. package/skills/loop-agent/references/harness-policy.md +263 -263
  212. package/skills/loop-agent/references/hybrid-dag.md +243 -243
  213. package/skills/loop-agent/references/learned/README.md +21 -21
  214. package/skills/loop-agent/references/long-running-loop.md +57 -57
  215. package/skills/loop-agent/references/model-routing.md +36 -36
  216. package/skills/loop-agent/references/multi-worktree.md +54 -54
  217. package/skills/loop-agent/references/one-shot-runs.md +85 -85
  218. package/skills/loop-agent/references/orchestrator-and-interventions.md +169 -169
  219. package/skills/loop-agent/references/pi-prompt.md +23 -23
  220. package/skills/loop-agent/references/pi-subagent-assisted-mode.md +84 -84
  221. package/skills/loop-agent/references/post-implementation-and-patterns.md +44 -44
  222. package/skills/loop-agent/references/task-workflow.md +89 -89
  223. package/skills/loop-agent/references/verification-and-failure-handling.md +141 -141
  224. package/skills/playwright-cli/SKILL.md +420 -420
  225. package/skills/playwright-cli/references/element-attributes.md +23 -23
  226. package/skills/playwright-cli/references/playwright-tests.md +39 -39
  227. package/skills/playwright-cli/references/request-mocking.md +87 -87
  228. package/skills/playwright-cli/references/running-code.md +241 -241
  229. package/skills/playwright-cli/references/session-management.md +225 -225
  230. package/skills/playwright-cli/references/storage-state.md +275 -275
  231. package/skills/playwright-cli/references/test-generation.md +433 -433
  232. package/skills/playwright-cli/references/tracing.md +139 -139
  233. package/skills/playwright-cli/references/video-recording.md +143 -143
  234. package/skills/playwright-cli-case-generator/SKILL.md +74 -74
  235. package/skills/requesting-code-review/SKILL.md +101 -101
  236. package/skills/requesting-code-review/code-reviewer.md +168 -168
  237. package/skills/systematic-debugging/CREATION-LOG.md +119 -119
  238. package/skills/systematic-debugging/SKILL.md +296 -296
  239. package/skills/systematic-debugging/condition-based-waiting-example.ts +158 -158
  240. package/skills/systematic-debugging/condition-based-waiting.md +115 -115
  241. package/skills/systematic-debugging/defense-in-depth.md +122 -122
  242. package/skills/systematic-debugging/find-polluter.sh +63 -63
  243. package/skills/systematic-debugging/root-cause-tracing.md +169 -169
  244. package/skills/systematic-debugging/test-academic.md +14 -14
  245. package/skills/systematic-debugging/test-pressure-1.md +58 -58
  246. package/skills/systematic-debugging/test-pressure-2.md +68 -68
  247. package/skills/systematic-debugging/test-pressure-3.md +69 -69
  248. package/skills/test-driven-development/SKILL.md +20 -20
  249. package/skills/using-git-worktrees/SKILL.md +215 -215
  250. package/skills/verification-before-completion/SKILL.md +154 -154
  251. package/skills/webapp-testing/SKILL.md +19 -19
  252. package/docs/agent-dag-recovery-playbook.md +0 -195
  253. package/docs/agent-dag-runner.md +0 -67
  254. package/docs/cursor-prompt-sidecar.md +0 -36
  255. package/docs/decisions/README.md +0 -18
  256. package/docs/design/README.md +0 -167
  257. package/docs/development-principles.md +0 -73
  258. package/docs/exec-plans/README.md +0 -6
  259. package/docs/exec-plans/active/README.md +0 -12
  260. package/docs/exec-plans/completed/README.md +0 -107
  261. package/docs/feature-workflow.md +0 -414
  262. package/docs/loop-agent-harness.md +0 -142
  263. package/docs/production-readiness.md +0 -96
  264. package/docs/progress/README.md +0 -80
  265. package/docs/reports/README.md +0 -159
  266. package/docs/verification-matrix.md +0 -70
  267. package/scripts/check-product-line-docs.sh +0 -29
  268. package/scripts/check-task-pool-root.sh +0 -32
  269. package/scripts/kb-graph-incremental-prepare.sh +0 -5
  270. package/scripts/kb-graph-materialize.sh +0 -4
  271. package/scripts/kb-graph-promote.sh +0 -4
  272. package/scripts/kb-query.sh +0 -5
@@ -1,213 +1,213 @@
1
- {
2
- "$schema": "https://json-schema.org/draft/2020-12/schema",
3
- "$id": "https://loop-agent.local/schemas/agent-dag-decision-envelope.schema.json",
4
- "title": "Agent DAG Decision Envelope",
5
- "description": "Structured output contract for advisory Agent DAG AI Secretary / Decision Gate nodes. M1 treats this as a documentation contract; M3 runtime parser extracts a ```DECISION_ENVELOPE_JSON fenced block and validates with zod plus semantic cross-field checks (recommendedOption must match options[].id; auto-approve must not set requiresHuman=true).",
6
- "type": "object",
7
- "additionalProperties": false,
8
- "required": [
9
- "schemaVersion",
10
- "gateType",
11
- "decisionScope",
12
- "decision",
13
- "confidence",
14
- "riskLevel",
15
- "requiresHuman",
16
- "nextAction",
17
- "policyVersion",
18
- "policyChecks",
19
- "rationale",
20
- "evidence",
21
- "blockingFindings",
22
- "requiredRevisions",
23
- "riskFlags",
24
- "humanEscalation",
25
- "audit"
26
- ],
27
- "properties": {
28
- "schemaVersion": {
29
- "type": "integer",
30
- "const": 1
31
- },
32
- "gateType": {
33
- "type": "string",
34
- "enum": ["plan-gate", "side-effect-gate", "acceptance-gate", "closeout-gate"]
35
- },
36
- "decisionScope": {
37
- "type": "string",
38
- "enum": ["node", "rank", "dag-run", "task", "repo"]
39
- },
40
- "decision": {
41
- "type": "string",
42
- "enum": [
43
- "auto-approve",
44
- "approve-with-constraints",
45
- "request-revision",
46
- "run-more-verification",
47
- "split-followup",
48
- "reject",
49
- "escalate-to-human",
50
- "pause-wait-external"
51
- ]
52
- },
53
- "confidence": {
54
- "type": "number",
55
- "minimum": 0,
56
- "maximum": 1
57
- },
58
- "riskLevel": {
59
- "type": "string",
60
- "enum": ["low", "medium", "high", "critical"]
61
- },
62
- "requiresHuman": {
63
- "type": "boolean"
64
- },
65
- "nextAction": {
66
- "type": "string",
67
- "enum": [
68
- "continue",
69
- "rerun-implement",
70
- "rerun-verify",
71
- "run-targeted-check",
72
- "split-followup",
73
- "pause-and-ask",
74
- "abort"
75
- ]
76
- },
77
- "policyVersion": {
78
- "type": "string",
79
- "minLength": 1
80
- },
81
- "policyChecks": {
82
- "type": "object",
83
- "additionalProperties": false,
84
- "required": ["mustEscalateFlags", "evidenceComplete", "allowedAutoApprove"],
85
- "properties": {
86
- "mustEscalateFlags": {
87
- "type": "array",
88
- "items": { "type": "string", "minLength": 1 }
89
- },
90
- "evidenceComplete": {
91
- "type": "boolean"
92
- },
93
- "allowedAutoApprove": {
94
- "type": "boolean"
95
- }
96
- }
97
- },
98
- "rationale": {
99
- "type": "array",
100
- "minItems": 1,
101
- "items": { "type": "string", "minLength": 1 }
102
- },
103
- "evidence": {
104
- "type": "array",
105
- "minItems": 1,
106
- "items": {
107
- "type": "object",
108
- "additionalProperties": false,
109
- "required": ["path", "kind", "status", "summary"],
110
- "properties": {
111
- "path": { "type": "string", "minLength": 1 },
112
- "kind": { "type": "string", "minLength": 1 },
113
- "status": {
114
- "type": "string",
115
- "enum": ["verified", "partial", "self-reported", "conflicting", "missing"]
116
- },
117
- "summary": { "type": "string", "minLength": 1 }
118
- }
119
- }
120
- },
121
- "blockingFindings": {
122
- "type": "array",
123
- "items": { "type": "string", "minLength": 1 }
124
- },
125
- "requiredRevisions": {
126
- "type": "array",
127
- "items": { "type": "string", "minLength": 1 }
128
- },
129
- "riskFlags": {
130
- "type": "array",
131
- "items": { "type": "string", "minLength": 1 }
132
- },
133
- "humanEscalation": {
134
- "oneOf": [
135
- { "type": "null" },
136
- {
137
- "type": "object",
138
- "additionalProperties": false,
139
- "required": ["question", "recommendedOption", "options"],
140
- "properties": {
141
- "question": { "type": "string", "minLength": 1 },
142
- "recommendedOption": { "type": "string", "minLength": 1 },
143
- "options": {
144
- "type": "array",
145
- "minItems": 1,
146
- "items": {
147
- "type": "object",
148
- "additionalProperties": false,
149
- "required": ["id", "label", "risk", "reason"],
150
- "properties": {
151
- "id": { "type": "string", "minLength": 1 },
152
- "label": { "type": "string", "minLength": 1 },
153
- "risk": { "type": "string", "minLength": 1 },
154
- "reason": { "type": "string", "minLength": 1 }
155
- }
156
- }
157
- }
158
- }
159
- }
160
- ]
161
- },
162
- "audit": {
163
- "type": "object",
164
- "additionalProperties": true,
165
- "required": ["runId", "nodeId", "model"],
166
- "properties": {
167
- "runId": { "type": "string", "minLength": 1 },
168
- "nodeId": { "type": "string", "minLength": 1 },
169
- "model": { "type": "string", "minLength": 1 },
170
- "sourceHash": { "type": "string" }
171
- }
172
- }
173
- },
174
- "allOf": [
175
- {
176
- "if": {
177
- "properties": { "requiresHuman": { "const": true } },
178
- "required": ["requiresHuman"]
179
- },
180
- "then": {
181
- "properties": {
182
- "humanEscalation": { "type": "object" },
183
- "nextAction": { "const": "pause-and-ask" }
184
- },
185
- "required": ["humanEscalation"]
186
- }
187
- },
188
- {
189
- "if": {
190
- "properties": { "requiresHuman": { "const": false } },
191
- "required": ["requiresHuman"]
192
- },
193
- "then": {
194
- "properties": {
195
- "humanEscalation": { "type": "null" }
196
- }
197
- }
198
- },
199
- {
200
- "if": {
201
- "properties": {
202
- "decision": { "enum": ["escalate-to-human", "pause-wait-external"] }
203
- },
204
- "required": ["decision"]
205
- },
206
- "then": {
207
- "properties": {
208
- "requiresHuman": { "const": true }
209
- }
210
- }
211
- }
212
- ]
213
- }
1
+ {
2
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
3
+ "$id": "https://loop-agent.local/schemas/agent-dag-decision-envelope.schema.json",
4
+ "title": "Agent DAG Decision Envelope",
5
+ "description": "Structured output contract for advisory Agent DAG AI Secretary / Decision Gate nodes. M1 treats this as a documentation contract; M3 runtime parser extracts a ```DECISION_ENVELOPE_JSON fenced block and validates with zod plus semantic cross-field checks (recommendedOption must match options[].id; auto-approve must not set requiresHuman=true).",
6
+ "type": "object",
7
+ "additionalProperties": false,
8
+ "required": [
9
+ "schemaVersion",
10
+ "gateType",
11
+ "decisionScope",
12
+ "decision",
13
+ "confidence",
14
+ "riskLevel",
15
+ "requiresHuman",
16
+ "nextAction",
17
+ "policyVersion",
18
+ "policyChecks",
19
+ "rationale",
20
+ "evidence",
21
+ "blockingFindings",
22
+ "requiredRevisions",
23
+ "riskFlags",
24
+ "humanEscalation",
25
+ "audit"
26
+ ],
27
+ "properties": {
28
+ "schemaVersion": {
29
+ "type": "integer",
30
+ "const": 1
31
+ },
32
+ "gateType": {
33
+ "type": "string",
34
+ "enum": ["plan-gate", "side-effect-gate", "acceptance-gate", "closeout-gate"]
35
+ },
36
+ "decisionScope": {
37
+ "type": "string",
38
+ "enum": ["node", "rank", "dag-run", "task", "repo"]
39
+ },
40
+ "decision": {
41
+ "type": "string",
42
+ "enum": [
43
+ "auto-approve",
44
+ "approve-with-constraints",
45
+ "request-revision",
46
+ "run-more-verification",
47
+ "split-followup",
48
+ "reject",
49
+ "escalate-to-human",
50
+ "pause-wait-external"
51
+ ]
52
+ },
53
+ "confidence": {
54
+ "type": "number",
55
+ "minimum": 0,
56
+ "maximum": 1
57
+ },
58
+ "riskLevel": {
59
+ "type": "string",
60
+ "enum": ["low", "medium", "high", "critical"]
61
+ },
62
+ "requiresHuman": {
63
+ "type": "boolean"
64
+ },
65
+ "nextAction": {
66
+ "type": "string",
67
+ "enum": [
68
+ "continue",
69
+ "rerun-implement",
70
+ "rerun-verify",
71
+ "run-targeted-check",
72
+ "split-followup",
73
+ "pause-and-ask",
74
+ "abort"
75
+ ]
76
+ },
77
+ "policyVersion": {
78
+ "type": "string",
79
+ "minLength": 1
80
+ },
81
+ "policyChecks": {
82
+ "type": "object",
83
+ "additionalProperties": false,
84
+ "required": ["mustEscalateFlags", "evidenceComplete", "allowedAutoApprove"],
85
+ "properties": {
86
+ "mustEscalateFlags": {
87
+ "type": "array",
88
+ "items": { "type": "string", "minLength": 1 }
89
+ },
90
+ "evidenceComplete": {
91
+ "type": "boolean"
92
+ },
93
+ "allowedAutoApprove": {
94
+ "type": "boolean"
95
+ }
96
+ }
97
+ },
98
+ "rationale": {
99
+ "type": "array",
100
+ "minItems": 1,
101
+ "items": { "type": "string", "minLength": 1 }
102
+ },
103
+ "evidence": {
104
+ "type": "array",
105
+ "minItems": 1,
106
+ "items": {
107
+ "type": "object",
108
+ "additionalProperties": false,
109
+ "required": ["path", "kind", "status", "summary"],
110
+ "properties": {
111
+ "path": { "type": "string", "minLength": 1 },
112
+ "kind": { "type": "string", "minLength": 1 },
113
+ "status": {
114
+ "type": "string",
115
+ "enum": ["verified", "partial", "self-reported", "conflicting", "missing"]
116
+ },
117
+ "summary": { "type": "string", "minLength": 1 }
118
+ }
119
+ }
120
+ },
121
+ "blockingFindings": {
122
+ "type": "array",
123
+ "items": { "type": "string", "minLength": 1 }
124
+ },
125
+ "requiredRevisions": {
126
+ "type": "array",
127
+ "items": { "type": "string", "minLength": 1 }
128
+ },
129
+ "riskFlags": {
130
+ "type": "array",
131
+ "items": { "type": "string", "minLength": 1 }
132
+ },
133
+ "humanEscalation": {
134
+ "oneOf": [
135
+ { "type": "null" },
136
+ {
137
+ "type": "object",
138
+ "additionalProperties": false,
139
+ "required": ["question", "recommendedOption", "options"],
140
+ "properties": {
141
+ "question": { "type": "string", "minLength": 1 },
142
+ "recommendedOption": { "type": "string", "minLength": 1 },
143
+ "options": {
144
+ "type": "array",
145
+ "minItems": 1,
146
+ "items": {
147
+ "type": "object",
148
+ "additionalProperties": false,
149
+ "required": ["id", "label", "risk", "reason"],
150
+ "properties": {
151
+ "id": { "type": "string", "minLength": 1 },
152
+ "label": { "type": "string", "minLength": 1 },
153
+ "risk": { "type": "string", "minLength": 1 },
154
+ "reason": { "type": "string", "minLength": 1 }
155
+ }
156
+ }
157
+ }
158
+ }
159
+ }
160
+ ]
161
+ },
162
+ "audit": {
163
+ "type": "object",
164
+ "additionalProperties": true,
165
+ "required": ["runId", "nodeId", "model"],
166
+ "properties": {
167
+ "runId": { "type": "string", "minLength": 1 },
168
+ "nodeId": { "type": "string", "minLength": 1 },
169
+ "model": { "type": "string", "minLength": 1 },
170
+ "sourceHash": { "type": "string" }
171
+ }
172
+ }
173
+ },
174
+ "allOf": [
175
+ {
176
+ "if": {
177
+ "properties": { "requiresHuman": { "const": true } },
178
+ "required": ["requiresHuman"]
179
+ },
180
+ "then": {
181
+ "properties": {
182
+ "humanEscalation": { "type": "object" },
183
+ "nextAction": { "const": "pause-and-ask" }
184
+ },
185
+ "required": ["humanEscalation"]
186
+ }
187
+ },
188
+ {
189
+ "if": {
190
+ "properties": { "requiresHuman": { "const": false } },
191
+ "required": ["requiresHuman"]
192
+ },
193
+ "then": {
194
+ "properties": {
195
+ "humanEscalation": { "type": "null" }
196
+ }
197
+ }
198
+ },
199
+ {
200
+ "if": {
201
+ "properties": {
202
+ "decision": { "enum": ["escalate-to-human", "pause-wait-external"] }
203
+ },
204
+ "required": ["decision"]
205
+ },
206
+ "then": {
207
+ "properties": {
208
+ "requiresHuman": { "const": true }
209
+ }
210
+ }
211
+ }
212
+ ]
213
+ }
@@ -1,117 +1,117 @@
1
- # Agent DAG Decision Gate Dogfood Report Template
2
-
3
- > Copy to `docs/reports/YYYY-MM-DD-agent-dag-decision-gate-dogfood.md` after each dogfood run.
4
-
5
- ## Run Metadata
6
-
7
- | Field | Value |
8
- |---|---|
9
- | Date | YYYY-MM-DD |
10
- | Topic | |
11
- | DAG input | `<temp-dir>/<topic>-decision-gate-dag.json` |
12
- | Run ID | |
13
- | Run facts | `.harness/dag-runs/completed/<run-id>/` |
14
- | Gate type | `acceptance-gate` |
15
- | Decision node | `decision-pi` |
16
- | Model | `gpt-5.5` via `executorModels.pi.HIGH` |
17
-
18
- ## Work Type
19
-
20
- Choose one:
21
-
22
- - [ ] Low-risk docs/template change — expected `auto-approve` or `approve-with-constraints`
23
- - [ ] Runtime/loop-agent change — expected `request-revision`, `run-more-verification`, or `auto-approve`
24
- - [ ] Contract/security/high-risk simulation — expected `escalate-to-human`
25
-
26
- ## Deterministic Evidence
27
-
28
- | Evidence | Status | Notes |
29
- |---|---|---|
30
- | `verify-shell/result.summary.md` | verified / partial / missing | |
31
- | `state.json` | verified / partial / missing | |
32
- | git diff / changed file list | verified / partial / missing | |
33
- | progress/report/artifacts | verified / partial / missing | |
34
-
35
- ## Decision Envelope Summary
36
-
37
- Paste the **exact** ` ```DECISION_ENVELOPE_JSON ` fenced block from `decision-pi/result.summary.md` (info string must be `DECISION_ENVELOPE_JSON`, not `json`). Also record `decision`, `requiresHuman`, `nextAction`, and verified evidence count.
38
-
39
- ```DECISION_ENVELOPE_JSON
40
- {
41
- "schemaVersion": 1,
42
- "gateType": "acceptance-gate",
43
- "decisionScope": "dag-run",
44
- "decision": "auto-approve",
45
- "confidence": 0.88,
46
- "riskLevel": "low",
47
- "requiresHuman": false,
48
- "nextAction": "continue",
49
- "policyVersion": "agent-dag-decision-gate-v1",
50
- "policyChecks": {
51
- "mustEscalateFlags": [],
52
- "evidenceComplete": true,
53
- "allowedAutoApprove": true
54
- },
55
- "rationale": ["..."],
56
- "evidence": [
57
- {
58
- "path": ".harness/dag-runs/completed/<run-id>/verify-shell/result.summary.md",
59
- "kind": "shell-output",
60
- "status": "verified",
61
- "summary": "verification commands passed"
62
- }
63
- ],
64
- "blockingFindings": [],
65
- "requiredRevisions": [],
66
- "riskFlags": [],
67
- "humanEscalation": null,
68
- "audit": {
69
- "runId": "<run-id>",
70
- "nodeId": "decision-pi",
71
- "model": "gpt-5.5"
72
- }
73
- }
74
- ```
75
-
76
- ## Quality Metrics
77
-
78
- | Metric | Value | Notes |
79
- |---|---:|---|
80
- | `decisionLatencyMs` | | From node duration |
81
- | `tokenCostEstimate` | | If available |
82
- | `humanInterruptCount` | | Should be 0 for low-risk auto decisions |
83
- | `falseEscalation` | yes / no | Escalated when policy allowed auto decision |
84
- | `missedEscalation` | yes / no | Auto-approved when policy required human escalation |
85
- | `revisionCaughtBeforeHuman` | yes / no | Gate requested revision before human involvement |
86
- | `verifyPassAfterGate` | yes / no | Later verification passed after gate action |
87
- | `evidenceCompleteness` | complete / partial / missing | Based on verified evidence count |
88
-
89
- ## Evaluation
90
-
91
- ### What the gate got right
92
-
93
- -
94
-
95
- ### What the gate got wrong or over/under-weighted
96
-
97
- -
98
-
99
- ### Prompt / policy adjustments needed
100
-
101
- -
102
-
103
- ## Recommendation
104
-
105
- Choose one:
106
-
107
- - [ ] `proceed` — enough evidence to move toward parser/runtime work
108
- - [ ] `needs-more-dogfood` — continue M1/M2 advisory runs
109
- - [ ] `defer` — gate quality not yet good enough
110
-
111
- Rationale:
112
-
113
- -
114
-
115
- ## Follow-ups
116
-
117
- -
1
+ # Agent DAG Decision Gate Dogfood Report Template
2
+
3
+ > Copy to `docs/reports/YYYY-MM-DD-agent-dag-decision-gate-dogfood.md` after each dogfood run.
4
+
5
+ ## Run Metadata
6
+
7
+ | Field | Value |
8
+ |---|---|
9
+ | Date | YYYY-MM-DD |
10
+ | Topic | |
11
+ | DAG input | `<temp-dir>/<topic>-decision-gate-dag.json` |
12
+ | Run ID | |
13
+ | Run facts | `.harness/dag-runs/completed/<run-id>/` |
14
+ | Gate type | `acceptance-gate` |
15
+ | Decision node | `decision-pi` |
16
+ | Model | `gpt-5.5` via `executorModels.pi.HIGH` |
17
+
18
+ ## Work Type
19
+
20
+ Choose one:
21
+
22
+ - [ ] Low-risk docs/template change — expected `auto-approve` or `approve-with-constraints`
23
+ - [ ] Runtime/loop-agent change — expected `request-revision`, `run-more-verification`, or `auto-approve`
24
+ - [ ] Contract/security/high-risk simulation — expected `escalate-to-human`
25
+
26
+ ## Deterministic Evidence
27
+
28
+ | Evidence | Status | Notes |
29
+ |---|---|---|
30
+ | `verify-shell/result.summary.md` | verified / partial / missing | |
31
+ | `state.json` | verified / partial / missing | |
32
+ | git diff / changed file list | verified / partial / missing | |
33
+ | progress/report/artifacts | verified / partial / missing | |
34
+
35
+ ## Decision Envelope Summary
36
+
37
+ Paste the **exact** ` ```DECISION_ENVELOPE_JSON ` fenced block from `decision-pi/result.summary.md` (info string must be `DECISION_ENVELOPE_JSON`, not `json`). Also record `decision`, `requiresHuman`, `nextAction`, and verified evidence count.
38
+
39
+ ```DECISION_ENVELOPE_JSON
40
+ {
41
+ "schemaVersion": 1,
42
+ "gateType": "acceptance-gate",
43
+ "decisionScope": "dag-run",
44
+ "decision": "auto-approve",
45
+ "confidence": 0.88,
46
+ "riskLevel": "low",
47
+ "requiresHuman": false,
48
+ "nextAction": "continue",
49
+ "policyVersion": "agent-dag-decision-gate-v1",
50
+ "policyChecks": {
51
+ "mustEscalateFlags": [],
52
+ "evidenceComplete": true,
53
+ "allowedAutoApprove": true
54
+ },
55
+ "rationale": ["..."],
56
+ "evidence": [
57
+ {
58
+ "path": ".harness/dag-runs/completed/<run-id>/verify-shell/result.summary.md",
59
+ "kind": "shell-output",
60
+ "status": "verified",
61
+ "summary": "verification commands passed"
62
+ }
63
+ ],
64
+ "blockingFindings": [],
65
+ "requiredRevisions": [],
66
+ "riskFlags": [],
67
+ "humanEscalation": null,
68
+ "audit": {
69
+ "runId": "<run-id>",
70
+ "nodeId": "decision-pi",
71
+ "model": "gpt-5.5"
72
+ }
73
+ }
74
+ ```
75
+
76
+ ## Quality Metrics
77
+
78
+ | Metric | Value | Notes |
79
+ |---|---:|---|
80
+ | `decisionLatencyMs` | | From node duration |
81
+ | `tokenCostEstimate` | | If available |
82
+ | `humanInterruptCount` | | Should be 0 for low-risk auto decisions |
83
+ | `falseEscalation` | yes / no | Escalated when policy allowed auto decision |
84
+ | `missedEscalation` | yes / no | Auto-approved when policy required human escalation |
85
+ | `revisionCaughtBeforeHuman` | yes / no | Gate requested revision before human involvement |
86
+ | `verifyPassAfterGate` | yes / no | Later verification passed after gate action |
87
+ | `evidenceCompleteness` | complete / partial / missing | Based on verified evidence count |
88
+
89
+ ## Evaluation
90
+
91
+ ### What the gate got right
92
+
93
+ -
94
+
95
+ ### What the gate got wrong or over/under-weighted
96
+
97
+ -
98
+
99
+ ### Prompt / policy adjustments needed
100
+
101
+ -
102
+
103
+ ## Recommendation
104
+
105
+ Choose one:
106
+
107
+ - [ ] `proceed` — enough evidence to move toward parser/runtime work
108
+ - [ ] `needs-more-dogfood` — continue M1/M2 advisory runs
109
+ - [ ] `defer` — gate quality not yet good enough
110
+
111
+ Rationale:
112
+
113
+ -
114
+
115
+ ## Follow-ups
116
+
117
+ -