@tea-agent/loop-agent 0.26.0 → 0.26.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (197) hide show
  1. package/CHANGELOG.md +1056 -1020
  2. package/README.md +8 -3
  3. package/bin/loop-agent.js +21 -21
  4. package/dist/application/dag/generate-task-dag.js +33 -0
  5. package/dist/cli/command-definitions.js +25 -10
  6. package/dist/cli/help.js +4 -3
  7. package/dist/cli/program.js +43 -17
  8. package/dist/commands/cursor-prompt.js +6 -6
  9. package/dist/commands/import-prd.js +7 -2
  10. package/dist/commands/init.js +7 -5
  11. package/dist/commands/loop-benchmark.js +11 -11
  12. package/dist/commands/pi-reuse-benchmark.js +16 -16
  13. package/dist/commands/task-source-prepare.js +474 -0
  14. package/dist/executors/dag-pi-executor.js +40 -5
  15. package/dist/executors/shell-executor.js +111 -0
  16. package/dist/executors/shell-presets.js +12 -4
  17. package/dist/executors/shell-write-guard.js +161 -25
  18. package/dist/sidecars/cursor-prompt/executor.js +1 -1
  19. package/dist/task/config-types.js +6 -0
  20. package/dist/task/contract/constants.js +1 -0
  21. package/dist/task/contract/project.js +8 -0
  22. package/dist/task/contract/schema.js +1 -0
  23. package/dist/task/frontend-preflight.js +131 -0
  24. package/dist/task/runtime.js +2 -4
  25. package/dist/task/source-prepare/build-draft.js +224 -0
  26. package/dist/task/source-prepare/completeness.js +195 -0
  27. package/dist/task/source-prepare/index.js +7 -0
  28. package/dist/task/source-prepare/parse-intent.js +373 -0
  29. package/dist/task/source-prepare/path-policy.js +197 -0
  30. package/dist/task/source-prepare/prepare.js +506 -0
  31. package/dist/task/source-prepare/reference-integrity.js +274 -0
  32. package/dist/task/source-prepare/types.js +7 -0
  33. package/dist/worker/observability/read-model.js +134 -0
  34. package/dist/worker/observe/static/copy.js +67 -67
  35. package/dist/worker/observe/static/dag-layout.d.ts +31 -31
  36. package/dist/worker/observe/static/dag-layout.js +83 -83
  37. package/dist/worker/observe/static/dom.js +220 -220
  38. package/dist/worker/observe/static/relations.js +133 -133
  39. package/dist/worker/observe/static/router.js +93 -93
  40. package/dist/worker/observe/static/run-processing.js +148 -148
  41. package/dist/worker/observe/static/state.js +61 -0
  42. package/dist/worker/observe/static/styles.css +8 -0
  43. package/dist/worker/observe/static/views/batch.js +227 -227
  44. package/dist/worker/observe/static/views/dag-graph.js +248 -172
  45. package/dist/worker/observe/static/views/dag-inspector.js +374 -157
  46. package/dist/worker/observe/static/views/dag.js +4 -11
  47. package/dist/worker/observe/static/views/failures.js +143 -143
  48. package/dist/worker/observe/static/views/feature.js +492 -492
  49. package/dist/worker/observe/static/views/run.js +453 -453
  50. package/dist/worker/observe/static/views/shell.js +7 -7
  51. package/dist/worker/observe/static/views/timeline.js +163 -163
  52. package/dist/workflows/dag/backend-test-pytest-collection.js +277 -0
  53. package/dist/workflows/dag/canvas-observer.js +275 -275
  54. package/dist/workflows/dag/convergence/controller.js +110 -21
  55. package/dist/workflows/dag/frontend-implementation-contract.js +218 -17
  56. package/dist/workflows/dag/frontend-review-context.js +7 -1
  57. package/dist/workflows/dag/frontend-verification-trace.js +14 -3
  58. package/dist/workflows/dag/frontend-worktree-diff.js +14 -3
  59. package/dist/workflows/dag/init-hybrid.js +96 -34
  60. package/dist/workflows/dag/output-protocol.js +180 -7
  61. package/dist/workflows/dag/runner.js +141 -52
  62. package/dist/workflows/dag/types.js +4 -0
  63. package/dist/workflows/dag/validate.js +3 -2
  64. package/docs/skills/README.md +7 -7
  65. package/docs/templates/adr.md +60 -60
  66. package/docs/templates/agent-dag-authority-surface-audit.prompt.md +94 -94
  67. package/docs/templates/agent-dag-decision-envelope.schema.json +213 -213
  68. package/docs/templates/agent-dag-decision-gate.prompt.md +246 -246
  69. package/docs/templates/agent-dag-process-supervisor.prompt.md +98 -98
  70. package/docs/templates/agent-dag-report.schema.json +473 -473
  71. package/docs/templates/agent-dag-review-verdict.prompt.md +68 -68
  72. package/docs/templates/backend-test-dag.json +100 -8
  73. package/docs/templates/backend-test-result.schema.json +99 -99
  74. package/docs/templates/feature-spec.md +53 -53
  75. package/docs/templates/frontend-design-contract.md +42 -42
  76. package/docs/templates/frontend-eval/fixtures/failures/01-type-build-error.md +17 -17
  77. package/docs/templates/frontend-eval/fixtures/failures/02-unit-component-test-fail.md +16 -16
  78. package/docs/templates/frontend-eval/fixtures/failures/03-fixture-schema-drift.md +16 -16
  79. package/docs/templates/frontend-eval/fixtures/failures/04-missing-loading-empty-error-state.md +16 -16
  80. package/docs/templates/frontend-eval/fixtures/failures/05-forbidden-write-writeset-expansion.md +16 -16
  81. package/docs/templates/frontend-eval/fixtures/failures/06-unapproved-dependency-add.md +16 -16
  82. package/docs/templates/frontend-eval/fixtures/failures/07-mock-production-on.md +21 -21
  83. package/docs/templates/frontend-eval/fixtures/functional/01-simple-component-style.md +29 -29
  84. package/docs/templates/frontend-eval/fixtures/functional/02-form-validation.md +28 -28
  85. package/docs/templates/frontend-eval/fixtures/functional/03-list-detail-page.md +28 -28
  86. package/docs/templates/frontend-eval/fixtures/functional/04-api-mock.md +29 -29
  87. package/docs/templates/frontend-eval/fixtures/functional/05-permission-auth-gated-ui.md +27 -27
  88. package/docs/templates/frontend-eval/fixtures/functional/06-ssr-server-client-boundary.md +28 -28
  89. package/docs/templates/frontend-eval/fixtures/functional/07-shared-public-component-api.md +28 -28
  90. package/docs/templates/frontend-eval/fixtures/functional/08-pure-local-no-remote.md +27 -27
  91. package/docs/templates/frontend-eval/metrics.md +138 -138
  92. package/docs/templates/frontend-eval/smoke-targets.md +53 -53
  93. package/docs/templates/frontend-task-constraints.md +35 -35
  94. package/docs/templates/frontend-task-requirement.md +70 -70
  95. package/docs/templates/init-evolution-review.md +35 -35
  96. package/docs/templates/init-managed-agents.md +10 -5
  97. package/docs/templates/interactive-ui-round2-experiment.md +66 -66
  98. package/docs/templates/knowledge-graph-bootstrap-dag.json +118 -118
  99. package/docs/templates/knowledge-sync-dag.json +178 -178
  100. package/docs/templates/knowledge-sync-draft.schema.json +71 -71
  101. package/docs/templates/product-line/closeout.yaml +9 -9
  102. package/docs/templates/product-line/design.md +13 -13
  103. package/docs/templates/product-line/links.md +10 -10
  104. package/docs/templates/product-line/requirement.md +17 -17
  105. package/docs/templates/product-line/test-plan.md +7 -7
  106. package/docs/templates/production-readiness-checklist.md +57 -57
  107. package/docs/templates/project-start-checklist.md +9 -9
  108. package/docs/templates/qa-report.md +48 -48
  109. package/docs/templates/sprint-contract.md +29 -29
  110. package/docs/templates/worker-dogfood-evidence.md +80 -80
  111. package/docs/templates/worker-dogfood-setup.md +68 -68
  112. package/package.json +1 -1
  113. package/scripts/kb-bootstrap-init-skeleton.sh +0 -0
  114. package/scripts/kb-graph-incremental-prepare.mjs +386 -386
  115. package/scripts/kb-graph-materialize.mjs +105 -105
  116. package/scripts/kb-graph-promote.mjs +164 -164
  117. package/scripts/kb-query.mjs +554 -554
  118. package/skills/ai-engineering-context/SKILL.md +48 -48
  119. package/skills/analyze-product-dependencies/SKILL.md +67 -67
  120. package/skills/analyze-product-dependencies/agents/openai.yaml +4 -4
  121. package/skills/analyze-product-dependencies/references/api-documentation-schema.md +30 -30
  122. package/skills/analyze-product-dependencies/references/dependency-analysis-schema.md +28 -28
  123. package/skills/analyze-product-dependencies/references/example.md +76 -76
  124. package/skills/analyze-product-dependencies/references/forward-test-cases.md +35 -35
  125. package/skills/analyze-product-dependencies/references/input-contract.md +11 -11
  126. package/skills/analyze-product-dependencies/references/scouting-rules.md +61 -61
  127. package/skills/analyze-product-dependencies/scripts/test-validators.mjs +267 -267
  128. package/skills/analyze-product-dependencies/scripts/validate-api-documentation.mjs +101 -101
  129. package/skills/analyze-product-dependencies/scripts/validate-dependency-analysis.mjs +142 -142
  130. package/skills/analyze-product-dependencies/scripts/validate-product-requirement-input.mjs +76 -76
  131. package/skills/analyze-product-dependencies/scripts/validation-helpers.mjs +146 -146
  132. package/skills/analyze-product-requirements/SKILL.md +90 -90
  133. package/skills/analyze-product-requirements/agents/openai.yaml +4 -4
  134. package/skills/analyze-product-requirements/references/acceptance-criteria.md +91 -91
  135. package/skills/analyze-product-requirements/references/clarification-and-knowledge.md +56 -56
  136. package/skills/analyze-product-requirements/references/example.md +86 -86
  137. package/skills/analyze-product-requirements/references/forward-test-cases.md +66 -66
  138. package/skills/analyze-product-requirements/references/product-analysis-schema.md +32 -32
  139. package/skills/analyze-product-requirements/references/product-requirement-schema.md +33 -33
  140. package/skills/analyze-product-requirements/references/requirement-clarification-schema.md +35 -35
  141. package/skills/analyze-product-requirements/scripts/test-validators.mjs +193 -193
  142. package/skills/analyze-product-requirements/scripts/validate-product-analysis.mjs +69 -69
  143. package/skills/analyze-product-requirements/scripts/validate-product-requirement.mjs +97 -97
  144. package/skills/analyze-product-requirements/scripts/validate-requirement-clarification.mjs +98 -98
  145. package/skills/analyze-product-requirements/scripts/validation-helpers.mjs +156 -156
  146. package/skills/browser-tools/browser-content.js +103 -103
  147. package/skills/browser-tools/browser-cookies.js +35 -35
  148. package/skills/browser-tools/browser-eval.js +53 -53
  149. package/skills/browser-tools/browser-hn-scraper.js +108 -108
  150. package/skills/browser-tools/browser-nav.js +44 -44
  151. package/skills/browser-tools/browser-pick.js +162 -162
  152. package/skills/browser-tools/browser-screenshot.js +34 -34
  153. package/skills/browser-tools/browser-start.js +86 -86
  154. package/skills/browser-tools/package-lock.json +2556 -2556
  155. package/skills/browser-tools/package.json +19 -19
  156. package/skills/code-review-core/SKILL.md +20 -20
  157. package/skills/codebase-scout/SKILL.md +19 -19
  158. package/skills/grill-me/SKILL.md +10 -10
  159. package/skills/loop-agent/SKILL.md +5 -2
  160. package/skills/loop-agent/references/README.md +67 -67
  161. package/skills/loop-agent/references/command-reference.md +17 -15
  162. package/skills/loop-agent/references/docs-converge.md +126 -126
  163. package/skills/loop-agent/references/harness-policy.md +3 -4
  164. package/skills/loop-agent/references/hybrid-dag.md +1 -1
  165. package/skills/loop-agent/references/learned/README.md +21 -21
  166. package/skills/loop-agent/references/long-running-loop.md +57 -57
  167. package/skills/loop-agent/references/one-shot-runs.md +85 -85
  168. package/skills/loop-agent/references/pi-prompt.md +23 -23
  169. package/skills/loop-agent/references/pi-subagent-assisted-mode.md +84 -84
  170. package/skills/loop-agent/references/post-implementation-and-patterns.md +1 -1
  171. package/skills/loop-agent/references/source-and-plan-practice.md +3 -2
  172. package/skills/loop-agent/references/task-workflow.md +7 -5
  173. package/skills/playwright-cli/SKILL.md +420 -420
  174. package/skills/playwright-cli/references/element-attributes.md +23 -23
  175. package/skills/playwright-cli/references/playwright-tests.md +39 -39
  176. package/skills/playwright-cli/references/request-mocking.md +87 -87
  177. package/skills/playwright-cli/references/running-code.md +241 -241
  178. package/skills/playwright-cli/references/session-management.md +225 -225
  179. package/skills/playwright-cli/references/storage-state.md +275 -275
  180. package/skills/playwright-cli/references/test-generation.md +433 -433
  181. package/skills/playwright-cli/references/tracing.md +139 -139
  182. package/skills/playwright-cli/references/video-recording.md +143 -143
  183. package/skills/requesting-code-review/SKILL.md +101 -101
  184. package/skills/requesting-code-review/code-reviewer.md +168 -168
  185. package/skills/systematic-debugging/CREATION-LOG.md +119 -119
  186. package/skills/systematic-debugging/condition-based-waiting-example.ts +158 -158
  187. package/skills/systematic-debugging/condition-based-waiting.md +115 -115
  188. package/skills/systematic-debugging/defense-in-depth.md +122 -122
  189. package/skills/systematic-debugging/find-polluter.sh +63 -63
  190. package/skills/systematic-debugging/root-cause-tracing.md +169 -169
  191. package/skills/systematic-debugging/test-academic.md +14 -14
  192. package/skills/systematic-debugging/test-pressure-1.md +58 -58
  193. package/skills/systematic-debugging/test-pressure-2.md +68 -68
  194. package/skills/systematic-debugging/test-pressure-3.md +69 -69
  195. package/skills/using-git-worktrees/SKILL.md +215 -215
  196. package/skills/verification-before-completion/SKILL.md +154 -154
  197. package/skills/webapp-testing/SKILL.md +19 -19
@@ -1,70 +1,70 @@
1
- # 前端任务需求模板
2
-
3
- ## 用户目标
4
-
5
- TODO
6
-
7
- ## 目标页面/组件/路由
8
-
9
- TODO
10
-
11
- ## 用户流程
12
-
13
- TODO
14
-
15
- ## 必须状态
16
-
17
- ### loading
18
-
19
- TODO
20
-
21
- ### empty
22
-
23
- TODO
24
-
25
- ### error
26
-
27
- TODO
28
-
29
- ### success
30
-
31
- TODO
32
-
33
- ### disabled
34
-
35
- TODO
36
-
37
- ## 目标运行环境
38
-
39
- ### desktop
40
-
41
- TODO
42
-
43
- ### mobile
44
-
45
- TODO
46
-
47
- ### tablet
48
-
49
- TODO
50
-
51
- ## 交互要求
52
-
53
- TODO
54
-
55
- ## 接口与 Mock 输入
56
-
57
- - 接口文档/schema:TODO
58
- - endpoint、method、关键请求/响应字段:TODO
59
- - 是否允许依赖真实后端:TODO
60
- - 是否要求离线或独立行为验证:TODO
61
- - success/empty/error/permission 状态:TODO
62
- - 后端当前就绪状态与 Real Integration Gap:TODO
63
-
64
- ## 验收标准
65
-
66
- TODO
67
-
68
- ## 非目标
69
-
70
- TODO
1
+ # 前端任务需求模板
2
+
3
+ ## 用户目标
4
+
5
+ TODO
6
+
7
+ ## 目标页面/组件/路由
8
+
9
+ TODO
10
+
11
+ ## 用户流程
12
+
13
+ TODO
14
+
15
+ ## 必须状态
16
+
17
+ ### loading
18
+
19
+ TODO
20
+
21
+ ### empty
22
+
23
+ TODO
24
+
25
+ ### error
26
+
27
+ TODO
28
+
29
+ ### success
30
+
31
+ TODO
32
+
33
+ ### disabled
34
+
35
+ TODO
36
+
37
+ ## 目标运行环境
38
+
39
+ ### desktop
40
+
41
+ TODO
42
+
43
+ ### mobile
44
+
45
+ TODO
46
+
47
+ ### tablet
48
+
49
+ TODO
50
+
51
+ ## 交互要求
52
+
53
+ TODO
54
+
55
+ ## 接口与 Mock 输入
56
+
57
+ - 接口文档/schema:TODO
58
+ - endpoint、method、关键请求/响应字段:TODO
59
+ - 是否允许依赖真实后端:TODO
60
+ - 是否要求离线或独立行为验证:TODO
61
+ - success/empty/error/permission 状态:TODO
62
+ - 后端当前就绪状态与 Real Integration Gap:TODO
63
+
64
+ ## 验收标准
65
+
66
+ TODO
67
+
68
+ ## 非目标
69
+
70
+ TODO
@@ -1,35 +1,35 @@
1
- # Init Evolution Review
2
-
3
- Date:
4
- Base: `<full commit SHA or resolvable --base ref>`
5
- Head: `<full commit SHA for the reviewed HEAD>`
6
-
7
- `bash scripts/check-init-evolution-needed.sh --strict --base <ref>` 接受 Base 精确匹配该 `<ref>`、Head 为当前 `HEAD` 或其可解析祖先提交的报告。历史报告、`working tree` 等不可解析文字不能为其他变更范围放行严格检查。当 Head 是祖先时,`reportHead..HEAD` 区间内出现新的 `model-review` 高影响路径会使报告失效;仅 `advisory` 或 `surface-check` 的后续变化不影响已完成的高影响审查。
8
-
9
- ## Changed Surface
10
-
11
- -
12
-
13
- ## Decision
14
-
15
- Choose one:
16
-
17
- - No init impact
18
- - Surface check only
19
- - Init update required
20
-
21
- Rationale:
22
-
23
- ## Updates Made
24
-
25
- -
26
-
27
- ## Verification
28
-
29
- ```bash
30
- # commands and results
31
- ```
32
-
33
- ## Residual Risk
34
-
35
- -
1
+ # Init Evolution Review
2
+
3
+ Date:
4
+ Base: `<full commit SHA or resolvable --base ref>`
5
+ Head: `<full commit SHA for the reviewed HEAD>`
6
+
7
+ `bash scripts/check-init-evolution-needed.sh --strict --base <ref>` 接受 Base 精确匹配该 `<ref>`、Head 为当前 `HEAD` 或其可解析祖先提交的报告。历史报告、`working tree` 等不可解析文字不能为其他变更范围放行严格检查。当 Head 是祖先时,`reportHead..HEAD` 区间内出现新的 `model-review` 高影响路径会使报告失效;仅 `advisory` 或 `surface-check` 的后续变化不影响已完成的高影响审查。
8
+
9
+ ## Changed Surface
10
+
11
+ -
12
+
13
+ ## Decision
14
+
15
+ Choose one:
16
+
17
+ - No init impact
18
+ - Surface check only
19
+ - Init update required
20
+
21
+ Rationale:
22
+
23
+ ## Updates Made
24
+
25
+ -
26
+
27
+ ## Verification
28
+
29
+ ```bash
30
+ # commands and results
31
+ ```
32
+
33
+ ## Residual Risk
34
+
35
+ -
@@ -86,19 +86,24 @@ pwd → `README.md` → `harness.json` → `__LOOP_AGENT_GOVERNANCE_ROOT__/READM
86
86
 
87
87
  ```bash
88
88
  loop-agent new-task <task-id> "任务标题"
89
- # 有原始 PRD 时(推荐):loop-agent import-prd <task-id> --file <path-to-prd.md>
89
+ # 推荐默认:详细 PRD import prepare(无需手写两个 source;默认无 LLM 写 source)
90
+ loop-agent import-prd <task-id> --file <path-to-prd.md>
90
91
  # 非微小或跨会话任务(推荐):loop-agent plan create <plan-id> "<title>"
91
- # write .harness/tasks/<task-id>/source/需求.md
92
- # write .harness/tasks/<task-id>/source/执行约束.md
92
+ loop-agent task source prepare <task-id> \
93
+ --use-imported-prd \
94
+ --allowed-path "<glob>" \
95
+ --forbidden-path ".harness/**" \
96
+ --verify "typecheck:npm run typecheck" \
97
+ --apply --json
93
98
  loop-agent dag run-task <task-id> --profile auto --strict-models
94
99
  loop-agent dag validate --dag .harness/tasks/<task-id>/dag.json --strict-models --strict-governance
95
100
  loop-agent run-dag --dag .harness/tasks/<task-id>/dag.json --cwd .
96
101
  # 有 plan 时收尾:loop-agent plan complete <plan-id> --summary "..."
97
102
  ```
98
103
 
99
- `source/需求.md` 与 `source/执行约束.md` 必需;有原始 PRD 时优先用 `import-prd` 原样归档。`import-prd` / `plan create` 不是 `dag run-task` 的硬依赖。写入前同步 `task.json.allowedPaths` / `task.json.forbiddenPaths` 并审查 writer `writeSet`。
104
+ `source/需求.md` 与 `source/执行约束.md` 仍必需(M8/M9),但默认由 `task source prepare --apply` 投影生成,而不是主会话手写。有原始 PRD 时优先 `import-prd` 原样归档。`import-prd` / `plan create` 不是 `dag run-task` 的硬依赖。写入前同步 `task.json.allowedPaths` / `task.json.forbiddenPaths` 并审查 writer `writeSet`。
100
105
 
101
- 凡是影响项目公共契约、执行入口、交付流水线、自动化/治理、数据模型、安全或权限模型、跨模块行为、用户可见工作流的改动,都必须在编辑实现文件前先创建任务、写好两个 source 文件、生成 DAG,并审查 DAG/writeSet。
106
+ 凡是影响项目公共契约、执行入口、交付流水线、自动化/治理、数据模型、安全或权限模型、跨模块行为、用户可见工作流的改动,都必须在编辑实现文件前先创建任务、完成 source prepare(或等价 managed contract)、生成 DAG,并审查 DAG/writeSet。
102
107
 
103
108
  ### 任务类型路由(taskKind)
104
109
 
@@ -1,66 +1,66 @@
1
- # Interactive UI Round-2 A/B/C Experiment
2
-
3
- ## Frozen inputs
4
-
5
- - Target repo / commit:
6
- - Controller version:
7
- - TaskSpec / acceptance hash:
8
- - Base DAG:
9
- - Model matrix:
10
-
11
- Prepare the fixture and three DAGs after installing a published controller that contains `interactive-ui` support:
12
-
13
- ```bash
14
- bash scripts-local/setup-drill-round2-react.sh /tmp/drill-round2-react-target
15
- npm run build
16
- node scripts-local/prepare-round2-ui-experiment.mjs \
17
- /tmp/drill-round2-react-target \
18
- features/F-2026-001/tasks/FE-001.yaml \
19
- /tmp/round2-ui-experiment \
20
- loop-agent
21
- node scripts-local/build-round2-ui-variants.mjs \
22
- /tmp/round2-ui-experiment/round2-ui-base.json \
23
- /tmp/round2-ui-experiment/variants
24
- ```
25
-
26
- Validate every generated variant with the same frozen controller before running it. Use a unique run ID for A, B, and C; do not rewrite `executorModels`.
27
-
28
- ## Variants
29
-
30
- | Variant | Writer prompt | Writer tier | Run ID | Result |
31
- | --- | --- | --- | --- | --- |
32
- | A | interactive-ui contract | MED | | |
33
- | B | default contract | HIGH | | |
34
- | C | interactive-ui contract | HIGH | | |
35
-
36
- ## Metrics
37
-
38
- | Metric | A | B | C |
39
- | --- | --- | --- | --- |
40
- | First-pass review gate pass | | | |
41
- | Framework-native `.tsx` component | | | |
42
- | Route/parent integration | | | |
43
- | DOM interaction tests | | | |
44
- | Helper-only escape | | | |
45
- | Duration | | | |
46
- | Tokens | | | |
47
-
48
- ## Required evidence per run
49
-
50
- - DAG JSON and run ID
51
- - implement/repair writer model and prompt profile
52
- - changed component path
53
- - integration path
54
- - interaction test path
55
- - DOM assertions mapped to AC-FE-001
56
- - review verdict and deterministic test output
57
- - diff boundary audit
58
-
59
- ## Decision
60
-
61
- - Production default:
62
- - Evidence:
63
- - Cost/quality trade-off:
64
- - Follow-up:
65
-
66
- Do not conclude from a single run when provider or environment failures occurred. Re-run the affected variant with the same frozen inputs and a new run ID.
1
+ # Interactive UI Round-2 A/B/C Experiment
2
+
3
+ ## Frozen inputs
4
+
5
+ - Target repo / commit:
6
+ - Controller version:
7
+ - TaskSpec / acceptance hash:
8
+ - Base DAG:
9
+ - Model matrix:
10
+
11
+ Prepare the fixture and three DAGs after installing a published controller that contains `interactive-ui` support:
12
+
13
+ ```bash
14
+ bash scripts-local/setup-drill-round2-react.sh /tmp/drill-round2-react-target
15
+ npm run build
16
+ node scripts-local/prepare-round2-ui-experiment.mjs \
17
+ /tmp/drill-round2-react-target \
18
+ features/F-2026-001/tasks/FE-001.yaml \
19
+ /tmp/round2-ui-experiment \
20
+ loop-agent
21
+ node scripts-local/build-round2-ui-variants.mjs \
22
+ /tmp/round2-ui-experiment/round2-ui-base.json \
23
+ /tmp/round2-ui-experiment/variants
24
+ ```
25
+
26
+ Validate every generated variant with the same frozen controller before running it. Use a unique run ID for A, B, and C; do not rewrite `executorModels`.
27
+
28
+ ## Variants
29
+
30
+ | Variant | Writer prompt | Writer tier | Run ID | Result |
31
+ | --- | --- | --- | --- | --- |
32
+ | A | interactive-ui contract | MED | | |
33
+ | B | default contract | HIGH | | |
34
+ | C | interactive-ui contract | HIGH | | |
35
+
36
+ ## Metrics
37
+
38
+ | Metric | A | B | C |
39
+ | --- | --- | --- | --- |
40
+ | First-pass review gate pass | | | |
41
+ | Framework-native `.tsx` component | | | |
42
+ | Route/parent integration | | | |
43
+ | DOM interaction tests | | | |
44
+ | Helper-only escape | | | |
45
+ | Duration | | | |
46
+ | Tokens | | | |
47
+
48
+ ## Required evidence per run
49
+
50
+ - DAG JSON and run ID
51
+ - implement/repair writer model and prompt profile
52
+ - changed component path
53
+ - integration path
54
+ - interaction test path
55
+ - DOM assertions mapped to AC-FE-001
56
+ - review verdict and deterministic test output
57
+ - diff boundary audit
58
+
59
+ ## Decision
60
+
61
+ - Production default:
62
+ - Evidence:
63
+ - Cost/quality trade-off:
64
+ - Follow-up:
65
+
66
+ Do not conclude from a single run when provider or environment failures occurred. Re-run the affected variant with the same frozen inputs and a new run ID.
@@ -1,118 +1,118 @@
1
- {
2
- "$schema": "./agent-dag.schema.json",
3
- "version": 2,
4
- "title": "Knowledge-graph bootstrap DAG template",
5
- "objective": "AI-assisted business knowledge graph initialization: inventory repo signals, propose entities/edges under staging only, validate, review-gate, promote without overwrite, materialize query indexes.",
6
- "successCriteria": [
7
- "preflight requires knowledge/bootstrap/scope.yaml and status.yaml",
8
- "inventory writes knowledge/bootstrap/inventory.json",
9
- "propose-pi writes only knowledge/bootstrap/staging/** with evidence and non-asserted confidence",
10
- "validate-shell fails on empty staging, missing edges.proposed/coverage-notes, or self-asserted confidence",
11
- "multi-perspective review (structure/evidence/safety) each emit VERDICT; aggregate gate requires all pass before promote",
12
- "promote merges new formal files only (no overwrite)",
13
- "materialize writes knowledge/graph/entities-index.yaml and edges.yaml"
14
- ],
15
- "globalConstraints": [
16
- "AI must never set confidence: asserted.",
17
- "Do not modify src/** or Feature testing verdict/cases via bootstrap.",
18
- "Promote must not overwrite existing formal knowledge files.",
19
- "Complete B0/B1 skeleton before running this DAG."
20
- ],
21
- "defaults": {
22
- "executor": "shell",
23
- "contextProfile": "slim",
24
- "writePolicy": "read-only"
25
- },
26
- "tasks": [
27
- {
28
- "id": "kg-bootstrap-preflight-shell",
29
- "depends_on": [],
30
- "role": "verifier",
31
- "executor": "shell",
32
- "complexity": "LOW",
33
- "writePolicy": "read-only",
34
- "subtask_prompt": "Require bootstrap skeleton."
35
- },
36
- {
37
- "id": "kg-bootstrap-inventory-shell",
38
- "depends_on": ["kg-bootstrap-preflight-shell"],
39
- "role": "verifier",
40
- "executor": "shell",
41
- "complexity": "LOW",
42
- "writePolicy": "exclusive",
43
- "writeSet": ["knowledge/bootstrap/inventory.json"],
44
- "subtask_prompt": "B2 inventory."
45
- },
46
- {
47
- "id": "kg-bootstrap-propose-pi",
48
- "depends_on": ["kg-bootstrap-inventory-shell"],
49
- "role": "implementer",
50
- "executor": "pi",
51
- "toolProfile": "write",
52
- "complexity": "HIGH",
53
- "writePolicy": "exclusive",
54
- "writeSet": [
55
- "knowledge/bootstrap/staging/**",
56
- "knowledge/bootstrap/runs/**"
57
- ],
58
- "subtask_prompt": "B3 AI proposals in staging only; never asserted."
59
- },
60
- {
61
- "id": "kg-bootstrap-validate-shell",
62
- "depends_on": ["kg-bootstrap-propose-pi"],
63
- "role": "verifier",
64
- "executor": "shell",
65
- "complexity": "LOW",
66
- "writePolicy": "read-only",
67
- "subtask_prompt": "B4 validate staging."
68
- },
69
- {
70
- "id": "kg-bootstrap-review-pi",
71
- "depends_on": ["kg-bootstrap-validate-shell"],
72
- "role": "reviewer",
73
- "executor": "pi",
74
- "complexity": "HIGH",
75
- "writePolicy": "read-only",
76
- "outputContract": "First line VERDICT: pass or VERDICT: request-revision.",
77
- "subtask_prompt": "Review staging quality."
78
- },
79
- {
80
- "id": "kg-bootstrap-review-gate-shell",
81
- "depends_on": ["kg-bootstrap-review-pi"],
82
- "role": "verifier",
83
- "executor": "shell",
84
- "complexity": "LOW",
85
- "writePolicy": "read-only",
86
- "subtask_prompt": "Gate on VERDICT: pass."
87
- },
88
- {
89
- "id": "kg-bootstrap-promote-shell",
90
- "depends_on": ["kg-bootstrap-review-gate-shell"],
91
- "role": "verifier",
92
- "executor": "shell",
93
- "complexity": "LOW",
94
- "writePolicy": "exclusive",
95
- "writeSet": [
96
- "knowledge/domains/**",
97
- "knowledge/services/**",
98
- "knowledge/modules/**",
99
- "knowledge/graph/edges.manual.yaml",
100
- "features/**/knowledge-links.yaml"
101
- ],
102
- "subtask_prompt": "B5 promote new files only."
103
- },
104
- {
105
- "id": "kg-bootstrap-materialize-shell",
106
- "depends_on": ["kg-bootstrap-promote-shell"],
107
- "role": "closeout",
108
- "executor": "shell",
109
- "complexity": "LOW",
110
- "writePolicy": "exclusive",
111
- "writeSet": [
112
- "knowledge/graph/entities-index.yaml",
113
- "knowledge/graph/edges.yaml"
114
- ],
115
- "subtask_prompt": "B6 materialize indexes."
116
- }
117
- ]
118
- }
1
+ {
2
+ "$schema": "./agent-dag.schema.json",
3
+ "version": 2,
4
+ "title": "Knowledge-graph bootstrap DAG template",
5
+ "objective": "AI-assisted business knowledge graph initialization: inventory repo signals, propose entities/edges under staging only, validate, review-gate, promote without overwrite, materialize query indexes.",
6
+ "successCriteria": [
7
+ "preflight requires knowledge/bootstrap/scope.yaml and status.yaml",
8
+ "inventory writes knowledge/bootstrap/inventory.json",
9
+ "propose-pi writes only knowledge/bootstrap/staging/** with evidence and non-asserted confidence",
10
+ "validate-shell fails on empty staging, missing edges.proposed/coverage-notes, or self-asserted confidence",
11
+ "multi-perspective review (structure/evidence/safety) each emit VERDICT; aggregate gate requires all pass before promote",
12
+ "promote merges new formal files only (no overwrite)",
13
+ "materialize writes knowledge/graph/entities-index.yaml and edges.yaml"
14
+ ],
15
+ "globalConstraints": [
16
+ "AI must never set confidence: asserted.",
17
+ "Do not modify src/** or Feature testing verdict/cases via bootstrap.",
18
+ "Promote must not overwrite existing formal knowledge files.",
19
+ "Complete B0/B1 skeleton before running this DAG."
20
+ ],
21
+ "defaults": {
22
+ "executor": "shell",
23
+ "contextProfile": "slim",
24
+ "writePolicy": "read-only"
25
+ },
26
+ "tasks": [
27
+ {
28
+ "id": "kg-bootstrap-preflight-shell",
29
+ "depends_on": [],
30
+ "role": "verifier",
31
+ "executor": "shell",
32
+ "complexity": "LOW",
33
+ "writePolicy": "read-only",
34
+ "subtask_prompt": "Require bootstrap skeleton."
35
+ },
36
+ {
37
+ "id": "kg-bootstrap-inventory-shell",
38
+ "depends_on": ["kg-bootstrap-preflight-shell"],
39
+ "role": "verifier",
40
+ "executor": "shell",
41
+ "complexity": "LOW",
42
+ "writePolicy": "exclusive",
43
+ "writeSet": ["knowledge/bootstrap/inventory.json"],
44
+ "subtask_prompt": "B2 inventory."
45
+ },
46
+ {
47
+ "id": "kg-bootstrap-propose-pi",
48
+ "depends_on": ["kg-bootstrap-inventory-shell"],
49
+ "role": "implementer",
50
+ "executor": "pi",
51
+ "toolProfile": "write",
52
+ "complexity": "HIGH",
53
+ "writePolicy": "exclusive",
54
+ "writeSet": [
55
+ "knowledge/bootstrap/staging/**",
56
+ "knowledge/bootstrap/runs/**"
57
+ ],
58
+ "subtask_prompt": "B3 AI proposals in staging only; never asserted."
59
+ },
60
+ {
61
+ "id": "kg-bootstrap-validate-shell",
62
+ "depends_on": ["kg-bootstrap-propose-pi"],
63
+ "role": "verifier",
64
+ "executor": "shell",
65
+ "complexity": "LOW",
66
+ "writePolicy": "read-only",
67
+ "subtask_prompt": "B4 validate staging."
68
+ },
69
+ {
70
+ "id": "kg-bootstrap-review-pi",
71
+ "depends_on": ["kg-bootstrap-validate-shell"],
72
+ "role": "reviewer",
73
+ "executor": "pi",
74
+ "complexity": "HIGH",
75
+ "writePolicy": "read-only",
76
+ "outputContract": "First line VERDICT: pass or VERDICT: request-revision.",
77
+ "subtask_prompt": "Review staging quality."
78
+ },
79
+ {
80
+ "id": "kg-bootstrap-review-gate-shell",
81
+ "depends_on": ["kg-bootstrap-review-pi"],
82
+ "role": "verifier",
83
+ "executor": "shell",
84
+ "complexity": "LOW",
85
+ "writePolicy": "read-only",
86
+ "subtask_prompt": "Gate on VERDICT: pass."
87
+ },
88
+ {
89
+ "id": "kg-bootstrap-promote-shell",
90
+ "depends_on": ["kg-bootstrap-review-gate-shell"],
91
+ "role": "verifier",
92
+ "executor": "shell",
93
+ "complexity": "LOW",
94
+ "writePolicy": "exclusive",
95
+ "writeSet": [
96
+ "knowledge/domains/**",
97
+ "knowledge/services/**",
98
+ "knowledge/modules/**",
99
+ "knowledge/graph/edges.manual.yaml",
100
+ "features/**/knowledge-links.yaml"
101
+ ],
102
+ "subtask_prompt": "B5 promote new files only."
103
+ },
104
+ {
105
+ "id": "kg-bootstrap-materialize-shell",
106
+ "depends_on": ["kg-bootstrap-promote-shell"],
107
+ "role": "closeout",
108
+ "executor": "shell",
109
+ "complexity": "LOW",
110
+ "writePolicy": "exclusive",
111
+ "writeSet": [
112
+ "knowledge/graph/entities-index.yaml",
113
+ "knowledge/graph/edges.yaml"
114
+ ],
115
+ "subtask_prompt": "B6 materialize indexes."
116
+ }
117
+ ]
118
+ }