@tea-agent/loop-agent 0.2.1 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (135) hide show
  1. package/AGENTS.md +91 -87
  2. package/CHANGELOG.md +89 -52
  3. package/README.md +195 -180
  4. package/bin/agent-worker.js +22 -0
  5. package/bin/loop-agent.js +21 -21
  6. package/dist/application/dag/args.js +6 -0
  7. package/dist/application/dag/generate-task-dag.js +2 -0
  8. package/dist/application/dag/run-dag.js +3 -0
  9. package/dist/application/dag/validate-dag.js +40 -0
  10. package/dist/cli/command-definitions.js +2 -2
  11. package/dist/cli/program.js +24 -4
  12. package/dist/commands/init.js +1011 -459
  13. package/dist/commands/loop-benchmark.js +11 -11
  14. package/dist/commands/pi-reuse-benchmark.js +16 -16
  15. package/dist/executors/cursor-executor.js +1 -1
  16. package/dist/executors/dag-pi-executor.js +8 -1
  17. package/dist/task/runtime.js +27 -27
  18. package/dist/worker/cli.js +119 -0
  19. package/dist/worker/loop-agent/command-result.js +1 -0
  20. package/dist/worker/loop-agent/loop-agent-client.js +105 -0
  21. package/dist/worker/loop-agent/parse-json.js +14 -0
  22. package/dist/worker/materialize/harness-task-materializer.js +157 -0
  23. package/dist/worker/pool/failure-routing.js +98 -0
  24. package/dist/worker/pool/run-store.js +117 -0
  25. package/dist/worker/pool/types.js +1 -0
  26. package/dist/worker/preflight.js +108 -0
  27. package/dist/worker/profile-mapping.js +76 -0
  28. package/dist/worker/progress-reporter.js +81 -0
  29. package/dist/worker/report/morning-report.js +69 -0
  30. package/dist/worker/repos/repo-resolver.js +23 -0
  31. package/dist/worker/run-task/run-task.js +359 -0
  32. package/dist/worker/runner/run-ready.js +216 -0
  33. package/dist/worker/task-graph/acceptance-schema.js +25 -0
  34. package/dist/worker/task-graph/ready-queue.js +23 -0
  35. package/dist/worker/task-graph/task-graph-schema.js +28 -0
  36. package/dist/worker/task-graph/types.js +1 -0
  37. package/dist/worker/task-graph/validate.js +188 -0
  38. package/dist/worker/task-spec/complexity-mapping.js +8 -0
  39. package/dist/worker/task-spec/schema.js +116 -0
  40. package/dist/worker/task-spec/types.js +1 -0
  41. package/dist/worker/task-spec/validate.js +352 -0
  42. package/dist/workflows/dag/canvas-observer.js +275 -275
  43. package/dist/workflows/dag/dynamic-runtime/loop-until.js +2 -1
  44. package/dist/workflows/dag/dynamic-runtime/map.js +1 -0
  45. package/dist/workflows/dag/init-hybrid.js +3 -3
  46. package/dist/workflows/dag/skills.js +3 -3
  47. package/dist/workflows/dag/types.js +2 -0
  48. package/dist/workflows/dynamic/compile.js +11 -0
  49. package/dist/workflows/dynamic/spec.js +1 -0
  50. package/docs/README.md +72 -65
  51. package/docs/agent-dag-recovery-playbook.md +184 -184
  52. package/docs/agent-dag-runner.md +42 -40
  53. package/docs/architecture/runtime-boundaries.md +147 -147
  54. package/docs/cursor-executor-usage.md +25 -25
  55. package/docs/decisions/README.md +3 -3
  56. package/docs/design/README.md +36 -36
  57. package/docs/development-principles.md +73 -71
  58. package/docs/dynamic-workflow-dag-engine-roadmap.md +1749 -1749
  59. package/docs/exec-plans/README.md +6 -6
  60. package/docs/exec-plans/active/README.md +7 -10
  61. package/docs/exec-plans/completed/README.md +19 -9
  62. package/docs/feature-workflow.md +186 -186
  63. package/docs/harness-methodology-debugging.md +153 -153
  64. package/docs/harness-methodology-tdd.md +130 -130
  65. package/docs/harness-methodology-verification.md +27 -27
  66. package/docs/init-surface.manifest.json +175 -0
  67. package/docs/loop-agent-harness.md +42 -42
  68. package/docs/production-readiness.md +96 -96
  69. package/docs/progress/README.md +3 -3
  70. package/docs/reports/README.md +5 -5
  71. package/docs/skills/README.md +6 -0
  72. package/docs/skills/vetted-skill-registry.md +26 -0
  73. package/docs/templates/adr.md +60 -60
  74. package/docs/templates/agent-dag-authority-surface-audit.prompt.md +94 -94
  75. package/docs/templates/agent-dag-decision-envelope.schema.json +213 -213
  76. package/docs/templates/agent-dag-decision-gate-dogfood-report.md +117 -117
  77. package/docs/templates/agent-dag-decision-gate.prompt.md +246 -246
  78. package/docs/templates/agent-dag-process-supervisor.prompt.md +98 -98
  79. package/docs/templates/agent-dag-report.schema.json +454 -454
  80. package/docs/templates/agent-dag-review-verdict.prompt.md +68 -68
  81. package/docs/templates/agent-dag.base.json +195 -195
  82. package/docs/templates/agent-dag.final-verification.json +190 -190
  83. package/docs/templates/agent-dag.schema.json +316 -316
  84. package/docs/templates/agent-dag.supervised-implementation.json +500 -500
  85. package/docs/templates/exec-plan.md +64 -64
  86. package/docs/templates/feature-spec.md +53 -53
  87. package/docs/templates/hybrid-dag.json +193 -193
  88. package/docs/templates/init-evolution-review.md +33 -0
  89. package/docs/templates/production-readiness-checklist.md +57 -57
  90. package/docs/templates/progress-log.md +17 -17
  91. package/docs/templates/project-start-checklist.md +9 -9
  92. package/docs/templates/qa-report.md +48 -48
  93. package/docs/templates/sprint-contract.md +29 -29
  94. package/docs/verification-matrix.md +41 -41
  95. package/examples/decision-gate-agent-dag.json +123 -123
  96. package/examples/example-dag.json +51 -51
  97. package/examples/hybrid-loop-agent-dag.json +194 -194
  98. package/harness.json +94 -92
  99. package/package.json +66 -62
  100. package/skills/ai-engineering-context/SKILL.md +48 -48
  101. package/skills/code-review-core/SKILL.md +20 -0
  102. package/skills/codebase-scout/SKILL.md +19 -0
  103. package/skills/init-capability-evolution/SKILL.md +69 -0
  104. package/skills/loop-agent/SKILL.md +147 -145
  105. package/skills/loop-agent/references/README.md +67 -67
  106. package/skills/loop-agent/references/command-reference.md +403 -357
  107. package/skills/loop-agent/references/harness-policy.md +259 -258
  108. package/skills/loop-agent/references/hybrid-dag.md +216 -216
  109. package/skills/loop-agent/references/learned/README.md +21 -21
  110. package/skills/loop-agent/references/long-running-loop.md +59 -59
  111. package/skills/loop-agent/references/model-routing.md +36 -36
  112. package/skills/loop-agent/references/multi-worktree.md +54 -54
  113. package/skills/loop-agent/references/one-shot-runs.md +85 -85
  114. package/skills/loop-agent/references/orchestrator-and-interventions.md +169 -169
  115. package/skills/loop-agent/references/pi-prompt.md +23 -23
  116. package/skills/loop-agent/references/pi-subagent-assisted-mode.md +81 -81
  117. package/skills/loop-agent/references/post-implementation-and-patterns.md +44 -44
  118. package/skills/loop-agent/references/task-workflow.md +84 -84
  119. package/skills/loop-agent/references/verification-and-failure-handling.md +128 -128
  120. package/skills/requesting-code-review/SKILL.md +101 -101
  121. package/skills/requesting-code-review/code-reviewer.md +168 -168
  122. package/skills/systematic-debugging/CREATION-LOG.md +119 -119
  123. package/skills/systematic-debugging/SKILL.md +296 -296
  124. package/skills/systematic-debugging/condition-based-waiting-example.ts +158 -158
  125. package/skills/systematic-debugging/condition-based-waiting.md +115 -115
  126. package/skills/systematic-debugging/defense-in-depth.md +122 -122
  127. package/skills/systematic-debugging/find-polluter.sh +63 -63
  128. package/skills/systematic-debugging/root-cause-tracing.md +169 -169
  129. package/skills/systematic-debugging/test-academic.md +14 -14
  130. package/skills/systematic-debugging/test-pressure-1.md +58 -58
  131. package/skills/systematic-debugging/test-pressure-2.md +68 -68
  132. package/skills/systematic-debugging/test-pressure-3.md +69 -69
  133. package/skills/test-driven-development/SKILL.md +20 -0
  134. package/skills/verification-before-completion/SKILL.md +154 -154
  135. package/skills/webapp-testing/SKILL.md +19 -0
@@ -1,117 +1,117 @@
1
- # Agent DAG Decision Gate Dogfood Report Template
2
-
3
- > Copy to `docs/reports/YYYY-MM-DD-agent-dag-decision-gate-dogfood.md` after each dogfood run.
4
-
5
- ## Run Metadata
6
-
7
- | Field | Value |
8
- |---|---|
9
- | Date | YYYY-MM-DD |
10
- | Topic | |
11
- | DAG input | `<temp-dir>/<topic>-decision-gate-dag.json` |
12
- | Run ID | |
13
- | Run facts | `.harness/dag-runs/completed/<run-id>/` |
14
- | Gate type | `acceptance-gate` |
15
- | Decision node | `decision-pi` |
16
- | Model | `gpt-5.5` via `executorModels.pi.HIGH` |
17
-
18
- ## Work Type
19
-
20
- Choose one:
21
-
22
- - [ ] Low-risk docs/template change — expected `auto-approve` or `approve-with-constraints`
23
- - [ ] Runtime/loop-agent change — expected `request-revision`, `run-more-verification`, or `auto-approve`
24
- - [ ] Contract/security/high-risk simulation — expected `escalate-to-human`
25
-
26
- ## Deterministic Evidence
27
-
28
- | Evidence | Status | Notes |
29
- |---|---|---|
30
- | `verify-shell/result.summary.md` | verified / partial / missing | |
31
- | `state.json` | verified / partial / missing | |
32
- | git diff / changed file list | verified / partial / missing | |
33
- | progress/report/artifacts | verified / partial / missing | |
34
-
35
- ## Decision Envelope Summary
36
-
37
- Paste the **exact** ` ```DECISION_ENVELOPE_JSON ` fenced block from `decision-pi/result.summary.md` (info string must be `DECISION_ENVELOPE_JSON`, not `json`). Also record `decision`, `requiresHuman`, `nextAction`, and verified evidence count.
38
-
39
- ```DECISION_ENVELOPE_JSON
40
- {
41
- "schemaVersion": 1,
42
- "gateType": "acceptance-gate",
43
- "decisionScope": "dag-run",
44
- "decision": "auto-approve",
45
- "confidence": 0.88,
46
- "riskLevel": "low",
47
- "requiresHuman": false,
48
- "nextAction": "continue",
49
- "policyVersion": "agent-dag-decision-gate-v1",
50
- "policyChecks": {
51
- "mustEscalateFlags": [],
52
- "evidenceComplete": true,
53
- "allowedAutoApprove": true
54
- },
55
- "rationale": ["..."],
56
- "evidence": [
57
- {
58
- "path": ".harness/dag-runs/completed/<run-id>/verify-shell/result.summary.md",
59
- "kind": "shell-output",
60
- "status": "verified",
61
- "summary": "verification commands passed"
62
- }
63
- ],
64
- "blockingFindings": [],
65
- "requiredRevisions": [],
66
- "riskFlags": [],
67
- "humanEscalation": null,
68
- "audit": {
69
- "runId": "<run-id>",
70
- "nodeId": "decision-pi",
71
- "model": "gpt-5.5"
72
- }
73
- }
74
- ```
75
-
76
- ## Quality Metrics
77
-
78
- | Metric | Value | Notes |
79
- |---|---:|---|
80
- | `decisionLatencyMs` | | From node duration |
81
- | `tokenCostEstimate` | | If available |
82
- | `humanInterruptCount` | | Should be 0 for low-risk auto decisions |
83
- | `falseEscalation` | yes / no | Escalated when policy allowed auto decision |
84
- | `missedEscalation` | yes / no | Auto-approved when policy required human escalation |
85
- | `revisionCaughtBeforeHuman` | yes / no | Gate requested revision before human involvement |
86
- | `verifyPassAfterGate` | yes / no | Later verification passed after gate action |
87
- | `evidenceCompleteness` | complete / partial / missing | Based on verified evidence count |
88
-
89
- ## Evaluation
90
-
91
- ### What the gate got right
92
-
93
- -
94
-
95
- ### What the gate got wrong or over/under-weighted
96
-
97
- -
98
-
99
- ### Prompt / policy adjustments needed
100
-
101
- -
102
-
103
- ## Recommendation
104
-
105
- Choose one:
106
-
107
- - [ ] `proceed` — enough evidence to move toward parser/runtime work
108
- - [ ] `needs-more-dogfood` — continue M1/M2 advisory runs
109
- - [ ] `defer` — gate quality not yet good enough
110
-
111
- Rationale:
112
-
113
- -
114
-
115
- ## Follow-ups
116
-
117
- -
1
+ # Agent DAG Decision Gate Dogfood Report Template
2
+
3
+ > Copy to `docs/reports/YYYY-MM-DD-agent-dag-decision-gate-dogfood.md` after each dogfood run.
4
+
5
+ ## Run Metadata
6
+
7
+ | Field | Value |
8
+ |---|---|
9
+ | Date | YYYY-MM-DD |
10
+ | Topic | |
11
+ | DAG input | `<temp-dir>/<topic>-decision-gate-dag.json` |
12
+ | Run ID | |
13
+ | Run facts | `.harness/dag-runs/completed/<run-id>/` |
14
+ | Gate type | `acceptance-gate` |
15
+ | Decision node | `decision-pi` |
16
+ | Model | `gpt-5.5` via `executorModels.pi.HIGH` |
17
+
18
+ ## Work Type
19
+
20
+ Choose one:
21
+
22
+ - [ ] Low-risk docs/template change — expected `auto-approve` or `approve-with-constraints`
23
+ - [ ] Runtime/loop-agent change — expected `request-revision`, `run-more-verification`, or `auto-approve`
24
+ - [ ] Contract/security/high-risk simulation — expected `escalate-to-human`
25
+
26
+ ## Deterministic Evidence
27
+
28
+ | Evidence | Status | Notes |
29
+ |---|---|---|
30
+ | `verify-shell/result.summary.md` | verified / partial / missing | |
31
+ | `state.json` | verified / partial / missing | |
32
+ | git diff / changed file list | verified / partial / missing | |
33
+ | progress/report/artifacts | verified / partial / missing | |
34
+
35
+ ## Decision Envelope Summary
36
+
37
+ Paste the **exact** ` ```DECISION_ENVELOPE_JSON ` fenced block from `decision-pi/result.summary.md` (info string must be `DECISION_ENVELOPE_JSON`, not `json`). Also record `decision`, `requiresHuman`, `nextAction`, and verified evidence count.
38
+
39
+ ```DECISION_ENVELOPE_JSON
40
+ {
41
+ "schemaVersion": 1,
42
+ "gateType": "acceptance-gate",
43
+ "decisionScope": "dag-run",
44
+ "decision": "auto-approve",
45
+ "confidence": 0.88,
46
+ "riskLevel": "low",
47
+ "requiresHuman": false,
48
+ "nextAction": "continue",
49
+ "policyVersion": "agent-dag-decision-gate-v1",
50
+ "policyChecks": {
51
+ "mustEscalateFlags": [],
52
+ "evidenceComplete": true,
53
+ "allowedAutoApprove": true
54
+ },
55
+ "rationale": ["..."],
56
+ "evidence": [
57
+ {
58
+ "path": ".harness/dag-runs/completed/<run-id>/verify-shell/result.summary.md",
59
+ "kind": "shell-output",
60
+ "status": "verified",
61
+ "summary": "verification commands passed"
62
+ }
63
+ ],
64
+ "blockingFindings": [],
65
+ "requiredRevisions": [],
66
+ "riskFlags": [],
67
+ "humanEscalation": null,
68
+ "audit": {
69
+ "runId": "<run-id>",
70
+ "nodeId": "decision-pi",
71
+ "model": "gpt-5.5"
72
+ }
73
+ }
74
+ ```
75
+
76
+ ## Quality Metrics
77
+
78
+ | Metric | Value | Notes |
79
+ |---|---:|---|
80
+ | `decisionLatencyMs` | | From node duration |
81
+ | `tokenCostEstimate` | | If available |
82
+ | `humanInterruptCount` | | Should be 0 for low-risk auto decisions |
83
+ | `falseEscalation` | yes / no | Escalated when policy allowed auto decision |
84
+ | `missedEscalation` | yes / no | Auto-approved when policy required human escalation |
85
+ | `revisionCaughtBeforeHuman` | yes / no | Gate requested revision before human involvement |
86
+ | `verifyPassAfterGate` | yes / no | Later verification passed after gate action |
87
+ | `evidenceCompleteness` | complete / partial / missing | Based on verified evidence count |
88
+
89
+ ## Evaluation
90
+
91
+ ### What the gate got right
92
+
93
+ -
94
+
95
+ ### What the gate got wrong or over/under-weighted
96
+
97
+ -
98
+
99
+ ### Prompt / policy adjustments needed
100
+
101
+ -
102
+
103
+ ## Recommendation
104
+
105
+ Choose one:
106
+
107
+ - [ ] `proceed` — enough evidence to move toward parser/runtime work
108
+ - [ ] `needs-more-dogfood` — continue M1/M2 advisory runs
109
+ - [ ] `defer` — gate quality not yet good enough
110
+
111
+ Rationale:
112
+
113
+ -
114
+
115
+ ## Follow-ups
116
+
117
+ -