@tea-agent/loop-agent 0.3.0 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +135 -133
- package/CHANGELOG.md +88 -63
- package/README.md +171 -168
- package/bin/agent-worker.js +22 -0
- package/bin/loop-agent.js +21 -21
- package/dist/commands/init.js +457 -457
- package/dist/commands/loop-benchmark.js +11 -11
- package/dist/commands/pi-reuse-benchmark.js +16 -16
- package/dist/executors/cursor-executor.js +1 -1
- package/dist/executors/dag-pi-executor.js +8 -1
- package/dist/task/runtime.js +27 -27
- package/dist/worker/cli.js +119 -0
- package/dist/worker/loop-agent/command-result.js +1 -0
- package/dist/worker/loop-agent/loop-agent-client.js +105 -0
- package/dist/worker/loop-agent/parse-json.js +14 -0
- package/dist/worker/materialize/harness-task-materializer.js +157 -0
- package/dist/worker/pool/failure-routing.js +98 -0
- package/dist/worker/pool/run-store.js +117 -0
- package/dist/worker/pool/types.js +1 -0
- package/dist/worker/preflight.js +108 -0
- package/dist/worker/profile-mapping.js +76 -0
- package/dist/worker/progress-reporter.js +81 -0
- package/dist/worker/report/morning-report.js +69 -0
- package/dist/worker/repos/repo-resolver.js +23 -0
- package/dist/worker/run-task/run-task.js +359 -0
- package/dist/worker/runner/run-ready.js +216 -0
- package/dist/worker/task-graph/acceptance-schema.js +25 -0
- package/dist/worker/task-graph/ready-queue.js +23 -0
- package/dist/worker/task-graph/task-graph-schema.js +28 -0
- package/dist/worker/task-graph/types.js +1 -0
- package/dist/worker/task-graph/validate.js +188 -0
- package/dist/worker/task-spec/complexity-mapping.js +8 -0
- package/dist/worker/task-spec/schema.js +116 -0
- package/dist/worker/task-spec/types.js +1 -0
- package/dist/worker/task-spec/validate.js +352 -0
- package/dist/workflows/dag/canvas-observer.js +275 -275
- package/docs/README.md +65 -61
- package/docs/agent-dag-recovery-playbook.md +184 -184
- package/docs/agent-dag-runner.md +42 -42
- package/docs/architecture/runtime-boundaries.md +147 -147
- package/docs/cursor-executor-usage.md +25 -25
- package/docs/decisions/README.md +3 -3
- package/docs/design/README.md +36 -36
- package/docs/development-principles.md +73 -71
- package/docs/dynamic-workflow-dag-engine-roadmap.md +1749 -1749
- package/docs/exec-plans/README.md +6 -6
- package/docs/exec-plans/active/README.md +7 -7
- package/docs/exec-plans/completed/README.md +19 -11
- package/docs/feature-workflow.md +186 -186
- package/docs/harness-methodology-debugging.md +153 -153
- package/docs/harness-methodology-tdd.md +130 -130
- package/docs/harness-methodology-verification.md +27 -27
- package/docs/loop-agent-harness.md +42 -42
- package/docs/production-readiness.md +96 -96
- package/docs/progress/README.md +3 -3
- package/docs/reports/README.md +5 -5
- package/docs/skills/README.md +6 -6
- package/docs/skills/vetted-skill-registry.md +22 -22
- package/docs/templates/adr.md +60 -60
- package/docs/templates/agent-dag-authority-surface-audit.prompt.md +94 -94
- package/docs/templates/agent-dag-decision-envelope.schema.json +213 -213
- package/docs/templates/agent-dag-decision-gate-dogfood-report.md +117 -117
- package/docs/templates/agent-dag-decision-gate.prompt.md +246 -246
- package/docs/templates/agent-dag-process-supervisor.prompt.md +98 -98
- package/docs/templates/agent-dag-report.schema.json +454 -454
- package/docs/templates/agent-dag-review-verdict.prompt.md +68 -68
- package/docs/templates/agent-dag.base.json +195 -195
- package/docs/templates/agent-dag.final-verification.json +190 -190
- package/docs/templates/agent-dag.schema.json +316 -316
- package/docs/templates/agent-dag.supervised-implementation.json +500 -500
- package/docs/templates/exec-plan.md +64 -64
- package/docs/templates/feature-spec.md +53 -53
- package/docs/templates/hybrid-dag.json +193 -193
- package/docs/templates/production-readiness-checklist.md +57 -57
- package/docs/templates/progress-log.md +17 -17
- package/docs/templates/project-start-checklist.md +9 -9
- package/docs/templates/qa-report.md +48 -48
- package/docs/templates/sprint-contract.md +29 -29
- package/docs/verification-matrix.md +41 -41
- package/examples/decision-gate-agent-dag.json +123 -123
- package/examples/example-dag.json +51 -51
- package/examples/hybrid-loop-agent-dag.json +194 -194
- package/harness.json +89 -89
- package/package.json +60 -58
- package/skills/ai-engineering-context/SKILL.md +48 -48
- package/skills/code-review-core/SKILL.md +20 -20
- package/skills/codebase-scout/SKILL.md +19 -19
- package/skills/loop-agent/SKILL.md +147 -145
- package/skills/loop-agent/references/README.md +67 -67
- package/skills/loop-agent/references/command-reference.md +368 -340
- package/skills/loop-agent/references/harness-policy.md +259 -258
- package/skills/loop-agent/references/hybrid-dag.md +216 -216
- package/skills/loop-agent/references/learned/README.md +21 -21
- package/skills/loop-agent/references/long-running-loop.md +59 -59
- package/skills/loop-agent/references/model-routing.md +36 -36
- package/skills/loop-agent/references/multi-worktree.md +54 -54
- package/skills/loop-agent/references/one-shot-runs.md +85 -85
- package/skills/loop-agent/references/orchestrator-and-interventions.md +169 -169
- package/skills/loop-agent/references/pi-prompt.md +23 -23
- package/skills/loop-agent/references/pi-subagent-assisted-mode.md +81 -81
- package/skills/loop-agent/references/post-implementation-and-patterns.md +44 -44
- package/skills/loop-agent/references/task-workflow.md +84 -84
- package/skills/loop-agent/references/verification-and-failure-handling.md +128 -128
- package/skills/requesting-code-review/SKILL.md +101 -101
- package/skills/requesting-code-review/code-reviewer.md +168 -168
- package/skills/systematic-debugging/CREATION-LOG.md +119 -119
- package/skills/systematic-debugging/SKILL.md +296 -296
- package/skills/systematic-debugging/condition-based-waiting-example.ts +158 -158
- package/skills/systematic-debugging/condition-based-waiting.md +115 -115
- package/skills/systematic-debugging/defense-in-depth.md +122 -122
- package/skills/systematic-debugging/find-polluter.sh +63 -63
- package/skills/systematic-debugging/root-cause-tracing.md +169 -169
- package/skills/systematic-debugging/test-academic.md +14 -14
- package/skills/systematic-debugging/test-pressure-1.md +58 -58
- package/skills/systematic-debugging/test-pressure-2.md +68 -68
- package/skills/systematic-debugging/test-pressure-3.md +69 -69
- package/skills/test-driven-development/SKILL.md +20 -20
- package/skills/verification-before-completion/SKILL.md +154 -154
- package/skills/webapp-testing/SKILL.md +19 -19
|
@@ -1,117 +1,117 @@
|
|
|
1
|
-
# Agent DAG Decision Gate Dogfood Report Template
|
|
2
|
-
|
|
3
|
-
> Copy to `docs/reports/YYYY-MM-DD-agent-dag-decision-gate-dogfood.md` after each dogfood run.
|
|
4
|
-
|
|
5
|
-
## Run Metadata
|
|
6
|
-
|
|
7
|
-
| Field | Value |
|
|
8
|
-
|---|---|
|
|
9
|
-
| Date | YYYY-MM-DD |
|
|
10
|
-
| Topic | |
|
|
11
|
-
| DAG input | `<temp-dir>/<topic>-decision-gate-dag.json` |
|
|
12
|
-
| Run ID | |
|
|
13
|
-
| Run facts | `.harness/dag-runs/completed/<run-id>/` |
|
|
14
|
-
| Gate type | `acceptance-gate` |
|
|
15
|
-
| Decision node | `decision-pi` |
|
|
16
|
-
| Model | `gpt-5.5` via `executorModels.pi.HIGH` |
|
|
17
|
-
|
|
18
|
-
## Work Type
|
|
19
|
-
|
|
20
|
-
Choose one:
|
|
21
|
-
|
|
22
|
-
- [ ] Low-risk docs/template change — expected `auto-approve` or `approve-with-constraints`
|
|
23
|
-
- [ ] Runtime/loop-agent change — expected `request-revision`, `run-more-verification`, or `auto-approve`
|
|
24
|
-
- [ ] Contract/security/high-risk simulation — expected `escalate-to-human`
|
|
25
|
-
|
|
26
|
-
## Deterministic Evidence
|
|
27
|
-
|
|
28
|
-
| Evidence | Status | Notes |
|
|
29
|
-
|---|---|---|
|
|
30
|
-
| `verify-shell/result.summary.md` | verified / partial / missing | |
|
|
31
|
-
| `state.json` | verified / partial / missing | |
|
|
32
|
-
| git diff / changed file list | verified / partial / missing | |
|
|
33
|
-
| progress/report/artifacts | verified / partial / missing | |
|
|
34
|
-
|
|
35
|
-
## Decision Envelope Summary
|
|
36
|
-
|
|
37
|
-
Paste the **exact** ` ```DECISION_ENVELOPE_JSON ` fenced block from `decision-pi/result.summary.md` (info string must be `DECISION_ENVELOPE_JSON`, not `json`). Also record `decision`, `requiresHuman`, `nextAction`, and verified evidence count.
|
|
38
|
-
|
|
39
|
-
```DECISION_ENVELOPE_JSON
|
|
40
|
-
{
|
|
41
|
-
"schemaVersion": 1,
|
|
42
|
-
"gateType": "acceptance-gate",
|
|
43
|
-
"decisionScope": "dag-run",
|
|
44
|
-
"decision": "auto-approve",
|
|
45
|
-
"confidence": 0.88,
|
|
46
|
-
"riskLevel": "low",
|
|
47
|
-
"requiresHuman": false,
|
|
48
|
-
"nextAction": "continue",
|
|
49
|
-
"policyVersion": "agent-dag-decision-gate-v1",
|
|
50
|
-
"policyChecks": {
|
|
51
|
-
"mustEscalateFlags": [],
|
|
52
|
-
"evidenceComplete": true,
|
|
53
|
-
"allowedAutoApprove": true
|
|
54
|
-
},
|
|
55
|
-
"rationale": ["..."],
|
|
56
|
-
"evidence": [
|
|
57
|
-
{
|
|
58
|
-
"path": ".harness/dag-runs/completed/<run-id>/verify-shell/result.summary.md",
|
|
59
|
-
"kind": "shell-output",
|
|
60
|
-
"status": "verified",
|
|
61
|
-
"summary": "verification commands passed"
|
|
62
|
-
}
|
|
63
|
-
],
|
|
64
|
-
"blockingFindings": [],
|
|
65
|
-
"requiredRevisions": [],
|
|
66
|
-
"riskFlags": [],
|
|
67
|
-
"humanEscalation": null,
|
|
68
|
-
"audit": {
|
|
69
|
-
"runId": "<run-id>",
|
|
70
|
-
"nodeId": "decision-pi",
|
|
71
|
-
"model": "gpt-5.5"
|
|
72
|
-
}
|
|
73
|
-
}
|
|
74
|
-
```
|
|
75
|
-
|
|
76
|
-
## Quality Metrics
|
|
77
|
-
|
|
78
|
-
| Metric | Value | Notes |
|
|
79
|
-
|---|---:|---|
|
|
80
|
-
| `decisionLatencyMs` | | From node duration |
|
|
81
|
-
| `tokenCostEstimate` | | If available |
|
|
82
|
-
| `humanInterruptCount` | | Should be 0 for low-risk auto decisions |
|
|
83
|
-
| `falseEscalation` | yes / no | Escalated when policy allowed auto decision |
|
|
84
|
-
| `missedEscalation` | yes / no | Auto-approved when policy required human escalation |
|
|
85
|
-
| `revisionCaughtBeforeHuman` | yes / no | Gate requested revision before human involvement |
|
|
86
|
-
| `verifyPassAfterGate` | yes / no | Later verification passed after gate action |
|
|
87
|
-
| `evidenceCompleteness` | complete / partial / missing | Based on verified evidence count |
|
|
88
|
-
|
|
89
|
-
## Evaluation
|
|
90
|
-
|
|
91
|
-
### What the gate got right
|
|
92
|
-
|
|
93
|
-
-
|
|
94
|
-
|
|
95
|
-
### What the gate got wrong or over/under-weighted
|
|
96
|
-
|
|
97
|
-
-
|
|
98
|
-
|
|
99
|
-
### Prompt / policy adjustments needed
|
|
100
|
-
|
|
101
|
-
-
|
|
102
|
-
|
|
103
|
-
## Recommendation
|
|
104
|
-
|
|
105
|
-
Choose one:
|
|
106
|
-
|
|
107
|
-
- [ ] `proceed` — enough evidence to move toward parser/runtime work
|
|
108
|
-
- [ ] `needs-more-dogfood` — continue M1/M2 advisory runs
|
|
109
|
-
- [ ] `defer` — gate quality not yet good enough
|
|
110
|
-
|
|
111
|
-
Rationale:
|
|
112
|
-
|
|
113
|
-
-
|
|
114
|
-
|
|
115
|
-
## Follow-ups
|
|
116
|
-
|
|
117
|
-
-
|
|
1
|
+
# Agent DAG Decision Gate Dogfood Report Template
|
|
2
|
+
|
|
3
|
+
> Copy to `docs/reports/YYYY-MM-DD-agent-dag-decision-gate-dogfood.md` after each dogfood run.
|
|
4
|
+
|
|
5
|
+
## Run Metadata
|
|
6
|
+
|
|
7
|
+
| Field | Value |
|
|
8
|
+
|---|---|
|
|
9
|
+
| Date | YYYY-MM-DD |
|
|
10
|
+
| Topic | |
|
|
11
|
+
| DAG input | `<temp-dir>/<topic>-decision-gate-dag.json` |
|
|
12
|
+
| Run ID | |
|
|
13
|
+
| Run facts | `.harness/dag-runs/completed/<run-id>/` |
|
|
14
|
+
| Gate type | `acceptance-gate` |
|
|
15
|
+
| Decision node | `decision-pi` |
|
|
16
|
+
| Model | `gpt-5.5` via `executorModels.pi.HIGH` |
|
|
17
|
+
|
|
18
|
+
## Work Type
|
|
19
|
+
|
|
20
|
+
Choose one:
|
|
21
|
+
|
|
22
|
+
- [ ] Low-risk docs/template change — expected `auto-approve` or `approve-with-constraints`
|
|
23
|
+
- [ ] Runtime/loop-agent change — expected `request-revision`, `run-more-verification`, or `auto-approve`
|
|
24
|
+
- [ ] Contract/security/high-risk simulation — expected `escalate-to-human`
|
|
25
|
+
|
|
26
|
+
## Deterministic Evidence
|
|
27
|
+
|
|
28
|
+
| Evidence | Status | Notes |
|
|
29
|
+
|---|---|---|
|
|
30
|
+
| `verify-shell/result.summary.md` | verified / partial / missing | |
|
|
31
|
+
| `state.json` | verified / partial / missing | |
|
|
32
|
+
| git diff / changed file list | verified / partial / missing | |
|
|
33
|
+
| progress/report/artifacts | verified / partial / missing | |
|
|
34
|
+
|
|
35
|
+
## Decision Envelope Summary
|
|
36
|
+
|
|
37
|
+
Paste the **exact** ` ```DECISION_ENVELOPE_JSON ` fenced block from `decision-pi/result.summary.md` (info string must be `DECISION_ENVELOPE_JSON`, not `json`). Also record `decision`, `requiresHuman`, `nextAction`, and verified evidence count.
|
|
38
|
+
|
|
39
|
+
```DECISION_ENVELOPE_JSON
|
|
40
|
+
{
|
|
41
|
+
"schemaVersion": 1,
|
|
42
|
+
"gateType": "acceptance-gate",
|
|
43
|
+
"decisionScope": "dag-run",
|
|
44
|
+
"decision": "auto-approve",
|
|
45
|
+
"confidence": 0.88,
|
|
46
|
+
"riskLevel": "low",
|
|
47
|
+
"requiresHuman": false,
|
|
48
|
+
"nextAction": "continue",
|
|
49
|
+
"policyVersion": "agent-dag-decision-gate-v1",
|
|
50
|
+
"policyChecks": {
|
|
51
|
+
"mustEscalateFlags": [],
|
|
52
|
+
"evidenceComplete": true,
|
|
53
|
+
"allowedAutoApprove": true
|
|
54
|
+
},
|
|
55
|
+
"rationale": ["..."],
|
|
56
|
+
"evidence": [
|
|
57
|
+
{
|
|
58
|
+
"path": ".harness/dag-runs/completed/<run-id>/verify-shell/result.summary.md",
|
|
59
|
+
"kind": "shell-output",
|
|
60
|
+
"status": "verified",
|
|
61
|
+
"summary": "verification commands passed"
|
|
62
|
+
}
|
|
63
|
+
],
|
|
64
|
+
"blockingFindings": [],
|
|
65
|
+
"requiredRevisions": [],
|
|
66
|
+
"riskFlags": [],
|
|
67
|
+
"humanEscalation": null,
|
|
68
|
+
"audit": {
|
|
69
|
+
"runId": "<run-id>",
|
|
70
|
+
"nodeId": "decision-pi",
|
|
71
|
+
"model": "gpt-5.5"
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
## Quality Metrics
|
|
77
|
+
|
|
78
|
+
| Metric | Value | Notes |
|
|
79
|
+
|---|---:|---|
|
|
80
|
+
| `decisionLatencyMs` | | From node duration |
|
|
81
|
+
| `tokenCostEstimate` | | If available |
|
|
82
|
+
| `humanInterruptCount` | | Should be 0 for low-risk auto decisions |
|
|
83
|
+
| `falseEscalation` | yes / no | Escalated when policy allowed auto decision |
|
|
84
|
+
| `missedEscalation` | yes / no | Auto-approved when policy required human escalation |
|
|
85
|
+
| `revisionCaughtBeforeHuman` | yes / no | Gate requested revision before human involvement |
|
|
86
|
+
| `verifyPassAfterGate` | yes / no | Later verification passed after gate action |
|
|
87
|
+
| `evidenceCompleteness` | complete / partial / missing | Based on verified evidence count |
|
|
88
|
+
|
|
89
|
+
## Evaluation
|
|
90
|
+
|
|
91
|
+
### What the gate got right
|
|
92
|
+
|
|
93
|
+
-
|
|
94
|
+
|
|
95
|
+
### What the gate got wrong or over/under-weighted
|
|
96
|
+
|
|
97
|
+
-
|
|
98
|
+
|
|
99
|
+
### Prompt / policy adjustments needed
|
|
100
|
+
|
|
101
|
+
-
|
|
102
|
+
|
|
103
|
+
## Recommendation
|
|
104
|
+
|
|
105
|
+
Choose one:
|
|
106
|
+
|
|
107
|
+
- [ ] `proceed` — enough evidence to move toward parser/runtime work
|
|
108
|
+
- [ ] `needs-more-dogfood` — continue M1/M2 advisory runs
|
|
109
|
+
- [ ] `defer` — gate quality not yet good enough
|
|
110
|
+
|
|
111
|
+
Rationale:
|
|
112
|
+
|
|
113
|
+
-
|
|
114
|
+
|
|
115
|
+
## Follow-ups
|
|
116
|
+
|
|
117
|
+
-
|