@tea-agent/loop-agent 0.35.1-beta.0 → 0.35.1-beta.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +110 -108
- package/CHANGELOG.md +24 -26
- package/README.md +165 -165
- package/bin/agent-worker.js +0 -0
- package/bin/loop-agent.js +57 -21
- package/dist/application/task-lifecycle/advance.js +0 -1
- package/dist/build-stamp.json +6 -0
- package/dist/cli/program.js +2 -2
- package/dist/commands/cursor-prompt.js +6 -6
- package/dist/commands/init-upgrade.js +19 -351
- package/dist/commands/init.js +67 -14
- package/dist/commands/loop-benchmark.js +11 -11
- package/dist/commands/pi-reuse-benchmark.js +16 -16
- package/dist/commands/run-dag-progress.js +0 -14
- package/dist/commands/task-advance.js +3 -33
- package/dist/executors/dag-pi-executor.js +44 -0
- package/dist/shared/operator/capabilities.js +1 -38
- package/dist/shared/package-metadata.js +42 -0
- package/dist/sidecars/cursor-prompt/executor.js +1 -1
- package/dist/worker/console/chat/pi-runtime.js +25 -41
- package/dist/worker/console/chat/routes.js +4 -27
- package/dist/worker/console/operation-runner.js +0 -24
- package/dist/worker/console/operator-actions.js +0 -58
- package/dist/worker/console/static/assets/index-CvsQgALl.js +56 -0
- package/dist/worker/console/static/assets/{index-Dups4sSM.css → index-hJqCPs_g.css} +1 -1
- package/dist/worker/console/static/index.html +2 -2
- package/dist/worker/console/static-src/app/useRecoveryConsole.js +5 -0
- package/dist/worker/console/static-src/operator-chat/useChatSessions.js +2 -13
- package/dist/worker/console/static-src/operator-chat/useComposer.js +7 -30
- package/dist/worker/loop-agent/loop-agent-client.js +17 -3
- package/dist/worker/observability/read-model.js +20 -0
- package/dist/worker/observe/static/copy.js +67 -67
- package/dist/worker/observe/static/dag-layout.d.ts +36 -36
- package/dist/worker/observe/static/dom.js +220 -220
- package/dist/worker/observe/static/relations.js +133 -133
- package/dist/worker/observe/static/run-processing.js +148 -148
- package/dist/worker/observe/static/views/batch.js +227 -227
- package/dist/worker/observe/static/views/failures.js +143 -143
- package/dist/worker/observe/static/views/feature.js +492 -492
- package/dist/worker/observe/static/views/run.js +453 -453
- package/dist/worker/observe/static/views/shell.js +7 -7
- package/dist/worker/observe/static/views/timeline.js +163 -163
- package/dist/worker/preflight.js +2 -1
- package/dist/workflows/dag/backend-test-scenario-param.js +33 -23
- package/dist/workflows/dag/canvas-observer.js +275 -275
- package/dist/workflows/dag/contract-output-registry.js +14 -0
- package/dist/workflows/dag/contract-validator-registrations.js +8 -0
- package/dist/workflows/dag/dynamic-runtime/shared.js +9 -1
- package/dist/workflows/dag/frontend-implementation-contract.js +233 -39
- package/dist/workflows/dag/frontend-prewrite-gate.js +364 -61
- package/dist/workflows/dag/frontend-recovery-plan.js +73 -0
- package/dist/workflows/dag/frontend-recovery-root-manifest.js +123 -0
- package/dist/workflows/dag/frontend-recovery-run.js +539 -0
- package/dist/workflows/dag/frontend-repair.js +219 -18
- package/dist/workflows/dag/frontend-verification-trace.js +47 -32
- package/dist/workflows/dag/frontend-writer-recovery.js +106 -0
- package/dist/workflows/dag/frontend-writer-rollback.js +821 -0
- package/dist/workflows/dag/init-hybrid.js +41 -24
- package/dist/workflows/dag/node-execution.js +89 -0
- package/dist/workflows/dag/recovery-recommendation.js +58 -0
- package/dist/workflows/dag/runner.js +245 -11
- package/dist/workflows/dag/scheduler.js +257 -3
- package/dist/workflows/dag/types.js +130 -2
- package/docs/architecture/evolution.md +73 -73
- package/docs/architecture/system-overview.md +100 -100
- package/docs/architecture/worker-and-feature.md +122 -122
- package/docs/skills/README.md +7 -7
- package/docs/templates/adr.md +60 -60
- package/docs/templates/agent-dag-authority-surface-audit.prompt.md +94 -94
- package/docs/templates/agent-dag-decision-envelope.schema.json +213 -213
- package/docs/templates/agent-dag-decision-gate.prompt.md +246 -246
- package/docs/templates/agent-dag-process-supervisor.prompt.md +98 -98
- package/docs/templates/agent-dag-report.schema.json +473 -473
- package/docs/templates/agent-dag-review-verdict.prompt.md +68 -68
- package/docs/templates/backend-test-result.schema.json +99 -99
- package/docs/templates/evaluation/agents-map-slim-v1.md +87 -87
- package/docs/templates/evaluation/agents-map-verbose-v0.md +153 -153
- package/docs/templates/feature-spec.md +53 -53
- package/docs/templates/frontend-design-contract.md +42 -42
- package/docs/templates/frontend-eval/fixtures/failures/01-type-build-error.md +17 -17
- package/docs/templates/frontend-eval/fixtures/failures/02-unit-component-test-fail.md +16 -16
- package/docs/templates/frontend-eval/fixtures/failures/03-fixture-schema-drift.md +16 -16
- package/docs/templates/frontend-eval/fixtures/failures/04-missing-loading-empty-error-state.md +16 -16
- package/docs/templates/frontend-eval/fixtures/failures/05-forbidden-write-writeset-expansion.md +16 -16
- package/docs/templates/frontend-eval/fixtures/failures/06-unapproved-dependency-add.md +16 -16
- package/docs/templates/frontend-eval/fixtures/failures/07-mock-production-on.md +21 -21
- package/docs/templates/frontend-eval/fixtures/functional/01-simple-component-style.md +29 -29
- package/docs/templates/frontend-eval/fixtures/functional/02-form-validation.md +28 -28
- package/docs/templates/frontend-eval/fixtures/functional/03-list-detail-page.md +28 -28
- package/docs/templates/frontend-eval/fixtures/functional/04-api-mock.md +29 -29
- package/docs/templates/frontend-eval/fixtures/functional/05-permission-auth-gated-ui.md +27 -27
- package/docs/templates/frontend-eval/fixtures/functional/06-ssr-server-client-boundary.md +28 -28
- package/docs/templates/frontend-eval/fixtures/functional/07-shared-public-component-api.md +28 -28
- package/docs/templates/frontend-eval/fixtures/functional/08-pure-local-no-remote.md +27 -27
- package/docs/templates/frontend-eval/metrics.md +138 -138
- package/docs/templates/frontend-eval/smoke-targets.md +53 -53
- package/docs/templates/frontend-task-constraints.md +35 -35
- package/docs/templates/frontend-task-requirement.md +70 -70
- package/docs/templates/init-evolution-review.md +35 -35
- package/docs/templates/init-managed-agents.md +154 -156
- package/docs/templates/interactive-ui-round2-experiment.md +66 -66
- package/docs/templates/knowledge-graph-bootstrap-dag.json +118 -118
- package/docs/templates/knowledge-sync-dag.json +178 -178
- package/docs/templates/knowledge-sync-draft.schema.json +71 -71
- package/docs/templates/product-line/closeout.yaml +9 -9
- package/docs/templates/product-line/design.md +13 -13
- package/docs/templates/product-line/links.md +10 -10
- package/docs/templates/product-line/requirement.md +17 -17
- package/docs/templates/product-line/test-plan.md +7 -7
- package/docs/templates/project-start-checklist.md +9 -9
- package/docs/templates/qa-report.md +48 -48
- package/docs/templates/sprint-contract.md +29 -29
- package/docs/templates/worker-dogfood-evidence.md +80 -80
- package/docs/templates/worker-dogfood-setup.md +68 -68
- package/harness.json +2 -5
- package/package.json +2 -2
- package/scripts/kb-bootstrap-init-skeleton.sh +0 -0
- package/scripts/kb-graph-incremental-prepare.mjs +0 -0
- package/scripts/kb-graph-materialize.mjs +105 -105
- package/scripts/kb-graph-promote.mjs +164 -164
- package/scripts/kb-query.mjs +554 -554
- package/skills/agent-worker/SKILL.md +48 -48
- package/skills/agent-worker/references/agent-worker-operator.md +159 -159
- package/skills/ai-engineering-context/SKILL.md +48 -48
- package/skills/analyze-product-dependencies/scripts/test-validators.mjs +0 -0
- package/skills/analyze-product-dependencies/scripts/validate-api-documentation.mjs +0 -0
- package/skills/analyze-product-dependencies/scripts/validate-dependency-analysis.mjs +0 -0
- package/skills/analyze-product-dependencies/scripts/validate-product-requirement-input.mjs +0 -0
- package/skills/analyze-product-requirements/scripts/compute-source-identity.mjs +0 -0
- package/skills/analyze-product-requirements/scripts/test-validators.mjs +0 -0
- package/skills/analyze-product-requirements/scripts/validate-product-analysis.mjs +0 -0
- package/skills/analyze-product-requirements/scripts/validate-product-requirement.mjs +0 -0
- package/skills/analyze-product-requirements/scripts/validate-requirement-clarification.mjs +0 -0
- package/skills/browser-tools/browser-content.js +103 -103
- package/skills/browser-tools/browser-cookies.js +35 -35
- package/skills/browser-tools/browser-eval.js +53 -53
- package/skills/browser-tools/browser-hn-scraper.js +108 -108
- package/skills/browser-tools/browser-nav.js +44 -44
- package/skills/browser-tools/browser-pick.js +162 -162
- package/skills/browser-tools/browser-screenshot.js +34 -34
- package/skills/browser-tools/browser-start.js +86 -86
- package/skills/browser-tools/package-lock.json +2556 -2556
- package/skills/browser-tools/package.json +19 -19
- package/skills/code-review-core/SKILL.md +20 -20
- package/skills/codebase-scout/SKILL.md +19 -19
- package/skills/grill-me/SKILL.md +10 -10
- package/skills/local-jacoco-coverage/scripts/run-coverage-analysis.sh +0 -0
- package/skills/local-jacoco-coverage/scripts/start-jacoco-agent.sh +0 -0
- package/skills/loop-agent/SKILL.md +0 -1
- package/skills/loop-agent/references/command-reference.md +639 -641
- package/skills/loop-agent/references/docs-converge.md +126 -126
- package/skills/loop-agent/references/learned/README.md +21 -21
- package/skills/loop-agent/references/pi-prompt.md +23 -23
- package/skills/loop-agent/references/pi-subagent-assisted-mode.md +84 -84
- package/skills/playwright-cli/references/element-attributes.md +23 -23
- package/skills/playwright-cli/references/playwright-tests.md +39 -39
- package/skills/playwright-cli/references/request-mocking.md +87 -87
- package/skills/playwright-cli/references/running-code.md +241 -241
- package/skills/playwright-cli/references/session-management.md +225 -225
- package/skills/playwright-cli/references/storage-state.md +275 -275
- package/skills/playwright-cli/references/test-generation.md +433 -433
- package/skills/requesting-code-review/SKILL.md +101 -101
- package/skills/requesting-code-review/code-reviewer.md +168 -168
- package/skills/systematic-debugging/CREATION-LOG.md +119 -119
- package/skills/systematic-debugging/condition-based-waiting-example.ts +158 -158
- package/skills/systematic-debugging/condition-based-waiting.md +115 -115
- package/skills/systematic-debugging/defense-in-depth.md +122 -122
- package/skills/systematic-debugging/find-polluter.sh +63 -63
- package/skills/systematic-debugging/root-cause-tracing.md +169 -169
- package/skills/systematic-debugging/test-academic.md +14 -14
- package/skills/systematic-debugging/test-pressure-1.md +58 -58
- package/skills/systematic-debugging/test-pressure-2.md +68 -68
- package/skills/systematic-debugging/test-pressure-3.md +69 -69
- package/skills/using-git-worktrees/SKILL.md +215 -215
- package/skills/verification-before-completion/SKILL.md +154 -154
- package/skills/webapp-testing/SKILL.md +19 -19
- package/dist/worker/console/operation-wait.js +0 -241
- package/dist/worker/console/static/assets/index-SjjjZnV3.js +0 -56
- package/dist/worker/console/static-src/operator-chat/slash-palette-nav.js +0 -141
|
@@ -1,154 +1,154 @@
|
|
|
1
|
-
---
|
|
2
|
-
name: verification-before-completion
|
|
3
|
-
description: 在宣称 work complete、fixed 或 passing,或在 commit / 创建 PR 之前使用——须先运行 verification commands 并确认 output,再作任何 success claims;始终 evidence before assertions
|
|
4
|
-
---
|
|
5
|
-
|
|
6
|
-
# Verification Before Completion
|
|
7
|
-
|
|
8
|
-
## Overview
|
|
9
|
-
|
|
10
|
-
未经验证就宣称 work complete 是不诚实,不是效率。
|
|
11
|
-
|
|
12
|
-
**Core principle:** 始终 evidence before claims。
|
|
13
|
-
|
|
14
|
-
**违反本条字面即违反其精神。**
|
|
15
|
-
|
|
16
|
-
## The Iron Law
|
|
17
|
-
|
|
18
|
-
```
|
|
19
|
-
NO COMPLETION CLAIMS WITHOUT FRESH VERIFICATION EVIDENCE
|
|
20
|
-
```
|
|
21
|
-
|
|
22
|
-
若本 message 中尚未运行 verification command,不得宣称 passes。
|
|
23
|
-
|
|
24
|
-
## The Gate Function
|
|
25
|
-
|
|
26
|
-
```
|
|
27
|
-
BEFORE claiming any status or expressing satisfaction:
|
|
28
|
-
|
|
29
|
-
1. IDENTIFY: What command proves this claim?
|
|
30
|
-
2. RUN: Execute the FULL command (fresh, complete)
|
|
31
|
-
3. READ: Full output, check exit code, count failures
|
|
32
|
-
4. VERIFY: Does output confirm the claim?
|
|
33
|
-
- If NO: State actual status with evidence
|
|
34
|
-
- If YES: State claim WITH evidence
|
|
35
|
-
5. ONLY THEN: Make the claim
|
|
36
|
-
|
|
37
|
-
Skip any step = lying, not verifying
|
|
38
|
-
```
|
|
39
|
-
|
|
40
|
-
## Common Failures
|
|
41
|
-
|
|
42
|
-
| Claim | Requires | Not Sufficient |
|
|
43
|
-
|-------|----------|----------------|
|
|
44
|
-
| Tests pass | Test command output: 0 failures | Previous run, "should pass" |
|
|
45
|
-
| Linter clean | Linter output: 0 errors | Partial check, extrapolation |
|
|
46
|
-
| Build succeeds | Build command: exit 0 | Linter passing, logs look good |
|
|
47
|
-
| Bug fixed | Test original symptom: passes | Code changed, assumed fixed |
|
|
48
|
-
| Regression test works | Red-green cycle verified | Test passes once |
|
|
49
|
-
| Agent completed | VCS diff shows changes | Agent reports "success" |
|
|
50
|
-
| Requirements met | Line-by-line checklist | Tests passing |
|
|
51
|
-
| Harness integrity | `scripts/check-repo.sh` exit 0 | Files look correct |
|
|
52
|
-
| Harness CI | `scripts/ci.sh` exit 0 | Individual checks pass |
|
|
53
|
-
| Contract handoff | Handoff checklist completed + progress/report updated | "Should be fine"
|
|
54
|
-
|
|
55
|
-
## Red Flags - STOP
|
|
56
|
-
|
|
57
|
-
- 使用 "should"、"probably"、"seems to"
|
|
58
|
-
- 验证前表达满意("Great!"、"Perfect!"、"Done!" 等)
|
|
59
|
-
- 未验证就要 commit/push/PR
|
|
60
|
-
- 信任 agent success reports
|
|
61
|
-
- 依赖 partial verification
|
|
62
|
-
- 认为 "just this once"
|
|
63
|
-
- 疲惫想结束工作
|
|
64
|
-
- **任何未运行 verification 却暗示 success 的措辞**
|
|
65
|
-
|
|
66
|
-
## Rationalization Prevention
|
|
67
|
-
|
|
68
|
-
| Excuse | Reality |
|
|
69
|
-
|--------|---------|
|
|
70
|
-
| "Should work now" | RUN the verification |
|
|
71
|
-
| "I'm confident" | Confidence ≠ evidence |
|
|
72
|
-
| "Just this once" | No exceptions |
|
|
73
|
-
| "Linter passed" | Linter ≠ compiler |
|
|
74
|
-
| "Agent said success" | Verify independently |
|
|
75
|
-
| "I'm tired" | Exhaustion ≠ excuse |
|
|
76
|
-
| "Partial check is enough" | Partial proves nothing |
|
|
77
|
-
| "Different words so rule doesn't apply" | Spirit over letter |
|
|
78
|
-
|
|
79
|
-
## Key Patterns
|
|
80
|
-
|
|
81
|
-
**Tests:**
|
|
82
|
-
```
|
|
83
|
-
✅ [Run test command] [See: 34/34 pass] "All tests pass"
|
|
84
|
-
❌ "Should pass now" / "Looks correct"
|
|
85
|
-
```
|
|
86
|
-
|
|
87
|
-
**Regression tests (TDD Red-Green):**
|
|
88
|
-
```
|
|
89
|
-
✅ Write → Run (pass) → Revert fix → Run (MUST FAIL) → Restore → Run (pass)
|
|
90
|
-
❌ "I've written a regression test" (without red-green verification)
|
|
91
|
-
```
|
|
92
|
-
|
|
93
|
-
**Build:**
|
|
94
|
-
```
|
|
95
|
-
✅ [Run build] [See: exit 0] "Build passes"
|
|
96
|
-
❌ "Linter passed" (linter doesn't check compilation)
|
|
97
|
-
```
|
|
98
|
-
|
|
99
|
-
**Requirements:**
|
|
100
|
-
```
|
|
101
|
-
✅ Re-read plan → Create checklist → Verify each → Report gaps or completion
|
|
102
|
-
❌ "Tests pass, phase complete"
|
|
103
|
-
```
|
|
104
|
-
|
|
105
|
-
**Agent delegation:**
|
|
106
|
-
```
|
|
107
|
-
✅ Agent reports success → Check VCS diff → Verify changes → Report actual state
|
|
108
|
-
❌ Trust agent report
|
|
109
|
-
```
|
|
110
|
-
|
|
111
|
-
## Harness-Specific Verification
|
|
112
|
-
|
|
113
|
-
在 harness-governed repo 中工作(存在 `harness.json`)时:
|
|
114
|
-
|
|
115
|
-
- **Docs/structure changes** → `bash scripts/check-repo.sh`
|
|
116
|
-
- **Full-repo delivery** → `bash scripts/ci.sh`
|
|
117
|
-
- **Cross-platform changes** → 验证 OpenCode 与 Pi-Agent 两条路径
|
|
118
|
-
- **Contract changes** → 验证 contract docs 已更新 + tests 对齐
|
|
119
|
-
- **Handoff** → 宣称 complete 前运行 `handoff check`
|
|
120
|
-
|
|
121
|
-
完整 command 选择见项目 `ai_workspace/loop-agent/verification-matrix.md`。
|
|
122
|
-
|
|
123
|
-
## Why This Matters
|
|
124
|
-
|
|
125
|
-
来自 24 条 failure memories:
|
|
126
|
-
- human partner 说 "I don't believe you" — trust 已破裂
|
|
127
|
-
- Undefined functions 已 ship — 会 crash
|
|
128
|
-
- Missing requirements 已 ship — 功能不完整
|
|
129
|
-
- 虚假完成浪费时间 → redirect → rework
|
|
130
|
-
- 违反:"Honesty is a core value. If you lie, you'll be replaced."
|
|
131
|
-
|
|
132
|
-
## When To Apply
|
|
133
|
-
|
|
134
|
-
**在以下情况之前 ALWAYS:**
|
|
135
|
-
- 任何 success/completion claims 的变体
|
|
136
|
-
- 任何表达满意
|
|
137
|
-
- 任何关于 work state 的正面陈述
|
|
138
|
-
- Commit、PR creation、task completion
|
|
139
|
-
- 进入 next task
|
|
140
|
-
- 委派给 agents
|
|
141
|
-
|
|
142
|
-
**规则适用于:**
|
|
143
|
-
- 精确短语
|
|
144
|
-
- paraphrases 与同义词
|
|
145
|
-
- success 的暗示
|
|
146
|
-
- 任何暗示 completion/correctness 的沟通
|
|
147
|
-
|
|
148
|
-
## The Bottom Line
|
|
149
|
-
|
|
150
|
-
**Verification 无捷径。**
|
|
151
|
-
|
|
152
|
-
Run the command. Read the output. THEN claim the result.
|
|
153
|
-
|
|
154
|
-
This is non-negotiable.
|
|
1
|
+
---
|
|
2
|
+
name: verification-before-completion
|
|
3
|
+
description: 在宣称 work complete、fixed 或 passing,或在 commit / 创建 PR 之前使用——须先运行 verification commands 并确认 output,再作任何 success claims;始终 evidence before assertions
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Verification Before Completion
|
|
7
|
+
|
|
8
|
+
## Overview
|
|
9
|
+
|
|
10
|
+
未经验证就宣称 work complete 是不诚实,不是效率。
|
|
11
|
+
|
|
12
|
+
**Core principle:** 始终 evidence before claims。
|
|
13
|
+
|
|
14
|
+
**违反本条字面即违反其精神。**
|
|
15
|
+
|
|
16
|
+
## The Iron Law
|
|
17
|
+
|
|
18
|
+
```
|
|
19
|
+
NO COMPLETION CLAIMS WITHOUT FRESH VERIFICATION EVIDENCE
|
|
20
|
+
```
|
|
21
|
+
|
|
22
|
+
若本 message 中尚未运行 verification command,不得宣称 passes。
|
|
23
|
+
|
|
24
|
+
## The Gate Function
|
|
25
|
+
|
|
26
|
+
```
|
|
27
|
+
BEFORE claiming any status or expressing satisfaction:
|
|
28
|
+
|
|
29
|
+
1. IDENTIFY: What command proves this claim?
|
|
30
|
+
2. RUN: Execute the FULL command (fresh, complete)
|
|
31
|
+
3. READ: Full output, check exit code, count failures
|
|
32
|
+
4. VERIFY: Does output confirm the claim?
|
|
33
|
+
- If NO: State actual status with evidence
|
|
34
|
+
- If YES: State claim WITH evidence
|
|
35
|
+
5. ONLY THEN: Make the claim
|
|
36
|
+
|
|
37
|
+
Skip any step = lying, not verifying
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
## Common Failures
|
|
41
|
+
|
|
42
|
+
| Claim | Requires | Not Sufficient |
|
|
43
|
+
|-------|----------|----------------|
|
|
44
|
+
| Tests pass | Test command output: 0 failures | Previous run, "should pass" |
|
|
45
|
+
| Linter clean | Linter output: 0 errors | Partial check, extrapolation |
|
|
46
|
+
| Build succeeds | Build command: exit 0 | Linter passing, logs look good |
|
|
47
|
+
| Bug fixed | Test original symptom: passes | Code changed, assumed fixed |
|
|
48
|
+
| Regression test works | Red-green cycle verified | Test passes once |
|
|
49
|
+
| Agent completed | VCS diff shows changes | Agent reports "success" |
|
|
50
|
+
| Requirements met | Line-by-line checklist | Tests passing |
|
|
51
|
+
| Harness integrity | `scripts/check-repo.sh` exit 0 | Files look correct |
|
|
52
|
+
| Harness CI | `scripts/ci.sh` exit 0 | Individual checks pass |
|
|
53
|
+
| Contract handoff | Handoff checklist completed + progress/report updated | "Should be fine"
|
|
54
|
+
|
|
55
|
+
## Red Flags - STOP
|
|
56
|
+
|
|
57
|
+
- 使用 "should"、"probably"、"seems to"
|
|
58
|
+
- 验证前表达满意("Great!"、"Perfect!"、"Done!" 等)
|
|
59
|
+
- 未验证就要 commit/push/PR
|
|
60
|
+
- 信任 agent success reports
|
|
61
|
+
- 依赖 partial verification
|
|
62
|
+
- 认为 "just this once"
|
|
63
|
+
- 疲惫想结束工作
|
|
64
|
+
- **任何未运行 verification 却暗示 success 的措辞**
|
|
65
|
+
|
|
66
|
+
## Rationalization Prevention
|
|
67
|
+
|
|
68
|
+
| Excuse | Reality |
|
|
69
|
+
|--------|---------|
|
|
70
|
+
| "Should work now" | RUN the verification |
|
|
71
|
+
| "I'm confident" | Confidence ≠ evidence |
|
|
72
|
+
| "Just this once" | No exceptions |
|
|
73
|
+
| "Linter passed" | Linter ≠ compiler |
|
|
74
|
+
| "Agent said success" | Verify independently |
|
|
75
|
+
| "I'm tired" | Exhaustion ≠ excuse |
|
|
76
|
+
| "Partial check is enough" | Partial proves nothing |
|
|
77
|
+
| "Different words so rule doesn't apply" | Spirit over letter |
|
|
78
|
+
|
|
79
|
+
## Key Patterns
|
|
80
|
+
|
|
81
|
+
**Tests:**
|
|
82
|
+
```
|
|
83
|
+
✅ [Run test command] [See: 34/34 pass] "All tests pass"
|
|
84
|
+
❌ "Should pass now" / "Looks correct"
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
**Regression tests (TDD Red-Green):**
|
|
88
|
+
```
|
|
89
|
+
✅ Write → Run (pass) → Revert fix → Run (MUST FAIL) → Restore → Run (pass)
|
|
90
|
+
❌ "I've written a regression test" (without red-green verification)
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
**Build:**
|
|
94
|
+
```
|
|
95
|
+
✅ [Run build] [See: exit 0] "Build passes"
|
|
96
|
+
❌ "Linter passed" (linter doesn't check compilation)
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
**Requirements:**
|
|
100
|
+
```
|
|
101
|
+
✅ Re-read plan → Create checklist → Verify each → Report gaps or completion
|
|
102
|
+
❌ "Tests pass, phase complete"
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
**Agent delegation:**
|
|
106
|
+
```
|
|
107
|
+
✅ Agent reports success → Check VCS diff → Verify changes → Report actual state
|
|
108
|
+
❌ Trust agent report
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
## Harness-Specific Verification
|
|
112
|
+
|
|
113
|
+
在 harness-governed repo 中工作(存在 `harness.json`)时:
|
|
114
|
+
|
|
115
|
+
- **Docs/structure changes** → `bash scripts/check-repo.sh`
|
|
116
|
+
- **Full-repo delivery** → `bash scripts/ci.sh`
|
|
117
|
+
- **Cross-platform changes** → 验证 OpenCode 与 Pi-Agent 两条路径
|
|
118
|
+
- **Contract changes** → 验证 contract docs 已更新 + tests 对齐
|
|
119
|
+
- **Handoff** → 宣称 complete 前运行 `handoff check`
|
|
120
|
+
|
|
121
|
+
完整 command 选择见项目 `ai_workspace/loop-agent/verification-matrix.md`。
|
|
122
|
+
|
|
123
|
+
## Why This Matters
|
|
124
|
+
|
|
125
|
+
来自 24 条 failure memories:
|
|
126
|
+
- human partner 说 "I don't believe you" — trust 已破裂
|
|
127
|
+
- Undefined functions 已 ship — 会 crash
|
|
128
|
+
- Missing requirements 已 ship — 功能不完整
|
|
129
|
+
- 虚假完成浪费时间 → redirect → rework
|
|
130
|
+
- 违反:"Honesty is a core value. If you lie, you'll be replaced."
|
|
131
|
+
|
|
132
|
+
## When To Apply
|
|
133
|
+
|
|
134
|
+
**在以下情况之前 ALWAYS:**
|
|
135
|
+
- 任何 success/completion claims 的变体
|
|
136
|
+
- 任何表达满意
|
|
137
|
+
- 任何关于 work state 的正面陈述
|
|
138
|
+
- Commit、PR creation、task completion
|
|
139
|
+
- 进入 next task
|
|
140
|
+
- 委派给 agents
|
|
141
|
+
|
|
142
|
+
**规则适用于:**
|
|
143
|
+
- 精确短语
|
|
144
|
+
- paraphrases 与同义词
|
|
145
|
+
- success 的暗示
|
|
146
|
+
- 任何暗示 completion/correctness 的沟通
|
|
147
|
+
|
|
148
|
+
## The Bottom Line
|
|
149
|
+
|
|
150
|
+
**Verification 无捷径。**
|
|
151
|
+
|
|
152
|
+
Run the command. Read the output. THEN claim the result.
|
|
153
|
+
|
|
154
|
+
This is non-negotiable.
|
|
@@ -1,19 +1,19 @@
|
|
|
1
|
-
---
|
|
2
|
-
name: webapp-testing
|
|
3
|
-
description: 任务明确涉及 browser 渲染行为时,用于前端或本地 web UI 验证。
|
|
4
|
-
---
|
|
5
|
-
|
|
6
|
-
# Webapp Testing
|
|
7
|
-
|
|
8
|
-
仅当任务包含 browser UI 或本地 web app 时使用本 skill。
|
|
9
|
-
|
|
10
|
-
## 规则
|
|
11
|
-
|
|
12
|
-
- 优先使用项目现有的 dev server 与 test tooling。
|
|
13
|
-
- 当 visual 或 interaction 行为重要时,用 browser 或文档化的 UI test 命令验证渲染行为。
|
|
14
|
-
- UI 有变更时,检查 desktop 与 mobile 布局的 overlap、clipping、blank state。
|
|
15
|
-
- 默认不添加 networked services 或第三方 scan。
|
|
16
|
-
|
|
17
|
-
## Output
|
|
18
|
-
|
|
19
|
-
报告确切的 server 命令、URL、browser/test 命令与观察结果。
|
|
1
|
+
---
|
|
2
|
+
name: webapp-testing
|
|
3
|
+
description: 任务明确涉及 browser 渲染行为时,用于前端或本地 web UI 验证。
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Webapp Testing
|
|
7
|
+
|
|
8
|
+
仅当任务包含 browser UI 或本地 web app 时使用本 skill。
|
|
9
|
+
|
|
10
|
+
## 规则
|
|
11
|
+
|
|
12
|
+
- 优先使用项目现有的 dev server 与 test tooling。
|
|
13
|
+
- 当 visual 或 interaction 行为重要时,用 browser 或文档化的 UI test 命令验证渲染行为。
|
|
14
|
+
- UI 有变更时,检查 desktop 与 mobile 布局的 overlap、clipping、blank state。
|
|
15
|
+
- 默认不添加 networked services 或第三方 scan。
|
|
16
|
+
|
|
17
|
+
## Output
|
|
18
|
+
|
|
19
|
+
报告确切的 server 命令、URL、browser/test 命令与观察结果。
|
|
@@ -1,241 +0,0 @@
|
|
|
1
|
-
import { projectOperationEventSummary, projectOperationForChat, } from "./chat/chat-event-store.js";
|
|
2
|
-
import { isTerminalOperationState, } from "./operation-store.js";
|
|
3
|
-
/**
|
|
4
|
-
* P2: read-only event-driven long poll on the canonical operation event ring
|
|
5
|
-
* (2026-08-13 Operator Chat long-run supervision). Server contract bounds:
|
|
6
|
-
* the model-facing `maxWaitMs` is clamped to [minWaitMs, maxWaitMsBound];
|
|
7
|
-
* defaults allow the 60–180s model supervision cadence and tests may inject
|
|
8
|
-
* short bounds.
|
|
9
|
-
*/
|
|
10
|
-
export const DEFAULT_OPERATION_WAIT_MIN_MS = 60_000;
|
|
11
|
-
export const DEFAULT_OPERATION_WAIT_MAX_MS = 180_000;
|
|
12
|
-
export class OperationWaitError extends Error {
|
|
13
|
-
code;
|
|
14
|
-
constructor(code, message) {
|
|
15
|
-
super(message);
|
|
16
|
-
this.code = code;
|
|
17
|
-
this.name = "OperationWaitError";
|
|
18
|
-
}
|
|
19
|
-
}
|
|
20
|
-
function isFocusedOperation(operation) {
|
|
21
|
-
return (isTerminalOperationState(operation.state) ||
|
|
22
|
-
operation.state === "needs-reconcile");
|
|
23
|
-
}
|
|
24
|
-
function settledSummary(input) {
|
|
25
|
-
const projectedEvents = input.newEvents.length > 0
|
|
26
|
-
? input.newEvents.map(projectOperationEventSummary)
|
|
27
|
-
: [];
|
|
28
|
-
const nextSeq = input.newEvents.length > 0
|
|
29
|
-
? input.newEvents[input.newEvents.length - 1].seq
|
|
30
|
-
: input.afterSeq;
|
|
31
|
-
return {
|
|
32
|
-
operationId: input.operationId,
|
|
33
|
-
state: input.operation.state,
|
|
34
|
-
changed: input.changed,
|
|
35
|
-
timedOut: input.timedOut,
|
|
36
|
-
events: projectedEvents,
|
|
37
|
-
nextSeq,
|
|
38
|
-
operation: projectOperationForChat(input.operation),
|
|
39
|
-
};
|
|
40
|
-
}
|
|
41
|
-
/**
|
|
42
|
-
* Deterministic read-only wait over the canonical operation event stream
|
|
43
|
-
* (AC-003 / AC-004). Completion paths:
|
|
44
|
-
* - existing events (seq > afterSeq) → immediate (changed: true);
|
|
45
|
-
* - operation terminal/needs-reconcile with unconsumed events → immediate summary (changed: true);
|
|
46
|
-
* - operation terminal/needs-reconcile with no new events → immediate summary (changed: false);
|
|
47
|
-
* - EVENT_CURSOR_EXPIRED when the cursor fell out of the retained ring;
|
|
48
|
-
* - first subscribed event/state change → immediate settle;
|
|
49
|
-
* - maxWaitMs elapsed with no change → timedOut: true summary (not a failure).
|
|
50
|
-
*
|
|
51
|
-
* Race safety: the listener is registered BEFORE listFrom, closing the
|
|
52
|
-
* listFrom/subscribe gap; every path settles exactly once through a guarded
|
|
53
|
-
* `finish`, and listener + timer are always cleaned up on settle.
|
|
54
|
-
*/
|
|
55
|
-
export async function waitForOperationChange(input) {
|
|
56
|
-
const operationId = input.operationId?.trim();
|
|
57
|
-
if (!operationId) {
|
|
58
|
-
throw new OperationWaitError("INVALID_INPUT", "operationId is required");
|
|
59
|
-
}
|
|
60
|
-
if (!Number.isInteger(input.afterSeq) || input.afterSeq < 0) {
|
|
61
|
-
throw new OperationWaitError("INVALID_INPUT", "afterSeq must be a non-negative integer");
|
|
62
|
-
}
|
|
63
|
-
const minWaitMs = input.minWaitMs ?? DEFAULT_OPERATION_WAIT_MIN_MS;
|
|
64
|
-
const maxWaitMsBound = input.maxWaitMsBound ?? DEFAULT_OPERATION_WAIT_MAX_MS;
|
|
65
|
-
if (!Number.isFinite(minWaitMs) ||
|
|
66
|
-
!Number.isFinite(maxWaitMsBound) ||
|
|
67
|
-
minWaitMs < 0 ||
|
|
68
|
-
maxWaitMsBound < minWaitMs) {
|
|
69
|
-
throw new OperationWaitError("INVALID_INPUT", "invalid wait bounds");
|
|
70
|
-
}
|
|
71
|
-
const rawMaxWaitMs = input.maxWaitMs ?? maxWaitMsBound;
|
|
72
|
-
if (!Number.isFinite(rawMaxWaitMs) || rawMaxWaitMs < 0) {
|
|
73
|
-
throw new OperationWaitError("INVALID_INPUT", "maxWaitMs must be a non-negative number");
|
|
74
|
-
}
|
|
75
|
-
const maxWaitMs = Math.max(minWaitMs, Math.min(maxWaitMsBound, Math.floor(rawMaxWaitMs)));
|
|
76
|
-
const afterSeq = input.afterSeq;
|
|
77
|
-
const operation = await input.operations.get(operationId);
|
|
78
|
-
if (!operation) {
|
|
79
|
-
throw new OperationWaitError("NOT_FOUND", `operation not found: ${operationId}`);
|
|
80
|
-
}
|
|
81
|
-
const events = input.events;
|
|
82
|
-
if (isFocusedOperation(operation)) {
|
|
83
|
-
// Flush unconsumed events before the terminal summary (AC-002): the
|
|
84
|
-
// caller's cursor must advance past every retained canonical event.
|
|
85
|
-
const listed = events.listFrom(operationId, afterSeq);
|
|
86
|
-
if ("error" in listed) {
|
|
87
|
-
throw new OperationWaitError("EVENT_CURSOR_EXPIRED", "event cursor expired; re-read operation snapshot");
|
|
88
|
-
}
|
|
89
|
-
if (listed.events.length > 0) {
|
|
90
|
-
return settledSummary({
|
|
91
|
-
operationId,
|
|
92
|
-
operation,
|
|
93
|
-
afterSeq,
|
|
94
|
-
changed: true,
|
|
95
|
-
timedOut: false,
|
|
96
|
-
newEvents: listed.events,
|
|
97
|
-
});
|
|
98
|
-
}
|
|
99
|
-
return settledSummary({
|
|
100
|
-
operationId,
|
|
101
|
-
operation,
|
|
102
|
-
afterSeq,
|
|
103
|
-
changed: false,
|
|
104
|
-
timedOut: false,
|
|
105
|
-
newEvents: [],
|
|
106
|
-
});
|
|
107
|
-
}
|
|
108
|
-
return new Promise((resolve, reject) => {
|
|
109
|
-
let settled = false;
|
|
110
|
-
let timer;
|
|
111
|
-
let unsubscribe;
|
|
112
|
-
const cleanup = () => {
|
|
113
|
-
if (timer !== undefined) {
|
|
114
|
-
clearTimeout(timer);
|
|
115
|
-
timer = undefined;
|
|
116
|
-
}
|
|
117
|
-
if (unsubscribe) {
|
|
118
|
-
unsubscribe();
|
|
119
|
-
unsubscribe = undefined;
|
|
120
|
-
}
|
|
121
|
-
};
|
|
122
|
-
const finish = (result) => {
|
|
123
|
-
if (settled)
|
|
124
|
-
return;
|
|
125
|
-
settled = true;
|
|
126
|
-
cleanup();
|
|
127
|
-
if (result instanceof OperationWaitError)
|
|
128
|
-
reject(result);
|
|
129
|
-
else
|
|
130
|
-
resolve(result);
|
|
131
|
-
};
|
|
132
|
-
const settleWithEvent = (event) => {
|
|
133
|
-
// Future-cursor guard (AC-003): events at or below the caller's
|
|
134
|
-
// afterSeq are already consumed and must not settle this wait.
|
|
135
|
-
if (event.seq <= afterSeq)
|
|
136
|
-
return;
|
|
137
|
-
// Listener path: async re-read of the operation for a fresh summary.
|
|
138
|
-
void (async () => {
|
|
139
|
-
try {
|
|
140
|
-
const current = await input.operations.get(operationId);
|
|
141
|
-
finish(settledSummary({
|
|
142
|
-
operationId,
|
|
143
|
-
operation: current ?? operation,
|
|
144
|
-
afterSeq,
|
|
145
|
-
changed: true,
|
|
146
|
-
timedOut: false,
|
|
147
|
-
newEvents: [event],
|
|
148
|
-
}));
|
|
149
|
-
}
|
|
150
|
-
catch (error) {
|
|
151
|
-
finish(error instanceof OperationWaitError
|
|
152
|
-
? error
|
|
153
|
-
: new OperationWaitError("INVALID_INPUT", error instanceof Error ? error.message : String(error)));
|
|
154
|
-
}
|
|
155
|
-
})();
|
|
156
|
-
};
|
|
157
|
-
// Terminal recheck before subscribing: events landing during this await
|
|
158
|
-
// are still in the ring and are caught by listFrom below.
|
|
159
|
-
void (async () => {
|
|
160
|
-
try {
|
|
161
|
-
const current = await input.operations.get(operationId);
|
|
162
|
-
if (current && isFocusedOperation(current)) {
|
|
163
|
-
// Same terminal flush as the initial path (AC-002): events
|
|
164
|
-
// landing during the await are still in the retained ring.
|
|
165
|
-
const listed = events.listFrom(operationId, afterSeq);
|
|
166
|
-
if ("error" in listed) {
|
|
167
|
-
finish(new OperationWaitError("EVENT_CURSOR_EXPIRED", "event cursor expired; re-read operation snapshot"));
|
|
168
|
-
return;
|
|
169
|
-
}
|
|
170
|
-
if (listed.events.length > 0) {
|
|
171
|
-
finish(settledSummary({
|
|
172
|
-
operationId,
|
|
173
|
-
operation: current,
|
|
174
|
-
afterSeq,
|
|
175
|
-
changed: true,
|
|
176
|
-
timedOut: false,
|
|
177
|
-
newEvents: listed.events,
|
|
178
|
-
}));
|
|
179
|
-
return;
|
|
180
|
-
}
|
|
181
|
-
finish(settledSummary({
|
|
182
|
-
operationId,
|
|
183
|
-
operation: current,
|
|
184
|
-
afterSeq,
|
|
185
|
-
changed: false,
|
|
186
|
-
timedOut: false,
|
|
187
|
-
newEvents: [],
|
|
188
|
-
}));
|
|
189
|
-
return;
|
|
190
|
-
}
|
|
191
|
-
// Subscribe BEFORE listFrom: any event appended after this point
|
|
192
|
-
// reaches the listener, closing the listFrom/subscribe race.
|
|
193
|
-
unsubscribe = events.subscribe(operationId, settleWithEvent);
|
|
194
|
-
const listed = events.listFrom(operationId, afterSeq);
|
|
195
|
-
if ("error" in listed) {
|
|
196
|
-
finish(new OperationWaitError("EVENT_CURSOR_EXPIRED", "event cursor expired; re-read operation snapshot"));
|
|
197
|
-
return;
|
|
198
|
-
}
|
|
199
|
-
if (listed.events.length > 0) {
|
|
200
|
-
finish(settledSummary({
|
|
201
|
-
operationId,
|
|
202
|
-
operation: current ?? operation,
|
|
203
|
-
afterSeq,
|
|
204
|
-
changed: true,
|
|
205
|
-
timedOut: false,
|
|
206
|
-
newEvents: listed.events,
|
|
207
|
-
}));
|
|
208
|
-
return;
|
|
209
|
-
}
|
|
210
|
-
// No events: arm the bounded wait; the listener settles on the
|
|
211
|
-
// first new event/state change, the timer settles on timeout.
|
|
212
|
-
timer = setTimeout(() => {
|
|
213
|
-
void (async () => {
|
|
214
|
-
try {
|
|
215
|
-
const latest = await input.operations.get(operationId);
|
|
216
|
-
finish(settledSummary({
|
|
217
|
-
operationId,
|
|
218
|
-
operation: latest ?? operation,
|
|
219
|
-
afterSeq,
|
|
220
|
-
changed: false,
|
|
221
|
-
timedOut: true,
|
|
222
|
-
newEvents: [],
|
|
223
|
-
}));
|
|
224
|
-
}
|
|
225
|
-
catch (error) {
|
|
226
|
-
finish(error instanceof OperationWaitError
|
|
227
|
-
? error
|
|
228
|
-
: new OperationWaitError("INVALID_INPUT", error instanceof Error ? error.message : String(error)));
|
|
229
|
-
}
|
|
230
|
-
})();
|
|
231
|
-
}, maxWaitMs);
|
|
232
|
-
timer.unref?.();
|
|
233
|
-
}
|
|
234
|
-
catch (error) {
|
|
235
|
-
finish(error instanceof OperationWaitError
|
|
236
|
-
? error
|
|
237
|
-
: new OperationWaitError("INVALID_INPUT", error instanceof Error ? error.message : String(error)));
|
|
238
|
-
}
|
|
239
|
-
})();
|
|
240
|
-
});
|
|
241
|
-
}
|