@tea-agent/loop-agent 0.35.0-beta.2 → 0.35.1-beta.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (154) hide show
  1. package/AGENTS.md +108 -108
  2. package/CHANGELOG.md +55 -4
  3. package/README.md +165 -165
  4. package/bin/agent-worker.js +0 -0
  5. package/bin/loop-agent.js +21 -21
  6. package/dist/application/task-lifecycle/advance.js +1 -0
  7. package/dist/commands/cursor-prompt.js +6 -6
  8. package/dist/commands/init-upgrade.js +351 -19
  9. package/dist/commands/init.js +14 -67
  10. package/dist/commands/loop-benchmark.js +11 -11
  11. package/dist/commands/pi-reuse-benchmark.js +16 -16
  12. package/dist/commands/run-dag-progress.js +14 -0
  13. package/dist/commands/task-advance.js +33 -3
  14. package/dist/shared/operator/capabilities.js +38 -1
  15. package/dist/sidecars/cursor-prompt/executor.js +1 -1
  16. package/dist/worker/console/chat/pi-runtime.js +41 -25
  17. package/dist/worker/console/chat/routes.js +27 -4
  18. package/dist/worker/console/operation-runner.js +24 -0
  19. package/dist/worker/console/operation-wait.js +241 -0
  20. package/dist/worker/console/operator-actions.js +58 -0
  21. package/dist/worker/console/static/assets/{index-hJqCPs_g.css → index-Dups4sSM.css} +1 -1
  22. package/dist/worker/console/static/assets/index-SjjjZnV3.js +56 -0
  23. package/dist/worker/console/static/index.html +2 -2
  24. package/dist/worker/console/static-src/operator-chat/slash-palette-nav.js +141 -0
  25. package/dist/worker/console/static-src/operator-chat/useChatSessions.js +13 -2
  26. package/dist/worker/console/static-src/operator-chat/useComposer.js +30 -7
  27. package/dist/worker/observe/static/copy.js +67 -67
  28. package/dist/worker/observe/static/dag-layout.d.ts +36 -36
  29. package/dist/worker/observe/static/dom.js +220 -220
  30. package/dist/worker/observe/static/relations.js +133 -133
  31. package/dist/worker/observe/static/run-processing.js +148 -148
  32. package/dist/worker/observe/static/views/batch.js +227 -227
  33. package/dist/worker/observe/static/views/failures.js +143 -143
  34. package/dist/worker/observe/static/views/feature.js +492 -492
  35. package/dist/worker/observe/static/views/run.js +453 -453
  36. package/dist/worker/observe/static/views/shell.js +7 -7
  37. package/dist/worker/observe/static/views/timeline.js +163 -163
  38. package/dist/workflows/dag/canvas-observer.js +275 -275
  39. package/dist/workflows/dag/frontend-prewrite-gate.js +9 -1
  40. package/dist/workflows/dag/init-hybrid.js +2 -0
  41. package/docs/architecture/evolution.md +73 -73
  42. package/docs/architecture/system-overview.md +100 -100
  43. package/docs/architecture/worker-and-feature.md +122 -122
  44. package/docs/skills/README.md +7 -7
  45. package/docs/templates/adr.md +60 -60
  46. package/docs/templates/agent-dag-authority-surface-audit.prompt.md +94 -94
  47. package/docs/templates/agent-dag-decision-envelope.schema.json +213 -213
  48. package/docs/templates/agent-dag-decision-gate.prompt.md +246 -246
  49. package/docs/templates/agent-dag-process-supervisor.prompt.md +98 -98
  50. package/docs/templates/agent-dag-report.schema.json +473 -473
  51. package/docs/templates/agent-dag-review-verdict.prompt.md +68 -68
  52. package/docs/templates/backend-test-result.schema.json +99 -99
  53. package/docs/templates/evaluation/agents-map-slim-v1.md +87 -87
  54. package/docs/templates/evaluation/agents-map-verbose-v0.md +153 -153
  55. package/docs/templates/feature-spec.md +53 -53
  56. package/docs/templates/frontend-design-contract.md +42 -42
  57. package/docs/templates/frontend-eval/fixtures/failures/01-type-build-error.md +17 -17
  58. package/docs/templates/frontend-eval/fixtures/failures/02-unit-component-test-fail.md +16 -16
  59. package/docs/templates/frontend-eval/fixtures/failures/03-fixture-schema-drift.md +16 -16
  60. package/docs/templates/frontend-eval/fixtures/failures/04-missing-loading-empty-error-state.md +16 -16
  61. package/docs/templates/frontend-eval/fixtures/failures/05-forbidden-write-writeset-expansion.md +16 -16
  62. package/docs/templates/frontend-eval/fixtures/failures/06-unapproved-dependency-add.md +16 -16
  63. package/docs/templates/frontend-eval/fixtures/failures/07-mock-production-on.md +21 -21
  64. package/docs/templates/frontend-eval/fixtures/functional/01-simple-component-style.md +29 -29
  65. package/docs/templates/frontend-eval/fixtures/functional/02-form-validation.md +28 -28
  66. package/docs/templates/frontend-eval/fixtures/functional/03-list-detail-page.md +28 -28
  67. package/docs/templates/frontend-eval/fixtures/functional/04-api-mock.md +29 -29
  68. package/docs/templates/frontend-eval/fixtures/functional/05-permission-auth-gated-ui.md +27 -27
  69. package/docs/templates/frontend-eval/fixtures/functional/06-ssr-server-client-boundary.md +28 -28
  70. package/docs/templates/frontend-eval/fixtures/functional/07-shared-public-component-api.md +28 -28
  71. package/docs/templates/frontend-eval/fixtures/functional/08-pure-local-no-remote.md +27 -27
  72. package/docs/templates/frontend-eval/metrics.md +138 -138
  73. package/docs/templates/frontend-eval/smoke-targets.md +53 -53
  74. package/docs/templates/frontend-task-constraints.md +35 -35
  75. package/docs/templates/frontend-task-requirement.md +70 -70
  76. package/docs/templates/init-evolution-review.md +35 -35
  77. package/docs/templates/init-managed-agents.md +156 -154
  78. package/docs/templates/interactive-ui-round2-experiment.md +66 -66
  79. package/docs/templates/knowledge-graph-bootstrap-dag.json +118 -118
  80. package/docs/templates/knowledge-sync-dag.json +178 -178
  81. package/docs/templates/knowledge-sync-draft.schema.json +71 -71
  82. package/docs/templates/product-line/closeout.yaml +9 -9
  83. package/docs/templates/product-line/design.md +13 -13
  84. package/docs/templates/product-line/links.md +10 -10
  85. package/docs/templates/product-line/requirement.md +17 -17
  86. package/docs/templates/product-line/test-plan.md +7 -7
  87. package/docs/templates/project-start-checklist.md +9 -9
  88. package/docs/templates/qa-report.md +48 -48
  89. package/docs/templates/sprint-contract.md +29 -29
  90. package/docs/templates/worker-dogfood-evidence.md +80 -80
  91. package/docs/templates/worker-dogfood-setup.md +68 -68
  92. package/harness.json +5 -2
  93. package/package.json +1 -1
  94. package/scripts/kb-bootstrap-init-skeleton.sh +0 -0
  95. package/scripts/kb-graph-incremental-prepare.mjs +0 -0
  96. package/scripts/kb-graph-materialize.mjs +105 -105
  97. package/scripts/kb-graph-promote.mjs +164 -164
  98. package/scripts/kb-query.mjs +554 -554
  99. package/skills/agent-worker/SKILL.md +48 -48
  100. package/skills/agent-worker/references/agent-worker-operator.md +159 -159
  101. package/skills/ai-engineering-context/SKILL.md +48 -48
  102. package/skills/analyze-product-dependencies/scripts/test-validators.mjs +0 -0
  103. package/skills/analyze-product-dependencies/scripts/validate-api-documentation.mjs +0 -0
  104. package/skills/analyze-product-dependencies/scripts/validate-dependency-analysis.mjs +0 -0
  105. package/skills/analyze-product-dependencies/scripts/validate-product-requirement-input.mjs +0 -0
  106. package/skills/analyze-product-requirements/scripts/compute-source-identity.mjs +0 -0
  107. package/skills/analyze-product-requirements/scripts/test-validators.mjs +0 -0
  108. package/skills/analyze-product-requirements/scripts/validate-product-analysis.mjs +0 -0
  109. package/skills/analyze-product-requirements/scripts/validate-product-requirement.mjs +0 -0
  110. package/skills/analyze-product-requirements/scripts/validate-requirement-clarification.mjs +0 -0
  111. package/skills/browser-tools/browser-content.js +103 -103
  112. package/skills/browser-tools/browser-cookies.js +35 -35
  113. package/skills/browser-tools/browser-eval.js +53 -53
  114. package/skills/browser-tools/browser-hn-scraper.js +108 -108
  115. package/skills/browser-tools/browser-nav.js +44 -44
  116. package/skills/browser-tools/browser-pick.js +162 -162
  117. package/skills/browser-tools/browser-screenshot.js +34 -34
  118. package/skills/browser-tools/browser-start.js +86 -86
  119. package/skills/browser-tools/package-lock.json +2556 -2556
  120. package/skills/browser-tools/package.json +19 -19
  121. package/skills/code-review-core/SKILL.md +20 -20
  122. package/skills/codebase-scout/SKILL.md +19 -19
  123. package/skills/grill-me/SKILL.md +10 -10
  124. package/skills/local-jacoco-coverage/scripts/run-coverage-analysis.sh +0 -0
  125. package/skills/local-jacoco-coverage/scripts/start-jacoco-agent.sh +0 -0
  126. package/skills/loop-agent/SKILL.md +1 -0
  127. package/skills/loop-agent/references/command-reference.md +641 -639
  128. package/skills/loop-agent/references/docs-converge.md +126 -126
  129. package/skills/loop-agent/references/learned/README.md +21 -21
  130. package/skills/loop-agent/references/pi-prompt.md +23 -23
  131. package/skills/loop-agent/references/pi-subagent-assisted-mode.md +84 -84
  132. package/skills/playwright-cli/references/element-attributes.md +23 -23
  133. package/skills/playwright-cli/references/playwright-tests.md +39 -39
  134. package/skills/playwright-cli/references/request-mocking.md +87 -87
  135. package/skills/playwright-cli/references/running-code.md +241 -241
  136. package/skills/playwright-cli/references/session-management.md +225 -225
  137. package/skills/playwright-cli/references/storage-state.md +275 -275
  138. package/skills/playwright-cli/references/test-generation.md +433 -433
  139. package/skills/requesting-code-review/SKILL.md +101 -101
  140. package/skills/requesting-code-review/code-reviewer.md +168 -168
  141. package/skills/systematic-debugging/CREATION-LOG.md +119 -119
  142. package/skills/systematic-debugging/condition-based-waiting-example.ts +158 -158
  143. package/skills/systematic-debugging/condition-based-waiting.md +115 -115
  144. package/skills/systematic-debugging/defense-in-depth.md +122 -122
  145. package/skills/systematic-debugging/find-polluter.sh +63 -63
  146. package/skills/systematic-debugging/root-cause-tracing.md +169 -169
  147. package/skills/systematic-debugging/test-academic.md +14 -14
  148. package/skills/systematic-debugging/test-pressure-1.md +58 -58
  149. package/skills/systematic-debugging/test-pressure-2.md +68 -68
  150. package/skills/systematic-debugging/test-pressure-3.md +69 -69
  151. package/skills/using-git-worktrees/SKILL.md +215 -215
  152. package/skills/verification-before-completion/SKILL.md +154 -154
  153. package/skills/webapp-testing/SKILL.md +19 -19
  154. package/dist/worker/console/static/assets/index-fsjzREob.js +0 -56
@@ -1,119 +1,119 @@
1
- # Creation Log: Systematic Debugging Skill
2
-
3
- 提取、结构化与 bulletproofing 关键 skill 的 reference example。
4
-
5
- ## Source Material
6
-
7
- 从 `~/.claude/CLAUDE.md` 提取 debugging framework:
8
- - 4-phase systematic process(Investigation → Pattern Analysis → Hypothesis → Implementation)
9
- - Core mandate:ALWAYS find root cause,NEVER fix symptoms
10
- - 设计以 resist time pressure 与 rationalization 的规则
11
-
12
- ## Extraction Decisions
13
-
14
- **What to include:**
15
- - 完整 4-phase framework 及所有 rules
16
- - Anti-shortcuts("NEVER fix symptom"、"STOP and re-analyze")
17
- - Pressure-resistant language("even if faster"、"even if I seem in a hurry")
18
- - 各 phase 的 concrete steps
19
-
20
- **What to leave out:**
21
- - Project-specific context
22
- - 同一 rule 的 repetitive variations
23
- - Narrative explanations(condensed 为 principles)
24
-
25
- ## Structure Following skill-creation/SKILL.md
26
-
27
- 1. **Rich when_to_use** — 含 symptoms 与 anti-patterns
28
- 2. **Type: technique** — 带 steps 的 concrete process
29
- 3. **Keywords** — "root cause"、"symptom"、"workaround"、"debugging"、"investigation"
30
- 4. **Flowchart** — "fix failed" 决策点 → re-analyze vs add more fixes
31
- 5. **Phase-by-phase breakdown** — Scannable checklist format
32
- 6. **Anti-patterns section** — 什么 NOT to do(对本 skill 关键)
33
-
34
- ## Bulletproofing Elements
35
-
36
- Framework 设计以 resist rationalization under pressure:
37
-
38
- ### Language Choices
39
- - "ALWAYS" / "NEVER"(非 "should" / "try to")
40
- - "even if faster" / "even if I seem in a hurry"
41
- - "STOP and re-analyze"(explicit pause)
42
- - "Don't skip past"(捕获 actual behavior)
43
-
44
- ### Structural Defenses
45
- - **Phase 1 required** — 不能 skip to implementation
46
- - **Single hypothesis rule** — 强制思考,防止 shotgun fixes
47
- - **Explicit failure mode** — "IF your first fix doesn't work" 及 mandatory action
48
- - **Anti-patterns section** — 展示 shortcuts 的确切样子
49
-
50
- ### Redundancy
51
- - Root cause mandate 在 overview + when_to_use + Phase 1 + implementation rules
52
- - "NEVER fix symptom" 在不同 contexts 出现 4 次
53
- - 各 phase 有 explicit "don't skip" guidance
54
-
55
- ## Testing Approach
56
-
57
- 按 .agents/skills/meta/testing-skills-with-subagents 创建 4 个 validation tests:
58
-
59
- ### Test 1: Academic Context (No Pressure)
60
- - Simple bug,无 time pressure
61
- - **Result:** Perfect compliance,complete investigation
62
-
63
- ### Test 2: Time Pressure + Obvious Quick Fix
64
- - User "in a hurry",symptom fix 看起来 easy
65
- - **Result:** Resisted shortcut,followed full process,found real root cause
66
-
67
- ### Test 3: Complex System + Uncertainty
68
- - Multi-layer failure, unclear 能否 find root cause
69
- - **Result:** Systematic investigation,traced through all layers,found source
70
-
71
- ### Test 4: Failed First Fix
72
- - Hypothesis 无效,temptation 加 more fixes
73
- - **Result:** Stopped,re-analyzed,formed new hypothesis(no shotgun)
74
-
75
- **All tests passed.** No rationalizations found.
76
-
77
- ## Iterations
78
-
79
- ### Initial Version
80
- - Complete 4-phase framework
81
- - Anti-patterns section
82
- - Flowchart for "fix failed" decision
83
-
84
- ### Enhancement 1: TDD Reference
85
- - Added link to .agents/skills/testing/test-driven-development
86
- - Note explaining TDD's "simplest code" ≠ debugging's "root cause"
87
- - Prevents confusion between methodologies
88
-
89
- ## Final Outcome
90
-
91
- Bulletproof skill that:
92
- - ✅ Clearly mandates root cause investigation
93
- - ✅ Resists time pressure rationalization
94
- - ✅ Provides concrete steps for each phase
95
- - ✅ Shows anti-patterns explicitly
96
- - ✅ Tested under multiple pressure scenarios
97
- - ✅ Clarifies relationship to TDD
98
- - ✅ Ready for use
99
-
100
- ## Key Insight
101
-
102
- **Most important bulletproofing:** Anti-patterns section 展示 moment 里 feel justified 的 exact shortcuts。当 Claude 想 "I'll just add this one quick fix",看到 listed as wrong 的 exact pattern 产生 cognitive friction。
103
-
104
- ## Usage Example
105
-
106
- 遇到 bug 时:
107
- 1. Load skill: .agents/skills/debugging/systematic-debugging
108
- 2. Read overview (10 sec) — reminded of mandate
109
- 3. Follow Phase 1 checklist — forced investigation
110
- 4. If tempted to skip — see anti-pattern,stop
111
- 5. Complete all phases — root cause found
112
-
113
- **Time investment:** 5-10 minutes
114
- **Time saved:** Hours of symptom-whack-a-mole
115
-
116
- ---
117
-
118
- *Created: 2025-10-03*
119
- *Purpose: Reference example for skill extraction and bulletproofing*
1
+ # Creation Log: Systematic Debugging Skill
2
+
3
+ 提取、结构化与 bulletproofing 关键 skill 的 reference example。
4
+
5
+ ## Source Material
6
+
7
+ 从 `~/.claude/CLAUDE.md` 提取 debugging framework:
8
+ - 4-phase systematic process(Investigation → Pattern Analysis → Hypothesis → Implementation)
9
+ - Core mandate:ALWAYS find root cause,NEVER fix symptoms
10
+ - 设计以 resist time pressure 与 rationalization 的规则
11
+
12
+ ## Extraction Decisions
13
+
14
+ **What to include:**
15
+ - 完整 4-phase framework 及所有 rules
16
+ - Anti-shortcuts("NEVER fix symptom"、"STOP and re-analyze")
17
+ - Pressure-resistant language("even if faster"、"even if I seem in a hurry")
18
+ - 各 phase 的 concrete steps
19
+
20
+ **What to leave out:**
21
+ - Project-specific context
22
+ - 同一 rule 的 repetitive variations
23
+ - Narrative explanations(condensed 为 principles)
24
+
25
+ ## Structure Following skill-creation/SKILL.md
26
+
27
+ 1. **Rich when_to_use** — 含 symptoms 与 anti-patterns
28
+ 2. **Type: technique** — 带 steps 的 concrete process
29
+ 3. **Keywords** — "root cause"、"symptom"、"workaround"、"debugging"、"investigation"
30
+ 4. **Flowchart** — "fix failed" 决策点 → re-analyze vs add more fixes
31
+ 5. **Phase-by-phase breakdown** — Scannable checklist format
32
+ 6. **Anti-patterns section** — 什么 NOT to do(对本 skill 关键)
33
+
34
+ ## Bulletproofing Elements
35
+
36
+ Framework 设计以 resist rationalization under pressure:
37
+
38
+ ### Language Choices
39
+ - "ALWAYS" / "NEVER"(非 "should" / "try to")
40
+ - "even if faster" / "even if I seem in a hurry"
41
+ - "STOP and re-analyze"(explicit pause)
42
+ - "Don't skip past"(捕获 actual behavior)
43
+
44
+ ### Structural Defenses
45
+ - **Phase 1 required** — 不能 skip to implementation
46
+ - **Single hypothesis rule** — 强制思考,防止 shotgun fixes
47
+ - **Explicit failure mode** — "IF your first fix doesn't work" 及 mandatory action
48
+ - **Anti-patterns section** — 展示 shortcuts 的确切样子
49
+
50
+ ### Redundancy
51
+ - Root cause mandate 在 overview + when_to_use + Phase 1 + implementation rules
52
+ - "NEVER fix symptom" 在不同 contexts 出现 4 次
53
+ - 各 phase 有 explicit "don't skip" guidance
54
+
55
+ ## Testing Approach
56
+
57
+ 按 .agents/skills/meta/testing-skills-with-subagents 创建 4 个 validation tests:
58
+
59
+ ### Test 1: Academic Context (No Pressure)
60
+ - Simple bug,无 time pressure
61
+ - **Result:** Perfect compliance,complete investigation
62
+
63
+ ### Test 2: Time Pressure + Obvious Quick Fix
64
+ - User "in a hurry",symptom fix 看起来 easy
65
+ - **Result:** Resisted shortcut,followed full process,found real root cause
66
+
67
+ ### Test 3: Complex System + Uncertainty
68
+ - Multi-layer failure, unclear 能否 find root cause
69
+ - **Result:** Systematic investigation,traced through all layers,found source
70
+
71
+ ### Test 4: Failed First Fix
72
+ - Hypothesis 无效,temptation 加 more fixes
73
+ - **Result:** Stopped,re-analyzed,formed new hypothesis(no shotgun)
74
+
75
+ **All tests passed.** No rationalizations found.
76
+
77
+ ## Iterations
78
+
79
+ ### Initial Version
80
+ - Complete 4-phase framework
81
+ - Anti-patterns section
82
+ - Flowchart for "fix failed" decision
83
+
84
+ ### Enhancement 1: TDD Reference
85
+ - Added link to .agents/skills/testing/test-driven-development
86
+ - Note explaining TDD's "simplest code" ≠ debugging's "root cause"
87
+ - Prevents confusion between methodologies
88
+
89
+ ## Final Outcome
90
+
91
+ Bulletproof skill that:
92
+ - ✅ Clearly mandates root cause investigation
93
+ - ✅ Resists time pressure rationalization
94
+ - ✅ Provides concrete steps for each phase
95
+ - ✅ Shows anti-patterns explicitly
96
+ - ✅ Tested under multiple pressure scenarios
97
+ - ✅ Clarifies relationship to TDD
98
+ - ✅ Ready for use
99
+
100
+ ## Key Insight
101
+
102
+ **Most important bulletproofing:** Anti-patterns section 展示 moment 里 feel justified 的 exact shortcuts。当 Claude 想 "I'll just add this one quick fix",看到 listed as wrong 的 exact pattern 产生 cognitive friction。
103
+
104
+ ## Usage Example
105
+
106
+ 遇到 bug 时:
107
+ 1. Load skill: .agents/skills/debugging/systematic-debugging
108
+ 2. Read overview (10 sec) — reminded of mandate
109
+ 3. Follow Phase 1 checklist — forced investigation
110
+ 4. If tempted to skip — see anti-pattern,stop
111
+ 5. Complete all phases — root cause found
112
+
113
+ **Time investment:** 5-10 minutes
114
+ **Time saved:** Hours of symptom-whack-a-mole
115
+
116
+ ---
117
+
118
+ *Created: 2025-10-03*
119
+ *Purpose: Reference example for skill extraction and bulletproofing*
@@ -1,158 +1,158 @@
1
- // Complete implementation of condition-based waiting utilities
2
- // From: Lace test infrastructure improvements (2025-10-03)
3
- // Context: Fixed 15 flaky tests by replacing arbitrary timeouts
4
-
5
- import type { ThreadManager } from '~/threads/thread-manager';
6
- import type { LaceEvent, LaceEventType } from '~/threads/types';
7
-
8
- /**
9
- * Wait for a specific event type to appear in thread
10
- *
11
- * @param threadManager - The thread manager to query
12
- * @param threadId - Thread to check for events
13
- * @param eventType - Type of event to wait for
14
- * @param timeoutMs - Maximum time to wait (default 5000ms)
15
- * @returns Promise resolving to the first matching event
16
- *
17
- * Example:
18
- * await waitForEvent(threadManager, agentThreadId, 'TOOL_RESULT');
19
- */
20
- export function waitForEvent(
21
- threadManager: ThreadManager,
22
- threadId: string,
23
- eventType: LaceEventType,
24
- timeoutMs = 5000
25
- ): Promise<LaceEvent> {
26
- return new Promise((resolve, reject) => {
27
- const startTime = Date.now();
28
-
29
- const check = () => {
30
- const events = threadManager.getEvents(threadId);
31
- const event = events.find((e) => e.type === eventType);
32
-
33
- if (event) {
34
- resolve(event);
35
- } else if (Date.now() - startTime > timeoutMs) {
36
- reject(new Error(`Timeout waiting for ${eventType} event after ${timeoutMs}ms`));
37
- } else {
38
- setTimeout(check, 10); // Poll every 10ms for efficiency
39
- }
40
- };
41
-
42
- check();
43
- });
44
- }
45
-
46
- /**
47
- * Wait for a specific number of events of a given type
48
- *
49
- * @param threadManager - The thread manager to query
50
- * @param threadId - Thread to check for events
51
- * @param eventType - Type of event to wait for
52
- * @param count - Number of events to wait for
53
- * @param timeoutMs - Maximum time to wait (default 5000ms)
54
- * @returns Promise resolving to all matching events once count is reached
55
- *
56
- * Example:
57
- * // Wait for 2 AGENT_MESSAGE events (initial response + continuation)
58
- * await waitForEventCount(threadManager, agentThreadId, 'AGENT_MESSAGE', 2);
59
- */
60
- export function waitForEventCount(
61
- threadManager: ThreadManager,
62
- threadId: string,
63
- eventType: LaceEventType,
64
- count: number,
65
- timeoutMs = 5000
66
- ): Promise<LaceEvent[]> {
67
- return new Promise((resolve, reject) => {
68
- const startTime = Date.now();
69
-
70
- const check = () => {
71
- const events = threadManager.getEvents(threadId);
72
- const matchingEvents = events.filter((e) => e.type === eventType);
73
-
74
- if (matchingEvents.length >= count) {
75
- resolve(matchingEvents);
76
- } else if (Date.now() - startTime > timeoutMs) {
77
- reject(
78
- new Error(
79
- `Timeout waiting for ${count} ${eventType} events after ${timeoutMs}ms (got ${matchingEvents.length})`
80
- )
81
- );
82
- } else {
83
- setTimeout(check, 10);
84
- }
85
- };
86
-
87
- check();
88
- });
89
- }
90
-
91
- /**
92
- * Wait for an event matching a custom predicate
93
- * Useful when you need to check event data, not just type
94
- *
95
- * @param threadManager - The thread manager to query
96
- * @param threadId - Thread to check for events
97
- * @param predicate - Function that returns true when event matches
98
- * @param description - Human-readable description for error messages
99
- * @param timeoutMs - Maximum time to wait (default 5000ms)
100
- * @returns Promise resolving to the first matching event
101
- *
102
- * Example:
103
- * // Wait for TOOL_RESULT with specific ID
104
- * await waitForEventMatch(
105
- * threadManager,
106
- * agentThreadId,
107
- * (e) => e.type === 'TOOL_RESULT' && e.data.id === 'call_123',
108
- * 'TOOL_RESULT with id=call_123'
109
- * );
110
- */
111
- export function waitForEventMatch(
112
- threadManager: ThreadManager,
113
- threadId: string,
114
- predicate: (event: LaceEvent) => boolean,
115
- description: string,
116
- timeoutMs = 5000
117
- ): Promise<LaceEvent> {
118
- return new Promise((resolve, reject) => {
119
- const startTime = Date.now();
120
-
121
- const check = () => {
122
- const events = threadManager.getEvents(threadId);
123
- const event = events.find(predicate);
124
-
125
- if (event) {
126
- resolve(event);
127
- } else if (Date.now() - startTime > timeoutMs) {
128
- reject(new Error(`Timeout waiting for ${description} after ${timeoutMs}ms`));
129
- } else {
130
- setTimeout(check, 10);
131
- }
132
- };
133
-
134
- check();
135
- });
136
- }
137
-
138
- // Usage example from actual debugging session:
139
- //
140
- // BEFORE (flaky):
141
- // ---------------
142
- // const messagePromise = agent.sendMessage('Execute tools');
143
- // await new Promise(r => setTimeout(r, 300)); // Hope tools start in 300ms
144
- // agent.abort();
145
- // await messagePromise;
146
- // await new Promise(r => setTimeout(r, 50)); // Hope results arrive in 50ms
147
- // expect(toolResults.length).toBe(2); // Fails randomly
148
- //
149
- // AFTER (reliable):
150
- // ----------------
151
- // const messagePromise = agent.sendMessage('Execute tools');
152
- // await waitForEventCount(threadManager, threadId, 'TOOL_CALL', 2); // Wait for tools to start
153
- // agent.abort();
154
- // await messagePromise;
155
- // await waitForEventCount(threadManager, threadId, 'TOOL_RESULT', 2); // Wait for results
156
- // expect(toolResults.length).toBe(2); // Always succeeds
157
- //
158
- // Result: 60% pass rate → 100%, 40% faster execution
1
+ // Complete implementation of condition-based waiting utilities
2
+ // From: Lace test infrastructure improvements (2025-10-03)
3
+ // Context: Fixed 15 flaky tests by replacing arbitrary timeouts
4
+
5
+ import type { ThreadManager } from '~/threads/thread-manager';
6
+ import type { LaceEvent, LaceEventType } from '~/threads/types';
7
+
8
+ /**
9
+ * Wait for a specific event type to appear in thread
10
+ *
11
+ * @param threadManager - The thread manager to query
12
+ * @param threadId - Thread to check for events
13
+ * @param eventType - Type of event to wait for
14
+ * @param timeoutMs - Maximum time to wait (default 5000ms)
15
+ * @returns Promise resolving to the first matching event
16
+ *
17
+ * Example:
18
+ * await waitForEvent(threadManager, agentThreadId, 'TOOL_RESULT');
19
+ */
20
+ export function waitForEvent(
21
+ threadManager: ThreadManager,
22
+ threadId: string,
23
+ eventType: LaceEventType,
24
+ timeoutMs = 5000
25
+ ): Promise<LaceEvent> {
26
+ return new Promise((resolve, reject) => {
27
+ const startTime = Date.now();
28
+
29
+ const check = () => {
30
+ const events = threadManager.getEvents(threadId);
31
+ const event = events.find((e) => e.type === eventType);
32
+
33
+ if (event) {
34
+ resolve(event);
35
+ } else if (Date.now() - startTime > timeoutMs) {
36
+ reject(new Error(`Timeout waiting for ${eventType} event after ${timeoutMs}ms`));
37
+ } else {
38
+ setTimeout(check, 10); // Poll every 10ms for efficiency
39
+ }
40
+ };
41
+
42
+ check();
43
+ });
44
+ }
45
+
46
+ /**
47
+ * Wait for a specific number of events of a given type
48
+ *
49
+ * @param threadManager - The thread manager to query
50
+ * @param threadId - Thread to check for events
51
+ * @param eventType - Type of event to wait for
52
+ * @param count - Number of events to wait for
53
+ * @param timeoutMs - Maximum time to wait (default 5000ms)
54
+ * @returns Promise resolving to all matching events once count is reached
55
+ *
56
+ * Example:
57
+ * // Wait for 2 AGENT_MESSAGE events (initial response + continuation)
58
+ * await waitForEventCount(threadManager, agentThreadId, 'AGENT_MESSAGE', 2);
59
+ */
60
+ export function waitForEventCount(
61
+ threadManager: ThreadManager,
62
+ threadId: string,
63
+ eventType: LaceEventType,
64
+ count: number,
65
+ timeoutMs = 5000
66
+ ): Promise<LaceEvent[]> {
67
+ return new Promise((resolve, reject) => {
68
+ const startTime = Date.now();
69
+
70
+ const check = () => {
71
+ const events = threadManager.getEvents(threadId);
72
+ const matchingEvents = events.filter((e) => e.type === eventType);
73
+
74
+ if (matchingEvents.length >= count) {
75
+ resolve(matchingEvents);
76
+ } else if (Date.now() - startTime > timeoutMs) {
77
+ reject(
78
+ new Error(
79
+ `Timeout waiting for ${count} ${eventType} events after ${timeoutMs}ms (got ${matchingEvents.length})`
80
+ )
81
+ );
82
+ } else {
83
+ setTimeout(check, 10);
84
+ }
85
+ };
86
+
87
+ check();
88
+ });
89
+ }
90
+
91
+ /**
92
+ * Wait for an event matching a custom predicate
93
+ * Useful when you need to check event data, not just type
94
+ *
95
+ * @param threadManager - The thread manager to query
96
+ * @param threadId - Thread to check for events
97
+ * @param predicate - Function that returns true when event matches
98
+ * @param description - Human-readable description for error messages
99
+ * @param timeoutMs - Maximum time to wait (default 5000ms)
100
+ * @returns Promise resolving to the first matching event
101
+ *
102
+ * Example:
103
+ * // Wait for TOOL_RESULT with specific ID
104
+ * await waitForEventMatch(
105
+ * threadManager,
106
+ * agentThreadId,
107
+ * (e) => e.type === 'TOOL_RESULT' && e.data.id === 'call_123',
108
+ * 'TOOL_RESULT with id=call_123'
109
+ * );
110
+ */
111
+ export function waitForEventMatch(
112
+ threadManager: ThreadManager,
113
+ threadId: string,
114
+ predicate: (event: LaceEvent) => boolean,
115
+ description: string,
116
+ timeoutMs = 5000
117
+ ): Promise<LaceEvent> {
118
+ return new Promise((resolve, reject) => {
119
+ const startTime = Date.now();
120
+
121
+ const check = () => {
122
+ const events = threadManager.getEvents(threadId);
123
+ const event = events.find(predicate);
124
+
125
+ if (event) {
126
+ resolve(event);
127
+ } else if (Date.now() - startTime > timeoutMs) {
128
+ reject(new Error(`Timeout waiting for ${description} after ${timeoutMs}ms`));
129
+ } else {
130
+ setTimeout(check, 10);
131
+ }
132
+ };
133
+
134
+ check();
135
+ });
136
+ }
137
+
138
+ // Usage example from actual debugging session:
139
+ //
140
+ // BEFORE (flaky):
141
+ // ---------------
142
+ // const messagePromise = agent.sendMessage('Execute tools');
143
+ // await new Promise(r => setTimeout(r, 300)); // Hope tools start in 300ms
144
+ // agent.abort();
145
+ // await messagePromise;
146
+ // await new Promise(r => setTimeout(r, 50)); // Hope results arrive in 50ms
147
+ // expect(toolResults.length).toBe(2); // Fails randomly
148
+ //
149
+ // AFTER (reliable):
150
+ // ----------------
151
+ // const messagePromise = agent.sendMessage('Execute tools');
152
+ // await waitForEventCount(threadManager, threadId, 'TOOL_CALL', 2); // Wait for tools to start
153
+ // agent.abort();
154
+ // await messagePromise;
155
+ // await waitForEventCount(threadManager, threadId, 'TOOL_RESULT', 2); // Wait for results
156
+ // expect(toolResults.length).toBe(2); // Always succeeds
157
+ //
158
+ // Result: 60% pass rate → 100%, 40% faster execution