pi-smart-compact 9.7.1 → 10.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (206) hide show
  1. package/ARCHITECTURE.md +973 -372
  2. package/CHANGELOG.md +721 -0
  3. package/LICENSE +8 -0
  4. package/README.md +128 -640
  5. package/SECURITY.md +34 -12
  6. package/SUPPORT.md +26 -9
  7. package/assets/DejaVu-LICENSE.txt +187 -0
  8. package/assets/DejaVuSansMono.ttf +0 -0
  9. package/assets/README.md +26 -0
  10. package/assets/skills/context-management/SKILL.md +34 -0
  11. package/dist/app/anchor-cache.d.ts +36 -0
  12. package/dist/app/anchor-cache.d.ts.map +1 -0
  13. package/dist/app/artifact-storage.d.ts +47 -0
  14. package/dist/app/artifact-storage.d.ts.map +1 -0
  15. package/dist/app/background-preparation.d.ts +39 -0
  16. package/dist/app/background-preparation.d.ts.map +1 -0
  17. package/dist/app/compaction-commit-store.d.ts +5 -1
  18. package/dist/app/compaction-commit-store.d.ts.map +1 -1
  19. package/dist/app/context-evidence.d.ts +57 -0
  20. package/dist/app/context-evidence.d.ts.map +1 -0
  21. package/dist/app/context-guide.d.ts +3 -0
  22. package/dist/app/context-guide.d.ts.map +1 -0
  23. package/dist/app/context-operations.d.ts +106 -0
  24. package/dist/app/context-operations.d.ts.map +1 -0
  25. package/dist/app/effective-state.d.ts +23 -0
  26. package/dist/app/effective-state.d.ts.map +1 -0
  27. package/dist/app/global-settings-runtime.d.ts +3 -3
  28. package/dist/app/global-settings-runtime.d.ts.map +1 -1
  29. package/dist/app/hindsight-memory.d.ts +100 -0
  30. package/dist/app/hindsight-memory.d.ts.map +1 -0
  31. package/dist/app/host-cache-ledger.d.ts +68 -0
  32. package/dist/app/host-cache-ledger.d.ts.map +1 -0
  33. package/dist/app/lazy-tools.d.ts +36 -0
  34. package/dist/app/lazy-tools.d.ts.map +1 -0
  35. package/dist/app/memory-backend.d.ts +58 -0
  36. package/dist/app/memory-backend.d.ts.map +1 -0
  37. package/dist/app/mnemopi-memory.d.ts +13 -0
  38. package/dist/app/mnemopi-memory.d.ts.map +1 -0
  39. package/dist/app/mnemopi-protocol.d.ts +78 -0
  40. package/dist/app/mnemopi-protocol.d.ts.map +1 -0
  41. package/dist/app/mnemopi-worker.d.ts +2 -0
  42. package/dist/app/mnemopi-worker.d.ts.map +1 -0
  43. package/dist/app/model-feasibility.d.ts +20 -0
  44. package/dist/app/model-feasibility.d.ts.map +1 -0
  45. package/dist/app/native-compaction.d.ts +88 -0
  46. package/dist/app/native-compaction.d.ts.map +1 -0
  47. package/dist/app/native-continuity-bridge.d.ts.map +1 -1
  48. package/dist/app/navigation-data.d.ts +28 -0
  49. package/dist/app/navigation-data.d.ts.map +1 -0
  50. package/dist/app/navigation-types.d.ts +60 -0
  51. package/dist/app/navigation-types.d.ts.map +1 -0
  52. package/dist/app/pending-slot.d.ts +11 -1
  53. package/dist/app/pending-slot.d.ts.map +1 -1
  54. package/dist/app/preflight.d.ts.map +1 -1
  55. package/dist/app/register-context-tools.d.ts +16 -3
  56. package/dist/app/register-context-tools.d.ts.map +1 -1
  57. package/dist/app/register-navigation.d.ts +20 -0
  58. package/dist/app/register-navigation.d.ts.map +1 -0
  59. package/dist/app/register-smart-compact-command.d.ts +17 -2
  60. package/dist/app/register-smart-compact-command.d.ts.map +1 -1
  61. package/dist/app/register-smart-compact-tool.d.ts.map +1 -1
  62. package/dist/app/register-smart-context-tool.d.ts +55 -0
  63. package/dist/app/register-smart-context-tool.d.ts.map +1 -0
  64. package/dist/app/run-context.d.ts +1 -0
  65. package/dist/app/run-context.d.ts.map +1 -1
  66. package/dist/app/run-smart-compact.d.ts +3 -3
  67. package/dist/app/run-smart-compact.d.ts.map +1 -1
  68. package/dist/app/session-handoff.d.ts +64 -0
  69. package/dist/app/session-handoff.d.ts.map +1 -0
  70. package/dist/app/session-lineage.d.ts +17 -0
  71. package/dist/app/session-lineage.d.ts.map +1 -0
  72. package/dist/app/session-run-lock.d.ts +0 -2
  73. package/dist/app/session-run-lock.d.ts.map +1 -1
  74. package/dist/app/settled-auto-trigger.d.ts +2 -0
  75. package/dist/app/settled-auto-trigger.d.ts.map +1 -1
  76. package/dist/app/smart-compact-input.d.ts +1 -1
  77. package/dist/app/smart-compact-input.d.ts.map +1 -1
  78. package/dist/app/smart-compact-policy.d.ts +1 -1
  79. package/dist/app/smart-compact-policy.d.ts.map +1 -1
  80. package/dist/app/steps/extract.d.ts +45 -1
  81. package/dist/app/steps/extract.d.ts.map +1 -1
  82. package/dist/app/steps/metrics.d.ts +1 -0
  83. package/dist/app/steps/metrics.d.ts.map +1 -1
  84. package/dist/app/steps/persist.d.ts.map +1 -1
  85. package/dist/app/steps/prepare.d.ts.map +1 -1
  86. package/dist/app/steps/recover.d.ts +9 -0
  87. package/dist/app/steps/recover.d.ts.map +1 -1
  88. package/dist/app/steps/synthesize.d.ts.map +1 -1
  89. package/dist/app/steps/tier.d.ts.map +1 -1
  90. package/dist/app/steps/verify.d.ts.map +1 -1
  91. package/dist/app/steps/visual.d.ts +4 -0
  92. package/dist/app/steps/visual.d.ts.map +1 -0
  93. package/dist/app/steps/window.d.ts.map +1 -1
  94. package/dist/app/tool-artifacts.d.ts +27 -0
  95. package/dist/app/tool-artifacts.d.ts.map +1 -0
  96. package/dist/app/visual-archive.d.ts +29 -0
  97. package/dist/app/visual-archive.d.ts.map +1 -0
  98. package/dist/constants.d.ts +96 -1
  99. package/dist/constants.d.ts.map +1 -1
  100. package/dist/domain/compaction-usage.d.ts +16 -0
  101. package/dist/domain/compaction-usage.d.ts.map +1 -0
  102. package/dist/domain/model-capacity.d.ts +12 -0
  103. package/dist/domain/model-capacity.d.ts.map +1 -0
  104. package/dist/domain/provider-evaluation.d.ts +7 -0
  105. package/dist/domain/provider-evaluation.d.ts.map +1 -1
  106. package/dist/domain/telemetry.d.ts +43 -2
  107. package/dist/domain/telemetry.d.ts.map +1 -1
  108. package/dist/domain/tool-semantics.d.ts +23 -0
  109. package/dist/domain/tool-semantics.d.ts.map +1 -1
  110. package/dist/index.d.ts.map +1 -1
  111. package/dist/index.js +15757 -6942
  112. package/dist/infra/ai-messages.d.ts +1 -1
  113. package/dist/infra/ai-messages.d.ts.map +1 -1
  114. package/dist/infra/context-graph.d.ts +38 -7
  115. package/dist/infra/context-graph.d.ts.map +1 -1
  116. package/dist/infra/fs.d.ts.map +1 -1
  117. package/dist/infra/hindsight-client.d.ts +73 -0
  118. package/dist/infra/hindsight-client.d.ts.map +1 -0
  119. package/dist/infra/hindsight-receipts.d.ts +68 -0
  120. package/dist/infra/hindsight-receipts.d.ts.map +1 -0
  121. package/dist/infra/llm-client.d.ts +26 -23
  122. package/dist/infra/llm-client.d.ts.map +1 -1
  123. package/dist/infra/memory-ref.d.ts +27 -0
  124. package/dist/infra/memory-ref.d.ts.map +1 -0
  125. package/dist/infra/native-protocol.d.ts +54 -0
  126. package/dist/infra/native-protocol.d.ts.map +1 -0
  127. package/dist/infra/optional-components.d.ts +15 -0
  128. package/dist/infra/optional-components.d.ts.map +1 -0
  129. package/dist/infra/paths.d.ts +2 -0
  130. package/dist/infra/paths.d.ts.map +1 -1
  131. package/dist/infra/services.d.ts +15 -5
  132. package/dist/infra/services.d.ts.map +1 -1
  133. package/dist/infra/visual-renderer.d.ts +16 -0
  134. package/dist/infra/visual-renderer.d.ts.map +1 -0
  135. package/dist/mnemopi-worker.js +213 -0
  136. package/dist/phases/explore.d.ts +12 -9
  137. package/dist/phases/explore.d.ts.map +1 -1
  138. package/dist/phases/synthesize.d.ts +18 -3
  139. package/dist/phases/synthesize.d.ts.map +1 -1
  140. package/dist/phases/verify.d.ts +5 -1
  141. package/dist/phases/verify.d.ts.map +1 -1
  142. package/dist/rtk.d.ts +7 -0
  143. package/dist/rtk.d.ts.map +1 -0
  144. package/dist/rtk.js +767 -0
  145. package/dist/types.d.ts +128 -4
  146. package/dist/types.d.ts.map +1 -1
  147. package/dist/ui/dashboard-format.d.ts +2 -1
  148. package/dist/ui/dashboard-format.d.ts.map +1 -1
  149. package/dist/ui/dashboard-insights.d.ts +9 -1
  150. package/dist/ui/dashboard-insights.d.ts.map +1 -1
  151. package/dist/ui/error-format.d.ts +7 -2
  152. package/dist/ui/error-format.d.ts.map +1 -1
  153. package/dist/ui/handoff-overlay.d.ts +26 -0
  154. package/dist/ui/handoff-overlay.d.ts.map +1 -0
  155. package/dist/ui/home-overlay.d.ts +54 -0
  156. package/dist/ui/home-overlay.d.ts.map +1 -0
  157. package/dist/ui/metrics-dashboard-overlay.d.ts.map +1 -1
  158. package/dist/ui/metrics-report.d.ts.map +1 -1
  159. package/dist/ui/navigation-overlay.d.ts +92 -0
  160. package/dist/ui/navigation-overlay.d.ts.map +1 -0
  161. package/dist/ui/overlays.d.ts +12 -2
  162. package/dist/ui/overlays.d.ts.map +1 -1
  163. package/dist/ui/profiles.d.ts +51 -0
  164. package/dist/ui/profiles.d.ts.map +1 -0
  165. package/dist/ui/settings-complex.d.ts +49 -3
  166. package/dist/ui/settings-complex.d.ts.map +1 -1
  167. package/dist/ui/settings-list.d.ts +28 -0
  168. package/dist/ui/settings-list.d.ts.map +1 -0
  169. package/dist/ui/settings-overlay.d.ts +13 -6
  170. package/dist/ui/settings-overlay.d.ts.map +1 -1
  171. package/dist/ui/storage-report.d.ts +4 -0
  172. package/dist/ui/storage-report.d.ts.map +1 -0
  173. package/dist/utils/backups.d.ts.map +1 -1
  174. package/dist/utils/cache.d.ts +6 -2
  175. package/dist/utils/cache.d.ts.map +1 -1
  176. package/dist/utils/config.d.ts +12 -0
  177. package/dist/utils/config.d.ts.map +1 -1
  178. package/dist/utils/helpers.d.ts.map +1 -1
  179. package/dist/utils/id-fingerprint.d.ts +3 -1
  180. package/dist/utils/id-fingerprint.d.ts.map +1 -1
  181. package/dist/utils/issues.d.ts +61 -0
  182. package/dist/utils/issues.d.ts.map +1 -0
  183. package/dist/utils/pruning.d.ts.map +1 -1
  184. package/dist/utils/session-log.d.ts +0 -2
  185. package/dist/utils/session-log.d.ts.map +1 -1
  186. package/dist/utils/state.d.ts +3 -1
  187. package/dist/utils/state.d.ts.map +1 -1
  188. package/dist/utils/tokens.d.ts +10 -2
  189. package/dist/utils/tokens.d.ts.map +1 -1
  190. package/docs/MIGRATING_TO_V8.md +7 -1
  191. package/docs/README.md +69 -0
  192. package/docs/RELEASE.md +173 -56
  193. package/docs/assets/banner.png +0 -0
  194. package/docs/assets/banner.svg +1158 -70
  195. package/docs/assets/pi-smart-compact.png +0 -0
  196. package/docs/assets/pi-smart-compact.svg +24 -0
  197. package/docs/configuration.md +637 -0
  198. package/docs/evaluation.md +408 -0
  199. package/docs/guide.md +860 -0
  200. package/docs/hindsight-memory.md +314 -0
  201. package/docs/identity.md +124 -0
  202. package/package.json +44 -11
  203. package/dist/provider-eval.js +0 -2122
  204. package/dist/provider-scenario-eval.js +0 -2900
  205. package/dist/telemetry-report.js +0 -1973
  206. package/docs/provider-evaluation-2026-08-06.md +0 -63
@@ -1,2900 +0,0 @@
1
- // @bun
2
- // scripts/provider-scenario-eval.ts
3
- import { ModelRegistry, ModelRuntime } from "@earendil-works/pi-coding-agent";
4
-
5
- // src/constants.ts
6
- var VERSION = "9.7.1";
7
- var SETTLED_TRIGGER_COOLDOWN_MS = 10 * 60000;
8
- var COMPACT_SYSTEM_PREFIX = "You are an expert conversation summarizer for a coding agent. " + "Produce structured markdown summaries. " + "Follow output format exactly. " + "Use EXACT names \u2014 never paraphrase code identifiers. " + "Trust deterministic extraction data over intuition.";
9
- var PROFILES = {
10
- light: {
11
- summaryBudgetTokens: 1e4,
12
- keepRecentTokens: 30000,
13
- minChunkTokens: 800,
14
- maxChunkTokens: 12000,
15
- singlePassMaxTokens: 40000,
16
- batchMaxTokens: 30000
17
- },
18
- balanced: {
19
- summaryBudgetTokens: 6000,
20
- keepRecentTokens: 20000,
21
- minChunkTokens: 500,
22
- maxChunkTokens: 8000,
23
- singlePassMaxTokens: 30000,
24
- batchMaxTokens: 24000
25
- },
26
- aggressive: {
27
- summaryBudgetTokens: 3000,
28
- keepRecentTokens: 1e4,
29
- minChunkTokens: 300,
30
- maxChunkTokens: 6000,
31
- singlePassMaxTokens: 20000,
32
- batchMaxTokens: 18000
33
- }
34
- };
35
- var DEFAULT_CONFIG = {
36
- mode: "auto",
37
- profile: "balanced",
38
- profiles: PROFILES,
39
- summaryModel: null,
40
- segmentationModel: null,
41
- verificationModel: null,
42
- summaryThinkingLevel: "minimal",
43
- segmentationThinkingLevel: "minimal",
44
- agentToolAccess: "inherit",
45
- autoTrigger: true,
46
- showStatus: true,
47
- autoTriggerStrategy: "native-hook",
48
- autoTriggerTimeoutMs: 120000,
49
- backupEnabled: true,
50
- backupDir: "",
51
- minContextPercent: 60,
52
- requireApproval: true,
53
- scrubSecrets: true,
54
- scrubPii: false,
55
- maxLlmCalls: 8,
56
- maxLlmInputTokens: 0,
57
- codexMaxCallMs: 0,
58
- maxLatencyMs: 0,
59
- pendingTtlMs: 300000,
60
- focusWeighting: true,
61
- zeroCallEnabled: true,
62
- contextGraphEnabled: true,
63
- telemetryChannel: "stable",
64
- adaptiveDamageFeedback: false,
65
- onlineDamageMonitor: true,
66
- pinPaths: []
67
- };
68
- var SINGLE_PASS_PREFIX = `Summarize this coding agent conversation. Produce ONE structured summary.
69
- ` + `
70
- Rules for Accuracy:
71
- ` + `1. Session Type: read-only tool calls = REVIEW, not implementation
72
- ` + `2. Status: Check for user complaints before marking "Done"
73
- ` + `3. Exact Names: Quote specific variable/function/parameter names, don't paraphrase
74
- ` + `4. Files: Use the VERIFIED file lists above (deterministically extracted, zero hallucination risk)
75
- ` + `
76
- Output EXACTLY this format:
77
-
78
- ` + `## Goal
79
- [What the user is trying to accomplish]
80
- ` + `## Constraints & Preferences
81
- - [CRITICAL: user requirements, preferences, constraints]
82
- ` + `## Progress
83
- ### Done
84
- - [x] [Completed tasks with file references]
85
- ### In Progress
86
- - [ ] [Current work state]
87
- ### Blocked
88
- - [Issues]
89
- ` + `## Key Decisions
90
- - **[Decision]**: [Rationale]
91
- ` + `## Files Modified
92
- - [Verified list from deterministic extraction]
93
- ` + `## Files Deleted
94
- - [Verified deleted paths from deterministic extraction]
95
- ` + `## Files Read
96
- - [Verified list from deterministic extraction]
97
- ` + `## Next Steps
98
- 1. [What should happen next]
99
- ` + `## Critical Context
100
- - [Specific data, patterns, info needed to continue]
101
- - [Error patterns or gotchas]
102
- ` + `## Topics Covered
103
- [Chronological bullet list with priority in brackets]
104
- `;
105
- var BATCH_PROMPT_PREFIX = `Summarize these conversation segments.
106
-
107
- Rules for Accuracy:
108
- ` + `1. Use EXACT file paths from extraction data
109
- ` + `2. Status: only mark "done" if there's clear evidence (successful test run, user confirmation)
110
- ` + `3. Quote specific values, don't paraphrase code
111
-
112
- ` + `For EACH segment produce EXACTLY:
113
- ` + `### CHUNK {NUMBER}: {TOPIC_NAME}
114
- ` + `**Priority**: [critical|high|normal|low]
115
- ` + `**Summary**: [2-4 sentences: what happened, errors, code changes with paths]
116
- ` + `**Decisions**: [comma-separated, or "None"]
117
- ` + `**Modified**: [comma-separated paths, or "None"]
118
- ` + `**Deleted**: [comma-separated paths, or "None"]
119
- ` + `**Read**: [comma-separated paths, or "None"]
120
- `;
121
- var ASSEMBLY_PROMPT_PREFIX = `Merge these topic summaries into ONE coherent summary.
122
-
123
- ` + `## IMMUTABLE CONTEXT (do not modify or contradict these facts)
124
- ` + `These are deterministically verified from the original conversation. They take priority over ANY summary content below.
125
-
126
- ` + `Rules:
127
- ` + `1. Preserve ALL critical/high info. Condense normal, minimize low.
128
- ` + `2. Chronological order.
129
- ` + `3. The pre-processed data below is GROUND TRUTH \u2014 trust it over individual summaries.
130
- ` + `4. Files Modified list is deterministically verified \u2014 if a summary says a file was modified but it's NOT in the list above, omit it.
131
- ` + `5. Key Decisions below are verified \u2014 preserve them exactly, do not paraphrase the decision text.
132
- ` + `6. Do NOT fabricate file paths, function names, or error messages not present in the verified data.
133
-
134
- ` + `Format:
135
- ` + `## Goal
136
- [Overall objective]
137
- ` + `## Constraints & Preferences
138
- - [CRITICAL requirements, preferences, constraints]
139
- ` + `## Progress
140
- ### Done
141
- - [x] [Completed tasks with file refs]
142
- ### In Progress
143
- - [ ] [Current work state]
144
- ### Blocked
145
- - [Issues]
146
- ` + `## Key Decisions
147
- - **[Decision]**: [Rationale]
148
- ` + `## Files Modified
149
- - [Verified deterministic list]
150
- ` + `## Files Deleted
151
- - [Verified deterministic deleted paths]
152
- ` + `## Files Read
153
- - [Verified deterministic list]
154
- ` + `## Next Steps
155
- 1. [What should happen next]
156
- ` + `## Critical Context
157
- - [Data, patterns, info needed]
158
- ` + `## Topics Covered
159
- [Chronological bullets with priority]
160
- `;
161
- var LOG_PREFIX = "[smart-compact]";
162
- var BACKUP_MAX_AGE_MS = 14 * 24 * 60 * 60 * 1000;
163
- var FIVE_MINUTES_MS = 5 * 60 * 1000;
164
- var ONE_HOUR_MS = 60 * 60 * 1000;
165
- var SEVEN_DAYS_MS = 7 * 24 * 60 * 60 * 1000;
166
- var THIRTY_DAYS_MS = 30 * 24 * 60 * 60 * 1000;
167
- var ID_PREFIX = {
168
- PROJECT: "proj-",
169
- COMPACT_SESSION: "sc-",
170
- MULTI_TOOL_USE_SYNTHETIC: "mtu_",
171
- OPEN_LOOP: "loop-",
172
- DECISION: "decision-",
173
- ERROR: "error-"
174
- };
175
- var TUNING = {
176
- EMA_PREV: 0.7,
177
- EMA_SAMPLE: 0.3,
178
- CALIBRATION_CLAMP_MIN: 0.3,
179
- CALIBRATION_CLAMP_MAX: 3,
180
- CONFIDENCE_HIGH: 0.85,
181
- CONFIDENCE_MEDIUM: 0.8,
182
- CONFIDENCE_LOW: 0.6
183
- };
184
- var TRUNC = {
185
- MESSAGE: 300,
186
- ERROR_DETAIL: 500,
187
- DECISION_SUMMARY: 200,
188
- USER_RESPONSE: 300,
189
- CONSTRAINT_TEXT: 300,
190
- OPEN_LOOP_SUMMARY: 120,
191
- TIMELINE_EVENT: 150,
192
- TIMELINE_ERROR: 100,
193
- SNIPPET: 80,
194
- PREVIEW: 150,
195
- PREVIEW_MID: 200,
196
- DETAIL: 300,
197
- PREVIEW_LONG: 400,
198
- PREVIEW_XL: 500,
199
- PREVIOUS_SUMMARY: 12000,
200
- CONTINUITY_CAPSULE: 4000,
201
- CHUNK_FALLBACK: 180,
202
- DECISION_DETAIL: 60,
203
- TOPIC_LABEL: 100,
204
- PROJ_ID_HASH: 12,
205
- CONV_HASH: 8,
206
- RESULT_GAPS: 5,
207
- SESSION_ID_DISPLAY: 20,
208
- ERROR_SNIPPET: 30,
209
- TIMELINE_DISPLAY: 10,
210
- EXPLORE_RESULTS: 15,
211
- BACKUP_PREVIEW_LINES: 5,
212
- FINGERPRINT_SEG: 2
213
- };
214
- var LIKELY_ERROR_RE = /(?:command not found|no such file|permission denied|syntax error|cannot find|module not found|compilation error|build failed|test failed|^FAIL\b|ERROR:)/i;
215
- var METRICS_BUFFER_MAX = 200;
216
- var RUNTIME_LOG_MAX_BYTES = 5 * 1024 * 1024;
217
- var EXPLORER_SYSTEM_PROMPT = `You are a conversation analyst. You have deterministic extraction data and can query the raw conversation using tools.
218
-
219
- ` + `Your job:
220
- ` + `1. Verify/enrich the extracted boundaries (merge, split, or add as needed)
221
- ` + `2. Identify cross-topic relationships
222
- ` + `3. Find implicit constraints (user tone, frustration, urgency)
223
- ` + `4. Assess completion status accurately
224
- ` + `5. Extract the narrative arc
225
-
226
- ` + `Use tools BEFORE forming conclusions. Finish within 3 tool rounds.
227
-
228
- ` + `After exploration, output ONLY a JSON object (no markdown):
229
- ` + '{"boundaries":[{"afterIndex":N,"topic":"...","priority":"critical|high|normal|low","confidence":0.0-1.0}],"mainGoal":"...","sessionType":"implementation|review|debugging|discussion","enrichedConstraints":[...],"crossReferences":[...],"statusAssessment":{"done":[...],"inProgress":[...],"blocked":[...]},"criticalContext":[...],"keyDecisions":[...]}';
230
-
231
- // src/utils/lru.ts
232
- function lruGet(m, key) {
233
- if (!m.has(key))
234
- return;
235
- const v = m.get(key);
236
- m.delete(key);
237
- m.set(key, v);
238
- return v;
239
- }
240
- function lruSet(m, key, value, max) {
241
- if (m.has(key))
242
- m.delete(key);
243
- m.set(key, value);
244
- while (m.size > max) {
245
- const oldest = m.keys().next().value;
246
- if (oldest === undefined)
247
- break;
248
- m.delete(oldest);
249
- }
250
- }
251
-
252
- // src/utils/tokens.ts
253
- var PROVIDER_MAP = {
254
- "zai-anthropic": {
255
- maxOutputTokens: 8192,
256
- supportsTools: "probe",
257
- jsonReliability: "high",
258
- instructionFollowing: "high",
259
- tokenRatioEstimate: 3.5,
260
- concurrencyLimit: 3,
261
- cacheStrategy: "anthropic",
262
- timeoutMultiplier: 1.2,
263
- singlePassTokenMultiplier: 1,
264
- multimodal: "metadata-only"
265
- },
266
- "kimi-coding": {
267
- maxOutputTokens: 8192,
268
- supportsTools: "probe",
269
- jsonReliability: "high",
270
- instructionFollowing: "high",
271
- tokenRatioEstimate: 3.5,
272
- concurrencyLimit: 2,
273
- cacheStrategy: "anthropic",
274
- timeoutMultiplier: 1.5,
275
- singlePassTokenMultiplier: 0.95,
276
- multimodal: "metadata-only"
277
- },
278
- anthropic: {
279
- maxOutputTokens: 8192,
280
- supportsTools: true,
281
- jsonReliability: "high",
282
- instructionFollowing: "high",
283
- tokenRatioEstimate: 3.5,
284
- concurrencyLimit: 3,
285
- cacheStrategy: "anthropic",
286
- timeoutMultiplier: 1.2,
287
- singlePassTokenMultiplier: 1,
288
- multimodal: "native"
289
- },
290
- openai: {
291
- maxOutputTokens: 16384,
292
- supportsTools: true,
293
- jsonReliability: "high",
294
- instructionFollowing: "high",
295
- tokenRatioEstimate: 4,
296
- concurrencyLimit: 5,
297
- cacheStrategy: "openai",
298
- timeoutMultiplier: 1,
299
- singlePassTokenMultiplier: 1.15,
300
- multimodal: "native"
301
- },
302
- google: {
303
- maxOutputTokens: 8192,
304
- supportsTools: true,
305
- jsonReliability: "high",
306
- instructionFollowing: "high",
307
- tokenRatioEstimate: 3.8,
308
- concurrencyLimit: 3,
309
- cacheStrategy: "openai",
310
- timeoutMultiplier: 1.15,
311
- singlePassTokenMultiplier: 1.1,
312
- multimodal: "native"
313
- },
314
- deepseek: {
315
- maxOutputTokens: 8192,
316
- supportsTools: true,
317
- jsonReliability: "medium",
318
- instructionFollowing: "medium",
319
- tokenRatioEstimate: 3.6,
320
- concurrencyLimit: 2,
321
- cacheStrategy: "none",
322
- timeoutMultiplier: 1.5,
323
- singlePassTokenMultiplier: 0.85,
324
- multimodal: "metadata-only"
325
- },
326
- minimax: {
327
- maxOutputTokens: 4096,
328
- supportsTools: "probe",
329
- jsonReliability: "medium",
330
- instructionFollowing: "medium",
331
- tokenRatioEstimate: 3.8,
332
- concurrencyLimit: 2,
333
- cacheStrategy: "anthropic",
334
- timeoutMultiplier: 1.6,
335
- singlePassTokenMultiplier: 0.8,
336
- multimodal: "metadata-only"
337
- },
338
- "xiaomi-token-plan": {
339
- maxOutputTokens: 8192,
340
- supportsTools: "probe",
341
- jsonReliability: "medium",
342
- instructionFollowing: "medium",
343
- tokenRatioEstimate: 3.3,
344
- concurrencyLimit: 2,
345
- cacheStrategy: "openai",
346
- timeoutMultiplier: 1.35,
347
- singlePassTokenMultiplier: 0.9,
348
- multimodal: "metadata-only"
349
- },
350
- "xiaomi-mimo": {
351
- maxOutputTokens: 8192,
352
- supportsTools: "probe",
353
- jsonReliability: "medium",
354
- instructionFollowing: "medium",
355
- tokenRatioEstimate: 3.3,
356
- concurrencyLimit: 2,
357
- cacheStrategy: "anthropic",
358
- timeoutMultiplier: 1.35,
359
- singlePassTokenMultiplier: 0.9,
360
- multimodal: "metadata-only"
361
- },
362
- crofai: {
363
- maxOutputTokens: 8192,
364
- supportsTools: "probe",
365
- jsonReliability: "medium",
366
- instructionFollowing: "medium",
367
- tokenRatioEstimate: 3.8,
368
- concurrencyLimit: 3,
369
- cacheStrategy: "none",
370
- timeoutMultiplier: 1.2,
371
- singlePassTokenMultiplier: 0.95,
372
- multimodal: "metadata-only"
373
- },
374
- mistral: {
375
- maxOutputTokens: 8192,
376
- supportsTools: true,
377
- jsonReliability: "high",
378
- instructionFollowing: "high",
379
- tokenRatioEstimate: 3.5,
380
- concurrencyLimit: 3,
381
- cacheStrategy: "openai",
382
- timeoutMultiplier: 1.2,
383
- singlePassTokenMultiplier: 1,
384
- multimodal: "metadata-only"
385
- },
386
- xai: {
387
- maxOutputTokens: 8192,
388
- supportsTools: true,
389
- jsonReliability: "medium",
390
- instructionFollowing: "high",
391
- tokenRatioEstimate: 3.8,
392
- concurrencyLimit: 3,
393
- cacheStrategy: "openai",
394
- timeoutMultiplier: 1.2,
395
- singlePassTokenMultiplier: 1,
396
- multimodal: "native"
397
- }
398
- };
399
- var PROVIDER_ALIASES = [
400
- { pattern: /anthropic/i, provider: "anthropic" },
401
- { pattern: /kimi/i, provider: "kimi-coding" },
402
- { pattern: /zai/i, provider: "zai-anthropic" },
403
- { pattern: /openai/i, provider: "openai" },
404
- { pattern: /gpt/i, provider: "openai" },
405
- { pattern: /google|gemini/i, provider: "google" },
406
- { pattern: /deepseek/i, provider: "deepseek" },
407
- { pattern: /minimax/i, provider: "minimax" },
408
- { pattern: /xiaomi-mimo/i, provider: "xiaomi-mimo" },
409
- { pattern: /xiaomi/i, provider: "xiaomi-token-plan" },
410
- { pattern: /crofai/i, provider: "crofai" },
411
- { pattern: /mistral/i, provider: "mistral" },
412
- { pattern: /xai|grok/i, provider: "xai" }
413
- ];
414
- var DEFAULT_CAPS = {
415
- maxOutputTokens: 8192,
416
- supportsTools: "probe",
417
- jsonReliability: "medium",
418
- instructionFollowing: "medium",
419
- tokenRatioEstimate: 3.8,
420
- concurrencyLimit: 2,
421
- cacheStrategy: "none",
422
- timeoutMultiplier: 1.35,
423
- singlePassTokenMultiplier: 0.9,
424
- multimodal: "metadata-only"
425
- };
426
- function getProviderCaps(provider) {
427
- if (PROVIDER_MAP[provider])
428
- return PROVIDER_MAP[provider];
429
- for (const { pattern, provider: key } of PROVIDER_ALIASES) {
430
- if (pattern.test(provider))
431
- return PROVIDER_MAP[key] ?? DEFAULT_CAPS;
432
- }
433
- return DEFAULT_CAPS;
434
- }
435
- class TokenCalibrationStore {
436
- maxEntries;
437
- factors = new Map;
438
- constructor(maxEntries = 128) {
439
- this.maxEntries = maxEntries;
440
- }
441
- clear() {
442
- this.factors.clear();
443
- }
444
- get(provider, model) {
445
- if (!provider)
446
- return 1;
447
- const exactKey = calibrationKey(provider, model);
448
- const exact = lruGet(this.factors, exactKey);
449
- if (exact !== undefined)
450
- return exact;
451
- return lruGet(this.factors, calibrationKey(provider)) ?? 1;
452
- }
453
- calibrate(estimated, actual, provider, model) {
454
- if (actual <= 0 || estimated <= 0 || !provider)
455
- return;
456
- const key = calibrationKey(provider, model);
457
- const prev = lruGet(this.factors, key) ?? 1;
458
- const target = prev * actual / estimated;
459
- const clamped = Math.max(TUNING.CALIBRATION_CLAMP_MIN, Math.min(TUNING.CALIBRATION_CLAMP_MAX, target));
460
- lruSet(this.factors, key, prev * TUNING.EMA_PREV + clamped * TUNING.EMA_SAMPLE, Math.max(1, this.maxEntries));
461
- }
462
- size() {
463
- return this.factors.size;
464
- }
465
- }
466
- var _fallbackCalibration = new TokenCalibrationStore;
467
- function calibrationKey(provider, model) {
468
- return model ? provider + "/" + model : provider + "/*";
469
- }
470
-
471
- // src/infra/llm-client.ts
472
- var _complete = null;
473
- var _completeSimple = null;
474
- var _stream = null;
475
- var _streamSimple = null;
476
- async function resolveComplete() {
477
- if (_complete)
478
- return _complete;
479
- const mod = await import("@earendil-works/pi-ai/compat");
480
- const fn = mod.complete;
481
- if (typeof fn !== "function")
482
- throw new Error("smart-compact: pi-ai /compat did not export complete()");
483
- _complete = fn;
484
- return fn;
485
- }
486
- async function resolveCompleteSimple() {
487
- if (_completeSimple)
488
- return _completeSimple;
489
- const mod = await import("@earendil-works/pi-ai/compat");
490
- const fn = mod.completeSimple;
491
- if (typeof fn !== "function")
492
- throw new Error("smart-compact: pi-ai /compat did not export completeSimple()");
493
- _completeSimple = fn;
494
- return fn;
495
- }
496
- async function resolveStream() {
497
- if (_stream)
498
- return _stream;
499
- const mod = await import("@earendil-works/pi-ai/compat");
500
- if (typeof mod.stream !== "function")
501
- throw new Error("smart-compact: pi-ai /compat did not export stream()");
502
- _stream = mod.stream;
503
- return _stream;
504
- }
505
- async function resolveStreamSimple() {
506
- if (_streamSimple)
507
- return _streamSimple;
508
- const mod = await import("@earendil-works/pi-ai/compat");
509
- if (typeof mod.streamSimple !== "function")
510
- throw new Error("smart-compact: pi-ai /compat did not export streamSimple()");
511
- _streamSimple = mod.streamSimple;
512
- return _streamSimple;
513
- }
514
- function isChatGptCodex(model) {
515
- if (model.api !== "openai-codex-responses")
516
- return false;
517
- return !model.baseUrl || model.baseUrl.includes("chatgpt.com");
518
- }
519
- function withCodexWireLimit(model, opts) {
520
- if (model.api !== "openai-codex-responses" || isChatGptCodex(model) || !opts.maxTokens)
521
- return opts;
522
- const previous = opts.onPayload;
523
- return {
524
- ...opts,
525
- onPayload: async (payload, requestModel) => {
526
- const transformed = await previous?.(payload, requestModel);
527
- const body = transformed ?? payload;
528
- return body && typeof body === "object" ? {
529
- ...body,
530
- max_output_tokens: opts.maxTokens
531
- } : body;
532
- }
533
- };
534
- }
535
- function resolveCodexWatchdogMs(maxTokens, configuredMs = 0) {
536
- if (configuredMs > 0)
537
- return configuredMs;
538
- return Math.min(90000, Math.max(15000, 1e4 + (maxTokens ?? 4096) * 8));
539
- }
540
- function streamedChars(event) {
541
- if (event.type === "text_delta" || event.type === "thinking_delta" || event.type === "toolcall_delta") {
542
- return event.delta.length;
543
- }
544
- return 0;
545
- }
546
- function assertSuccessful(message) {
547
- if (message.stopReason === "error" || message.stopReason === "aborted") {
548
- throw new Error(message.errorMessage || "LLM request failed");
549
- }
550
- return message;
551
- }
552
- function resolveProviderWatchdogMs(provider, maxTokens, configuredMs = 0) {
553
- const multiplier = provider && configuredMs <= 0 ? getProviderCaps(provider).timeoutMultiplier : 1;
554
- return Math.round(resolveCodexWatchdogMs(maxTokens, configuredMs) * multiplier);
555
- }
556
- async function withProviderDeadline(opts, invoke, provider) {
557
- if (opts.signal?.aborted)
558
- throw new Error("LLM request aborted before dispatch");
559
- const controller = new AbortController;
560
- const watchdogMs = resolveProviderWatchdogMs(provider, opts.maxTokens, opts.codexWatchdogMs);
561
- const abort = Promise.withResolvers();
562
- const abortFromCaller = () => {
563
- controller.abort(opts.signal?.reason);
564
- abort.reject(new Error("LLM request aborted by caller"));
565
- };
566
- opts.signal?.addEventListener("abort", abortFromCaller, { once: true });
567
- const timeout = Promise.withResolvers();
568
- const timer = setTimeout(() => {
569
- controller.abort("provider-watchdog");
570
- timeout.reject(new Error("Provider watchdog stopped generation after " + watchdogMs + "ms"));
571
- }, watchdogMs);
572
- if (typeof timer === "object" && "unref" in timer)
573
- timer.unref();
574
- try {
575
- return await Promise.race([
576
- invoke({ ...opts, signal: controller.signal }),
577
- abort.promise,
578
- timeout.promise
579
- ]);
580
- } finally {
581
- clearTimeout(timer);
582
- opts.signal?.removeEventListener("abort", abortFromCaller);
583
- }
584
- }
585
- async function completeChatGptCodex(model, body, opts) {
586
- const controller = new AbortController;
587
- const watchdogMs = resolveCodexWatchdogMs(opts.maxTokens, opts.codexWatchdogMs);
588
- let watchdogReason = null;
589
- let visibleChars = 0;
590
- const abortFromCaller = () => controller.abort(opts.signal?.reason);
591
- opts.signal?.addEventListener("abort", abortFromCaller, { once: true });
592
- if (opts.signal?.aborted)
593
- abortFromCaller();
594
- const timer = setTimeout(() => {
595
- watchdogReason = "time";
596
- controller.abort("codex-watchdog");
597
- }, watchdogMs);
598
- timer.unref?.();
599
- try {
600
- const limited = { ...opts, signal: controller.signal };
601
- const events = opts.reasoning === undefined ? (await resolveStream())(model, body, limited) : (await resolveStreamSimple())(model, body, limited);
602
- let final;
603
- for await (const event of events) {
604
- visibleChars += streamedChars(event);
605
- if (!watchdogReason && opts.maxTokens && visibleChars > opts.maxTokens * 3) {
606
- watchdogReason = "visible-output";
607
- controller.abort("codex-visible-output-cap");
608
- }
609
- if (event.type === "done")
610
- final = event.message;
611
- else if (event.type === "error")
612
- final = event.error;
613
- }
614
- if (watchdogReason) {
615
- throw new Error("Codex " + watchdogReason + " watchdog stopped generation after " + watchdogMs + "ms / " + visibleChars + " streamed chars");
616
- }
617
- if (!final)
618
- throw new Error("Codex stream ended without a final message");
619
- return assertSuccessful(final);
620
- } finally {
621
- clearTimeout(timer);
622
- opts.signal?.removeEventListener("abort", abortFromCaller);
623
- }
624
- }
625
- var rawLlmClient = {
626
- complete: async (model, body, originalOpts) => {
627
- const opts = withCodexWireLimit(model, originalOpts);
628
- return withProviderDeadline(opts, async (bounded) => {
629
- if (isChatGptCodex(model))
630
- return completeChatGptCodex(model, body, bounded);
631
- const response = bounded.reasoning === undefined ? await (await resolveComplete())(model, body, bounded) : await (await resolveCompleteSimple())(model, body, bounded);
632
- return assertSuccessful(response);
633
- }, model.provider);
634
- }
635
- };
636
- var defaultLlmClient = rawLlmClient;
637
- var _client = defaultLlmClient;
638
- function getLlmClient() {
639
- return _client;
640
- }
641
-
642
- // src/utils/cache.ts
643
- import fs2 from "fs";
644
-
645
- // src/utils/type-guards.ts
646
- function isRecord(value) {
647
- return typeof value === "object" && value !== null;
648
- }
649
- function isTextBlock(c) {
650
- return isRecord(c) && c.type === "text" && typeof c.text === "string";
651
- }
652
- function isToolCallBlock(c) {
653
- return isRecord(c) && c.type === "toolCall" && typeof c.name === "string" && isRecord(c.arguments);
654
- }
655
- var KNOWN_METHODS = new Set(["eesv", "single-pass", "heuristic"]);
656
- var KNOWN_PROFILES = new Set(["light", "balanced", "aggressive"]);
657
- var KNOWN_MODES = new Set(["balanced", "aggressive", "fast", "thorough"]);
658
-
659
- // src/utils/file-needles.ts
660
- var GENERIC_BASENAMES = new Set([
661
- "index.ts",
662
- "index.js",
663
- "index.tsx",
664
- "index.jsx",
665
- "types.ts",
666
- "helpers.ts",
667
- "utils.ts",
668
- "main.ts",
669
- "main.js",
670
- "mod.rs",
671
- "lib.rs",
672
- "__init__.py"
673
- ]);
674
- var MIN_BARE_BASENAME_LEN = 5;
675
- function normalizePath(filePath) {
676
- return filePath.replace(/\\/g, "/").replace(/^\.\//, "").toLowerCase();
677
- }
678
- function buildPathNeedles(filePath) {
679
- const parts = normalizePath(filePath).split("/").filter(Boolean);
680
- if (parts.length === 0)
681
- return [];
682
- const needles = [];
683
- const basename = parts[parts.length - 1];
684
- if (!GENERIC_BASENAMES.has(basename) && basename.length >= MIN_BARE_BASENAME_LEN) {
685
- needles.push(basename);
686
- }
687
- for (let j = parts.length - 2;j >= 0; j--) {
688
- needles.push(parts.slice(j).join("/"));
689
- }
690
- return needles;
691
- }
692
- var MAX_INDEXED_SUFFIX_CHARS = 1024;
693
- function buildPathNeedleOwnershipIndex(allPaths) {
694
- const owners = new Map;
695
- const normalizedPaths = allPaths.map(normalizePath);
696
- let hasUnindexedSuffixes = false;
697
- for (const normalized of normalizedPaths) {
698
- const suffixes = new Set([normalized]);
699
- for (let index = 0;index < normalized.length; index++) {
700
- if (normalized[index] !== "/")
701
- continue;
702
- if (normalized.length - index - 1 <= MAX_INDEXED_SUFFIX_CHARS) {
703
- suffixes.add(normalized.slice(index + 1));
704
- } else {
705
- hasUnindexedSuffixes = true;
706
- }
707
- }
708
- for (const suffix of suffixes) {
709
- owners.set(suffix, (owners.get(suffix) ?? 0) + 1);
710
- }
711
- }
712
- return { counts: owners, normalizedPaths, hasUnindexedSuffixes };
713
- }
714
- function buildUniquePathNeedlesFromIndex(filePath, owners) {
715
- return buildPathNeedles(filePath).filter((needle) => {
716
- if (!owners.hasUnindexedSuffixes)
717
- return owners.counts.get(needle) === 1;
718
- let count = 0;
719
- for (const candidate of owners.normalizedPaths) {
720
- if (candidate === needle || candidate.endsWith("/" + needle))
721
- count++;
722
- if (count > 1)
723
- return false;
724
- }
725
- return count === 1;
726
- });
727
- }
728
- var PATH_CANDIDATE_CHAR_RE = /[\w./-]/;
729
- function buildKnownPathReferenceIndex(knownPaths) {
730
- const segmentSuffixes = new Set;
731
- const boundarySuffixes = new Set;
732
- const normalizedPaths = [];
733
- let hasUnindexedSuffixes = false;
734
- for (const path of knownPaths) {
735
- const normalizedPath = normalizePath(path).replace(/^\/+/, "");
736
- if (!normalizedPath)
737
- continue;
738
- normalizedPaths.push(normalizedPath);
739
- segmentSuffixes.add(normalizedPath);
740
- for (let index = 0;index < normalizedPath.length; index++) {
741
- if (normalizedPath[index] === "/") {
742
- if (normalizedPath.length - index - 1 <= MAX_INDEXED_SUFFIX_CHARS) {
743
- segmentSuffixes.add(normalizedPath.slice(index + 1));
744
- } else {
745
- hasUnindexedSuffixes = true;
746
- }
747
- }
748
- if (index > 0 && !PATH_CANDIDATE_CHAR_RE.test(normalizedPath[index - 1])) {
749
- if (normalizedPath.length - index <= MAX_INDEXED_SUFFIX_CHARS) {
750
- boundarySuffixes.add(normalizedPath.slice(index));
751
- } else {
752
- hasUnindexedSuffixes = true;
753
- }
754
- }
755
- }
756
- }
757
- return {
758
- segmentSuffixes,
759
- sortedSegmentSuffixes: [...segmentSuffixes].sort(),
760
- boundarySuffixes,
761
- normalizedPaths,
762
- hasUnindexedSuffixes
763
- };
764
- }
765
- function matchesKnownPathReference(normalizedRef, normalizedPaths) {
766
- const pathShaped = normalizedRef.includes("/");
767
- return normalizedPaths.some((normalizedPath) => {
768
- if (normalizedPath === normalizedRef || normalizedPath.endsWith("/" + normalizedRef))
769
- return true;
770
- if (normalizedPath.endsWith(normalizedRef)) {
771
- const boundary = normalizedPath[normalizedPath.length - normalizedRef.length - 1];
772
- if (boundary && !PATH_CANDIDATE_CHAR_RE.test(boundary))
773
- return true;
774
- }
775
- if (!pathShaped)
776
- return false;
777
- return normalizedPath.startsWith(normalizedRef + "/") || normalizedPath.includes("/" + normalizedRef + "/");
778
- });
779
- }
780
- function sortedHasPrefix(values, prefix) {
781
- let low = 0;
782
- let high = values.length;
783
- while (low < high) {
784
- const middle = low + high >>> 1;
785
- if (values[middle] < prefix)
786
- low = middle + 1;
787
- else
788
- high = middle;
789
- }
790
- return values[low]?.startsWith(prefix) ?? false;
791
- }
792
- function isKnownPathReferenceInIndex(ref, index) {
793
- const normalizedRef = normalizePath(ref).replace(/^\/+/, "");
794
- if (!normalizedRef)
795
- return false;
796
- if (index.segmentSuffixes.has(normalizedRef) || index.boundarySuffixes.has(normalizedRef)) {
797
- return true;
798
- }
799
- if (normalizedRef.includes("/") && sortedHasPrefix(index.sortedSegmentSuffixes, normalizedRef + "/"))
800
- return true;
801
- return index.hasUnindexedSuffixes ? matchesKnownPathReference(normalizedRef, index.normalizedPaths) : false;
802
- }
803
-
804
- // src/domain/tool-semantics.ts
805
- var PATH_KEYS = [
806
- "path",
807
- "file_path",
808
- "filePath",
809
- "filename",
810
- "file",
811
- "target_file",
812
- "file_uri",
813
- "absolute_path"
814
- ];
815
- var PAYLOAD_KEYS = [
816
- "content",
817
- "newText",
818
- "oldText",
819
- "new_str",
820
- "old_str",
821
- "new_string",
822
- "old_string",
823
- "edits",
824
- "patch",
825
- "replacement"
826
- ];
827
- var COMMAND_KEYS = ["command", "cmd", "script"];
828
- function hasPresent(args, keys) {
829
- return keys.some((k) => args[k] != null);
830
- }
831
- function extractToolPath(args) {
832
- if (!args || typeof args !== "object")
833
- return;
834
- const a = args;
835
- for (const k of PATH_KEYS) {
836
- const v = a[k];
837
- if (typeof v === "string" && v.length > 0)
838
- return v;
839
- }
840
- return;
841
- }
842
- function normalizeToolName(name) {
843
- if (typeof name !== "string")
844
- return "";
845
- return name.replace(/^functions[.:/_-]+/i, "").replace(/([a-z0-9])([A-Z])/g, "$1_$2").toLowerCase().replace(/[^a-z0-9]+/g, "_").replace(/^_+|_+$/g, "");
846
- }
847
- function nameHas(name, hints) {
848
- const words = name.split("_");
849
- return hints.some((hint) => words.includes(hint));
850
- }
851
- function classifyToolOperation(args, toolName) {
852
- const a = args && typeof args === "object" ? args : {};
853
- const name = normalizeToolName(toolName);
854
- const hasPath = extractToolPath(a) !== undefined;
855
- if (hasPath && hasPresent(a, PAYLOAD_KEYS))
856
- return "mutate";
857
- if (hasPresent(a, COMMAND_KEYS))
858
- return "execute";
859
- if (hasPath && hasPresent(a, ["text"]) && nameHas(name, ["write", "edit", "patch", "replace", "append", "create", "update", "insert"]))
860
- return "mutate";
861
- if (hasPath && nameHas(name, ["delete", "remove", "unlink"]))
862
- return "delete";
863
- if (hasPresent(a, ["pattern", "query", "glob"]) || nameHas(name, ["grep", "search", "find", "glob", "rg"]))
864
- return "search";
865
- if (nameHas(name, ["list", "ls", "tree"]))
866
- return "list";
867
- if (hasPath || nameHas(name, ["read"]))
868
- return "read";
869
- return "unknown";
870
- }
871
-
872
- // src/utils/logger.ts
873
- var DEBUG = process.env.DEBUG?.includes("smart-compact") ?? false;
874
- function ts() {
875
- return new Date().toISOString();
876
- }
877
- function warn(msg, err) {
878
- const detail = err instanceof Error ? err.message : err ?? "";
879
- console.error(ts() + " " + LOG_PREFIX + " " + msg + (detail ? ": " + detail : ""));
880
- }
881
-
882
- // src/infra/fs.ts
883
- import fs from "fs";
884
- function readJsonlTail(target, limit, maxBytes = 512 * 1024) {
885
- if (limit <= 0 || !fs.existsSync(target))
886
- return [];
887
- const stat = fs.statSync(target);
888
- const length = Math.min(stat.size, maxBytes);
889
- const buffer = Buffer.alloc(length);
890
- const fd = fs.openSync(target, "r");
891
- try {
892
- fs.readSync(fd, buffer, 0, length, stat.size - length);
893
- } finally {
894
- fs.closeSync(fd);
895
- }
896
- let text = buffer.toString("utf8");
897
- if (stat.size > length) {
898
- const newline = text.indexOf(`
899
- `);
900
- text = newline >= 0 ? text.slice(newline + 1) : "";
901
- }
902
- const values = [];
903
- for (const line of text.split(`
904
- `)) {
905
- if (!line)
906
- continue;
907
- try {
908
- values.push(JSON.parse(line));
909
- } catch {}
910
- }
911
- return values.slice(-limit);
912
- }
913
-
914
- // src/infra/paths.ts
915
- import path from "path";
916
- import os from "os";
917
- function home() {
918
- const configured = process.env.HOME?.trim() || process.env.USERPROFILE?.trim();
919
- return configured || os.homedir();
920
- }
921
- function piAgentDir() {
922
- return path.join(home(), ".pi", "agent");
923
- }
924
- function cacheDir() {
925
- return path.join(piAgentDir(), ".cache");
926
- }
927
- function smartCompactCacheDir() {
928
- return path.join(cacheDir(), "smart-compact");
929
- }
930
- function metricsLogFile() {
931
- return path.join(cacheDir(), "compact-metrics.jsonl");
932
- }
933
- function damageReportsFile() {
934
- return path.join(smartCompactCacheDir(), "damage-reports.jsonl");
935
- }
936
- // src/utils/helpers.ts
937
- function normalizeFactKey(text) {
938
- return text.toLowerCase().replace(/\s+/g, " ").trim();
939
- }
940
-
941
- // src/utils/file-ref-detect.ts
942
- var CODE_EXT_RE = /\.(ts|tsx|js|jsx|mjs|cjs|rs|py|go|java|rb|cs|cpp|c|h|hpp|swift|kt|scala|php|css|scss|html|json|yaml|yml|toml|md|mdx|sh|sql|tf|ini|env|lock|gradle|xml)$/i;
943
- var VERSION_RE = /^v?\d+(?:\.\d+)+(?:[-+][\w.-]+)?$/i;
944
- function isAsciiWordCode(code) {
945
- return code >= 48 && code <= 57 || code >= 65 && code <= 90 || code === 95 || code >= 97 && code <= 122;
946
- }
947
- function isCandidateCode(code) {
948
- return isAsciiWordCode(code) || code === 45 || code === 46 || code === 47;
949
- }
950
- function isLikelyFileRef(candidate) {
951
- if (candidate.startsWith("//") || VERSION_RE.test(candidate))
952
- return false;
953
- if (candidate.includes("/")) {
954
- const last = candidate.split("/").pop() ?? "";
955
- return last.length > 0 && !VERSION_RE.test(last);
956
- }
957
- return CODE_EXT_RE.test(candidate);
958
- }
959
- function extractFileRefs(summary) {
960
- const refs = [];
961
- let cursor = 0;
962
- while (cursor < summary.length) {
963
- while (cursor < summary.length && !isCandidateCode(summary.charCodeAt(cursor)))
964
- cursor++;
965
- const runStart = cursor;
966
- while (cursor < summary.length && isCandidateCode(summary.charCodeAt(cursor)))
967
- cursor++;
968
- const runEnd = cursor;
969
- let extensionDot = -1;
970
- for (let index = runStart + 1;index + 1 < runEnd; index++) {
971
- if (summary.charCodeAt(index) === 46 && isAsciiWordCode(summary.charCodeAt(index + 1))) {
972
- extensionDot = index;
973
- }
974
- }
975
- if (extensionDot < 0)
976
- continue;
977
- let matchEnd = extensionDot + 2;
978
- while (matchEnd < runEnd && isAsciiWordCode(summary.charCodeAt(matchEnd))) {
979
- matchEnd++;
980
- }
981
- const candidate = summary.slice(runStart, matchEnd);
982
- if (/[\\/]/.test(summary[matchEnd] ?? ""))
983
- continue;
984
- if (isLikelyFileRef(candidate))
985
- refs.push(candidate);
986
- }
987
- return refs;
988
- }
989
-
990
- // src/domain/summary-parse.ts
991
- import { createHash } from "crypto";
992
-
993
- // src/domain/summary-schema.ts
994
- function classifyHeading(raw) {
995
- const text = raw.replace(/^#+\s*/, "").replace(/[:\s]+$/, "").trim().toLowerCase();
996
- if (!text)
997
- return "unknown";
998
- if (text === "goal" || text === "goals" || text === "objective" || text === "objectives")
999
- return "goal";
1000
- if (text.startsWith("constraint") || text.includes("preference"))
1001
- return "constraints";
1002
- if (text === "progress" || text === "status")
1003
- return "progress";
1004
- if (text.includes("key decision") || text === "decisions")
1005
- return "decisions";
1006
- if (text.includes("file") && text.includes("modif"))
1007
- return "files-modified";
1008
- if (text.includes("file") && (text.includes("read") || text.includes("viewed")))
1009
- return "files-read";
1010
- if (text.includes("file") && (text.includes("delet") || text.includes("remov")))
1011
- return "files-deleted";
1012
- if (text.includes("next step") || text === "next actions")
1013
- return "next-steps";
1014
- if (text.includes("critical context") || text === "important context")
1015
- return "critical-context";
1016
- if (text === "topics" || text.includes("topics covered"))
1017
- return "topics";
1018
- if (text.includes("open loop") || text.includes("unresolved"))
1019
- return "open-loops";
1020
- if (text.includes("changes since") || text === "changes")
1021
- return "changes";
1022
- if (text.includes("verification"))
1023
- return "verification-note";
1024
- return "unknown";
1025
- }
1026
-
1027
- // src/domain/summary-parse.ts
1028
- var HEADING_RE = /^(#{1,3})\s+(.+?)\s*$/;
1029
- function summaryEvidenceLine(value, maxLength) {
1030
- return value.replace(/[\u0000-\u0008\u000B\u000C\u000E-\u001F\u007F]/g, " ").replace(/\s+/g, " ").trim().replace(/^(?:(?:#{1,6}|[-*+]|>)\s+)+/, "").slice(0, maxLength).trim();
1031
- }
1032
- function summaryPathLine(value) {
1033
- return JSON.stringify(value);
1034
- }
1035
- function compactPathLine(value, maxLength, digest) {
1036
- const minimal = JSON.stringify("#" + digest);
1037
- if (minimal.length >= maxLength)
1038
- return minimal;
1039
- const chars = Array.from(value.replace(/\\/g, "/"));
1040
- let low = 0;
1041
- let high = chars.length;
1042
- let best = minimal;
1043
- while (low <= high) {
1044
- const length = Math.floor((low + high) / 2);
1045
- const candidate = JSON.stringify("\u2026/" + chars.slice(-length).join("") + "#" + digest);
1046
- if (candidate.length <= maxLength) {
1047
- best = candidate;
1048
- low = length + 1;
1049
- } else {
1050
- high = length - 1;
1051
- }
1052
- }
1053
- return best;
1054
- }
1055
- function buildSummaryPathEvidence(paths, budgetTokens = PROFILES.balanced.summaryBudgetTokens) {
1056
- const unique = Array.from(new Set(paths.filter(Boolean)));
1057
- if (!unique.length)
1058
- return new Map;
1059
- const full = unique.map((path) => [path, summaryPathLine(path)]);
1060
- const minimumPerLine = JSON.stringify("#" + "x".repeat(12)).length + 3;
1061
- const budgetChars = Math.max(unique.length * minimumPerLine, Math.min(20000, Math.max(4000, Math.floor(budgetTokens * 2))));
1062
- if (full.reduce((total, [, line]) => total + line.length + 3, 0) <= budgetChars) {
1063
- return new Map(full);
1064
- }
1065
- const digests = new Map;
1066
- const owners = new Map;
1067
- for (const path of unique) {
1068
- const fullDigest = createHash("sha256").update(path).digest("base64url");
1069
- let digest = fullDigest.slice(0, 12);
1070
- const owner = owners.get(digest);
1071
- if (owner && owner !== path) {
1072
- digest = fullDigest;
1073
- digests.set(owner, createHash("sha256").update(owner).digest("base64url"));
1074
- }
1075
- owners.set(digest, path);
1076
- digests.set(path, digest);
1077
- }
1078
- const perPath = Math.max(JSON.stringify("#" + "x".repeat(12)).length, Math.floor((budgetChars - unique.length * 3) / unique.length));
1079
- return new Map(unique.map((path) => [
1080
- path,
1081
- compactPathLine(path, perPath, digests.get(path) ?? "")
1082
- ]));
1083
- }
1084
- function mergeBodies(first, second) {
1085
- const seen = new Set;
1086
- return [first, second].filter(Boolean).flatMap((body) => body.split(`
1087
- `)).filter((line) => seen.has(line) ? false : (seen.add(line), true)).join(`
1088
- `).trim();
1089
- }
1090
- function parseSummary(markdown) {
1091
- const sections = [];
1092
- const lines = markdown.split(`
1093
- `);
1094
- let currentHeading = "";
1095
- let currentKind = "unknown";
1096
- let bodyLines = [];
1097
- let fence = null;
1098
- let started = false;
1099
- const flush = () => {
1100
- if (!started)
1101
- return;
1102
- const body = bodyLines.join(`
1103
- `).trim();
1104
- const existing = currentKind === "unknown" ? undefined : sections.find((s) => s.kind === currentKind);
1105
- if (existing)
1106
- existing.body = mergeBodies(existing.body, body);
1107
- else
1108
- sections.push({
1109
- kind: currentKind,
1110
- heading: currentHeading.trim(),
1111
- body
1112
- });
1113
- };
1114
- for (const line of lines) {
1115
- const fenceMatch = line.match(/^\s{0,3}(`{3,}|~{3,})(.*)$/);
1116
- if (fenceMatch) {
1117
- const marker = fenceMatch[1][0];
1118
- const markerLength = fenceMatch[1].length;
1119
- if (!fence) {
1120
- fence = { marker, length: markerLength };
1121
- } else if (marker === fence.marker && markerLength >= fence.length && !fenceMatch[2].trim()) {
1122
- fence = null;
1123
- }
1124
- if (started)
1125
- bodyLines.push(line);
1126
- continue;
1127
- }
1128
- if (!fence) {
1129
- const heading = line.match(HEADING_RE);
1130
- if (heading) {
1131
- const kind = classifyHeading(heading[2]);
1132
- if (heading[1].length <= 2 || kind !== "unknown") {
1133
- flush();
1134
- currentHeading = "## " + heading[2].trim();
1135
- currentKind = kind;
1136
- bodyLines = [];
1137
- started = true;
1138
- continue;
1139
- }
1140
- }
1141
- }
1142
- if (started)
1143
- bodyLines.push(line);
1144
- }
1145
- flush();
1146
- return { sections };
1147
- }
1148
- function findSection(summary, kind) {
1149
- const parsed = typeof summary === "string" ? parseSummary(summary) : summary;
1150
- return parsed.sections.find((s) => s.kind === kind);
1151
- }
1152
-
1153
- // src/utils/extraction.ts
1154
- function nestedToolCallId(wrapperId, messageIndex, toolIndex, nestedId) {
1155
- return typeof nestedId === "string" ? nestedId : wrapperId ? wrapperId + "_" + toolIndex : ID_PREFIX.MULTI_TOOL_USE_SYNTHETIC + messageIndex + "_" + toolIndex;
1156
- }
1157
- function flattenToolCallBlock(b) {
1158
- if (!isToolCallBlock(b))
1159
- return [];
1160
- if (b.name === "multi_tool_use.parallel" && Array.isArray(b.arguments?.tool_uses)) {
1161
- return b.arguments.tool_uses.map((u) => {
1162
- const recipient = u?.recipient_name ?? "";
1163
- return {
1164
- name: recipient.replace(/^functions\./, ""),
1165
- id: u?.id ?? undefined,
1166
- arguments: u?.parameters ?? {}
1167
- };
1168
- });
1169
- }
1170
- return [{ name: b.name, id: b.id, arguments: b.arguments }];
1171
- }
1172
- function extractText(content) {
1173
- if (typeof content === "string")
1174
- return content;
1175
- if (!Array.isArray(content))
1176
- return "";
1177
- return content.map((b) => {
1178
- if (typeof b === "string")
1179
- return b;
1180
- if (isTextBlock(b))
1181
- return b.text;
1182
- return "";
1183
- }).join("");
1184
- }
1185
- function buildToolCallIndex(msgs) {
1186
- const idx = new Map;
1187
- for (let i = 0;i < msgs.length; i++) {
1188
- const m = msgs[i];
1189
- if (m.role !== "assistant")
1190
- continue;
1191
- const blocks = Array.isArray(m.content) ? m.content : [];
1192
- for (const b of blocks) {
1193
- if (!isToolCallBlock(b))
1194
- continue;
1195
- if (b.id) {
1196
- idx.set(b.id, { name: b.name, arguments: b.arguments, msgIndex: i });
1197
- }
1198
- if (b.name === "multi_tool_use.parallel" && Array.isArray(b.arguments?.tool_uses)) {
1199
- const nested = flattenToolCallBlock(b);
1200
- for (let t = 0;t < nested.length; t++) {
1201
- const tool = nested[t];
1202
- const id = nestedToolCallId(b.id, i, t, tool.id);
1203
- idx.set(id, {
1204
- name: tool.name,
1205
- arguments: tool.arguments,
1206
- msgIndex: i
1207
- });
1208
- }
1209
- }
1210
- }
1211
- }
1212
- return idx;
1213
- }
1214
- var CONSTRAINT_PATTERNS = [
1215
- {
1216
- re: /\b(?:must|need|require|has to|important)\b.*\b(?:be|use|have|include|support)\b/i,
1217
- cat: "requirement",
1218
- conf: TUNING.CONFIDENCE_HIGH
1219
- },
1220
- {
1221
- re: /\b(?:don't|never|avoid|shouldn't|must not|do not|no\s+(?:need|want))\b/i,
1222
- cat: "prohibition",
1223
- conf: TUNING.CONFIDENCE_MEDIUM
1224
- },
1225
- {
1226
- re: /\b(?:prefer|like|want|would rather|should)\b.*\b(?:use|be|have|with)\b/i,
1227
- cat: "preference",
1228
- conf: TUNING.CONFIDENCE_LOW
1229
- },
1230
- {
1231
- re: /(?<![A-Za-z0-9_])(?:yapma|kullanma|sak\u0131n|sak\u0131nha|asla(?:\s+(?:kullanma|yapma|getirme))?|bunu yapma)(?![A-Za-z0-9_])/iu,
1232
- cat: "prohibition",
1233
- conf: TUNING.CONFIDENCE_MEDIUM
1234
- },
1235
- {
1236
- re: /(?<![A-Za-z0-9_])(?:kritik|kritikal|\u00F6nemli|onemli|\u015Fart|sart|zorunlu|\u015Fart ko\u015Ful|\u00F6nemli \u015Fart|kesinlikle|kesinlikle \u015Fart|b\u00F6yle olsun|b\u00F6yle yap\u0131n|\u015F\u00F6yle olsun|\u015F\u00F6yle yap\u0131n)(?![A-Za-z0-9_])/iu,
1237
- cat: "requirement",
1238
- conf: TUNING.CONFIDENCE_MEDIUM
1239
- },
1240
- {
1241
- re: /(?<![A-Za-z0-9_])(?:tercih|isterim|olsun|kullanal\u0131m|yapal\u0131m|istiyorum)(?![A-Za-z0-9_])/iu,
1242
- cat: "preference",
1243
- conf: TUNING.CONFIDENCE_LOW
1244
- }
1245
- ];
1246
- function isDiagnosticConstraintText(text) {
1247
- const candidate = text.replace(/^\s*[-*]\s+/, "").trim();
1248
- return /^(?:\[[^\]]+\]\s*)?(?:npm\s+(?:error|warn|notice|audit|verbose|info)\b|(?:rg|grep):|command exited\b)/i.test(candidate);
1249
- }
1250
- var OWN_OUTPUT_RE = /(?:verification stopped apply|yield check stopped apply|smart compact failed|do not bypass verification|conversation unchanged|review \/smart-compact metrics|restart pi with debug=smart-compact)/i;
1251
- var COMPLETED_CHECKLIST_RE = /^\[[xX\u2713\u2714]\]\s*\S/;
1252
- function isNonLiveConstraintText(text) {
1253
- const candidate = text.replace(/^\s*[-*]\s+/, "").trim();
1254
- return isDiagnosticConstraintText(candidate) || isCompactionStatusText(candidate) || OWN_OUTPUT_RE.test(candidate) || COMPLETED_CHECKLIST_RE.test(candidate);
1255
- }
1256
- function isCompactionStatusText(text) {
1257
- const candidate = summaryEvidenceLine(text, TRUNC.MESSAGE).replace(/^["'`]+/, "").trim();
1258
- return /^(?:EESV Compact\b|Smart compact (?:skipped|prepared|run finished)\b|Auto-compacting\b)/i.test(candidate);
1259
- }
1260
-
1261
- // src/infra/clock.ts
1262
- var systemClock = {
1263
- now: () => Date.now()
1264
- };
1265
-
1266
- // src/infra/services.ts
1267
- import crypto from "crypto";
1268
-
1269
- // src/domain/scrub.ts
1270
- var SECRET_PATTERNS = [
1271
- {
1272
- kind: "private-key",
1273
- regex: /-----BEGIN [A-Z ]*PRIVATE KEY-----[\s\S]*?-----END [A-Z ]*PRIVATE KEY-----/g
1274
- },
1275
- { kind: "aws-access-key", regex: /\bAKIA[0-9A-Z]{16}\b/g },
1276
- { kind: "google-api-key", regex: /\bAIza[0-9A-Za-z_-]{30,}\b/g },
1277
- { kind: "stripe-key", regex: /\b[rs]k_(?:live|test)_[0-9A-Za-z]{16,}\b/g },
1278
- { kind: "gitlab-token", regex: /\bglpat-[0-9A-Za-z_-]{20,}\b/g },
1279
- { kind: "npm-token", regex: /\bnpm_[0-9A-Za-z]{30,}\b/g },
1280
- { kind: "github-token", regex: /\bgh[pousr]_[A-Za-z0-9]{20,}\b/g },
1281
- { kind: "api-key", regex: /\bsk-(?:ant-)?[A-Za-z0-9_-]{20,}\b/g },
1282
- { kind: "slack-token", regex: /\bxox[baprs]-[A-Za-z0-9-]{10,}\b/g },
1283
- {
1284
- kind: "jwt",
1285
- regex: /\beyJ[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]{10,}\b/g
1286
- },
1287
- {
1288
- kind: "bearer-token",
1289
- regex: /\bBearer\s+[A-Za-z0-9._~+/-]{12,}=*/gi,
1290
- replacement: () => "Bearer [REDACTED:bearer-token]"
1291
- },
1292
- {
1293
- kind: "connection-password",
1294
- regex: /\b([a-z][a-z0-9+.-]*:\/\/[^:\s/@]+:)[^@\s/]+(@)/gi,
1295
- replacement: (prefix, suffix) => prefix + "[REDACTED:password]" + suffix
1296
- },
1297
- {
1298
- kind: "credential",
1299
- regex: /\b((?:[A-Za-z0-9]+[_-])*(?:api[_-]?key|access[_-]?token|auth[_-]?token|token|password|passwd|secret(?:[_-]?(?:access)?[_-]?key)?|client[_-]?secret)(?:[_-][A-Za-z0-9]+)*)\s*([:=])\s*["']?([^\s"']{16,})["']?/gi,
1300
- replacement: (name, separator, value, match) => /^[A-Za-z_$][\w$]*(?:\.[A-Za-z_$][\w$]*)+$/.test(value) ? match : name + separator + "[REDACTED:credential]"
1301
- }
1302
- ];
1303
- function passesLuhn(candidate) {
1304
- const digits = candidate.replace(/\D/g, "");
1305
- if (digits.length < 13 || digits.length > 19)
1306
- return false;
1307
- if (/^(\d)\1+$/.test(digits))
1308
- return false;
1309
- let sum = 0, double = false;
1310
- for (let i = digits.length - 1;i >= 0; i--) {
1311
- let d = digits.charCodeAt(i) - 48;
1312
- if (double) {
1313
- d *= 2;
1314
- if (d > 9)
1315
- d -= 9;
1316
- }
1317
- sum += d;
1318
- double = !double;
1319
- }
1320
- return sum % 10 === 0;
1321
- }
1322
- var PII_PATTERNS = [
1323
- { kind: "email", regex: /\b[A-Z0-9._%+-]+@[A-Z0-9.-]+\.[A-Z]{2,}\b/gi },
1324
- {
1325
- kind: "payment-card",
1326
- regex: /\b(?:\d[ -]*?){13,19}\b/g,
1327
- replacement: (candidate) => passesLuhn(candidate) ? "[REDACTED:payment-card]" : candidate
1328
- },
1329
- { kind: "phone", regex: /(?<![\w.])(?:\+?\d[\d ()-]{8,}\d)(?![\w.])/g }
1330
- ];
1331
- function redact(text, patterns) {
1332
- const counts = new Map;
1333
- let value = text;
1334
- for (const pattern of patterns) {
1335
- value = value.replace(pattern.regex, (...args) => {
1336
- const match = String(args[0]);
1337
- let replacement = "[REDACTED:" + pattern.kind + "]";
1338
- if (pattern.replacement) {
1339
- const groups = args.slice(1, -2).map(String);
1340
- replacement = pattern.replacement(...groups, match);
1341
- }
1342
- if (replacement === match)
1343
- return match;
1344
- counts.set(pattern.kind, (counts.get(pattern.kind) ?? 0) + 1);
1345
- return replacement;
1346
- });
1347
- }
1348
- return {
1349
- value,
1350
- findings: [...counts].map(([kind, count]) => ({ kind, count }))
1351
- };
1352
- }
1353
- function mergeFindings(target, findings) {
1354
- for (const finding of findings)
1355
- target.set(finding.kind, (target.get(finding.kind) ?? 0) + finding.count);
1356
- }
1357
- var SECRET_KEY_NAMES = {
1358
- api_key: true,
1359
- apikey: true,
1360
- access_token: true,
1361
- auth_token: true,
1362
- authorization: true,
1363
- password: true,
1364
- passwd: true,
1365
- secret: true,
1366
- secret_key: true,
1367
- secret_access_key: true,
1368
- client_secret: true,
1369
- private_key: true,
1370
- database_url: true,
1371
- connection_string: true,
1372
- token: true,
1373
- refresh_token: true,
1374
- session_token: true,
1375
- credential: true,
1376
- credentials: true,
1377
- cookie: true,
1378
- set_cookie: true,
1379
- otp: true,
1380
- one_time_password: true,
1381
- passcode: true
1382
- };
1383
- function normalizeObjectKey(key) {
1384
- return key.replace(/([a-z0-9])([A-Z])/g, "$1_$2").toLowerCase().replace(/[^a-z0-9]+/g, "_").replace(/^_+|_+$/g, "");
1385
- }
1386
- function isSecretBearingKey(key) {
1387
- const normalized = normalizeObjectKey(key);
1388
- if (SECRET_KEY_NAMES[normalized])
1389
- return true;
1390
- return /(?:^|_)(?:api_key|access_token|auth_token|password|passwd|secret_access_key|client_secret|private_key|refresh_token|session_token|one_time_password|passcode)(?:_|$)/.test(normalized);
1391
- }
1392
-
1393
- class SecretScrubber {
1394
- secretsEnabled;
1395
- piiEnabled;
1396
- total = 0;
1397
- constructor(secretsEnabled = true, piiEnabled = false) {
1398
- this.secretsEnabled = secretsEnabled;
1399
- this.piiEnabled = piiEnabled;
1400
- }
1401
- scrubText(text) {
1402
- let value = text;
1403
- const findings = new Map;
1404
- if (this.secretsEnabled) {
1405
- const result = redact(value, SECRET_PATTERNS);
1406
- value = result.value;
1407
- mergeFindings(findings, result.findings);
1408
- }
1409
- if (this.piiEnabled) {
1410
- const result = redact(value, PII_PATTERNS);
1411
- value = result.value;
1412
- mergeFindings(findings, result.findings);
1413
- }
1414
- const merged = [...findings].map(([kind, count]) => ({ kind, count }));
1415
- this.total += merged.reduce((sum, finding) => sum + finding.count, 0);
1416
- return { value, findings: merged };
1417
- }
1418
- scrubValue(input) {
1419
- const findings = new Map;
1420
- const seen = new WeakMap;
1421
- const recordCredential = () => {
1422
- findings.set("credential", (findings.get("credential") ?? 0) + 1);
1423
- this.total++;
1424
- };
1425
- const visit = (value) => {
1426
- if (typeof value === "string") {
1427
- const result = this.scrubText(value);
1428
- mergeFindings(findings, result.findings);
1429
- return result.value;
1430
- }
1431
- if (value == null || typeof value !== "object")
1432
- return value;
1433
- const cached = seen.get(value);
1434
- if (cached !== undefined)
1435
- return cached;
1436
- if (Array.isArray(value)) {
1437
- const output = [];
1438
- seen.set(value, output);
1439
- for (const item of value)
1440
- output.push(visit(item));
1441
- return output;
1442
- }
1443
- const output = {};
1444
- seen.set(value, output);
1445
- for (const [key, item] of Object.entries(value)) {
1446
- const carriesSecret = typeof item === "string" && item.length >= 8;
1447
- if (this.secretsEnabled && isSecretBearingKey(key) && carriesSecret) {
1448
- output[key] = "[REDACTED:credential]";
1449
- recordCredential();
1450
- } else {
1451
- output[key] = visit(item);
1452
- }
1453
- }
1454
- return output;
1455
- };
1456
- const value = visit(input);
1457
- return {
1458
- value,
1459
- findings: [...findings].map(([kind, count]) => ({ kind, count }))
1460
- };
1461
- }
1462
- count() {
1463
- return this.total;
1464
- }
1465
- }
1466
-
1467
- // src/infra/services.ts
1468
- class ToolSupportCache {
1469
- ttlMs;
1470
- maxEntries;
1471
- entries = new Map;
1472
- constructor(ttlMs = ONE_HOUR_MS, maxEntries = 128) {
1473
- this.ttlMs = ttlMs;
1474
- this.maxEntries = maxEntries;
1475
- }
1476
- get(key, now) {
1477
- const entry = this.entries.get(key);
1478
- if (!entry)
1479
- return;
1480
- if (now - entry.timestamp > this.ttlMs) {
1481
- this.entries.delete(key);
1482
- return;
1483
- }
1484
- this.entries.delete(key);
1485
- this.entries.set(key, entry);
1486
- return entry.result;
1487
- }
1488
- set(key, value, now) {
1489
- this.entries.delete(key);
1490
- this.entries.set(key, { result: value, timestamp: now });
1491
- while (this.entries.size > Math.max(1, this.maxEntries)) {
1492
- const oldest = this.entries.keys().next().value;
1493
- if (oldest === undefined)
1494
- break;
1495
- this.entries.delete(oldest);
1496
- }
1497
- }
1498
- clear() {
1499
- this.entries.clear();
1500
- }
1501
- size() {
1502
- return this.entries.size;
1503
- }
1504
- }
1505
-
1506
- class MetricsSink {
1507
- buf = [];
1508
- maxEntries;
1509
- constructor(maxEntries = METRICS_BUFFER_MAX) {
1510
- this.maxEntries = maxEntries;
1511
- }
1512
- record(metric) {
1513
- this.buf.push(metric);
1514
- if (this.buf.length > this.maxEntries) {
1515
- this.buf.splice(0, this.buf.length - Math.floor(this.maxEntries / 2));
1516
- }
1517
- }
1518
- snapshot() {
1519
- return [...this.buf];
1520
- }
1521
- clear() {
1522
- this.buf.length = 0;
1523
- }
1524
- summary() {
1525
- const n = this.buf.length;
1526
- if (!n)
1527
- return { totalCalls: 0, totalInput: 0, totalOutput: 0, totalCacheHit: 0, totalCacheWrite: 0, avgLatency: 0, cacheHitRate: 0 };
1528
- let totalInput = 0, totalOutput = 0, totalCacheHit = 0, totalCacheWrite = 0, totalLatency = 0;
1529
- for (const m of this.buf) {
1530
- totalInput += m.inputTokens;
1531
- totalOutput += m.outputTokens;
1532
- totalCacheHit += m.cacheHitTokens;
1533
- totalCacheWrite += m.cacheWriteTokens ?? 0;
1534
- totalLatency += m.latencyMs;
1535
- }
1536
- const promptInput = totalInput + totalCacheHit + totalCacheWrite;
1537
- const cacheHitRate = promptInput > 0 ? totalCacheHit / promptInput : 0;
1538
- return {
1539
- totalCalls: n,
1540
- totalInput,
1541
- totalOutput,
1542
- totalCacheHit,
1543
- totalCacheWrite,
1544
- avgLatency: Math.round(totalLatency / n),
1545
- cacheHitRate
1546
- };
1547
- }
1548
- }
1549
-
1550
- class BudgetExceededError extends Error {
1551
- reason;
1552
- constructor(reason) {
1553
- super("Smart Compact " + reason + " budget exhausted");
1554
- this.reason = reason;
1555
- this.name = "BudgetExceededError";
1556
- }
1557
- }
1558
-
1559
- class BudgetGuard {
1560
- maxCalls;
1561
- maxLatencyMs;
1562
- clock;
1563
- maxInputTokens;
1564
- maxOutputTokens;
1565
- calls = 0;
1566
- inputTokens = 0;
1567
- outputTokens = 0;
1568
- reservedOutputTokens = 0;
1569
- startedAt;
1570
- lastReason = null;
1571
- constructor(maxCalls = 0, maxLatencyMs = 0, clock = systemClock, maxInputTokens = 0, maxOutputTokens = 0) {
1572
- this.maxCalls = maxCalls;
1573
- this.maxLatencyMs = maxLatencyMs;
1574
- this.clock = clock;
1575
- this.maxInputTokens = maxInputTokens;
1576
- this.maxOutputTokens = maxOutputTokens;
1577
- this.startedAt = clock.now();
1578
- }
1579
- reserveCall(estimatedInputTokens = 0, expectedOutputTokens = 0) {
1580
- if (this.maxLatencyMs > 0 && this.clock.now() - this.startedAt >= this.maxLatencyMs) {
1581
- this.lastReason = "latency";
1582
- throw new BudgetExceededError("latency");
1583
- }
1584
- if (this.maxCalls > 0 && this.calls >= this.maxCalls) {
1585
- this.lastReason = "calls";
1586
- throw new BudgetExceededError("calls");
1587
- }
1588
- const outputReservation = Math.max(0, expectedOutputTokens);
1589
- if (this.maxInputTokens > 0 && this.inputTokens + estimatedInputTokens > this.maxInputTokens || this.maxOutputTokens > 0 && (this.outputTokens + this.reservedOutputTokens >= this.maxOutputTokens || this.outputTokens + this.reservedOutputTokens + outputReservation > this.maxOutputTokens)) {
1590
- this.lastReason = "tokens";
1591
- throw new BudgetExceededError("tokens");
1592
- }
1593
- this.calls++;
1594
- this.inputTokens += estimatedInputTokens;
1595
- this.reservedOutputTokens += outputReservation;
1596
- return outputReservation;
1597
- }
1598
- reconcileInput(estimated, actual) {
1599
- this.inputTokens = Math.max(0, this.inputTokens - estimated + actual);
1600
- if (this.maxInputTokens > 0 && this.inputTokens >= this.maxInputTokens)
1601
- this.lastReason = "tokens";
1602
- }
1603
- reconcileOutput(reserved, actual) {
1604
- this.reservedOutputTokens = Math.max(0, this.reservedOutputTokens - Math.max(0, reserved));
1605
- this.outputTokens += Math.max(0, actual);
1606
- if (this.maxOutputTokens > 0 && this.outputTokens + this.reservedOutputTokens >= this.maxOutputTokens)
1607
- this.lastReason = "tokens";
1608
- }
1609
- commitFailedOutput(reserved) {
1610
- const amount = Math.max(0, reserved);
1611
- this.reservedOutputTokens = Math.max(0, this.reservedOutputTokens - amount);
1612
- this.outputTokens += amount;
1613
- if (this.maxOutputTokens > 0 && this.outputTokens >= this.maxOutputTokens)
1614
- this.lastReason = "tokens";
1615
- }
1616
- recordOutput(actual) {
1617
- this.reconcileOutput(0, actual);
1618
- }
1619
- setLimits(maxCalls, maxInputTokens, maxOutputTokens = 0) {
1620
- this.maxCalls = maxCalls;
1621
- this.maxInputTokens = maxInputTokens;
1622
- this.maxOutputTokens = maxOutputTokens;
1623
- }
1624
- callCount() {
1625
- return this.calls;
1626
- }
1627
- remainingCalls() {
1628
- return this.maxCalls > 0 ? Math.max(0, this.maxCalls - this.calls) : Number.POSITIVE_INFINITY;
1629
- }
1630
- inputTokenCount() {
1631
- return this.inputTokens;
1632
- }
1633
- outputTokenCount() {
1634
- return this.outputTokens;
1635
- }
1636
- remainingOutputTokens() {
1637
- return this.maxOutputTokens > 0 ? Math.max(0, this.maxOutputTokens - this.outputTokens - this.reservedOutputTokens) : Number.POSITIVE_INFINITY;
1638
- }
1639
- reason() {
1640
- return this.lastReason;
1641
- }
1642
- }
1643
-
1644
- class ExtractionCacheStats {
1645
- hits = 0;
1646
- misses = 0;
1647
- recordHit() {
1648
- this.hits++;
1649
- }
1650
- recordMiss() {
1651
- this.misses++;
1652
- }
1653
- snapshot() {
1654
- const total = this.hits + this.misses;
1655
- return { hits: this.hits, misses: this.misses, hitRate: total > 0 ? this.hits / total : 0 };
1656
- }
1657
- clear() {
1658
- this.hits = 0;
1659
- this.misses = 0;
1660
- }
1661
- }
1662
- function makeCompactSessionId() {
1663
- return "sc-" + Date.now().toString(36) + "-" + crypto.randomBytes(4).toString("hex");
1664
- }
1665
- function createServices(overrides = {}) {
1666
- return {
1667
- clock: overrides.clock ?? systemClock,
1668
- llm: overrides.llm ?? { complete: (...args) => getLlmClient().complete(...args) },
1669
- toolSupport: overrides.toolSupport ?? new ToolSupportCache,
1670
- metrics: overrides.metrics ?? new MetricsSink,
1671
- extractionCacheStats: overrides.extractionCacheStats ?? new ExtractionCacheStats,
1672
- tokenCalibration: overrides.tokenCalibration ?? new TokenCalibrationStore,
1673
- budget: overrides.budget ?? new BudgetGuard,
1674
- scrubber: overrides.scrubber ?? new SecretScrubber,
1675
- thinkingLevels: overrides.thinkingLevels ?? {
1676
- summaryThinkingLevel: DEFAULT_CONFIG.summaryThinkingLevel,
1677
- segmentationThinkingLevel: DEFAULT_CONFIG.segmentationThinkingLevel
1678
- },
1679
- codexWatchdogMs: overrides.codexWatchdogMs ?? DEFAULT_CONFIG.codexMaxCallMs,
1680
- compactSessionId: overrides.compactSessionId ?? makeCompactSessionId()
1681
- };
1682
- }
1683
- var processToolSupport = new ToolSupportCache;
1684
- var processTokenCalibration = new TokenCalibrationStore;
1685
- var _default = createServices();
1686
-
1687
- // src/domain/telemetry.ts
1688
- function p95(values) {
1689
- if (!values.length)
1690
- return 0;
1691
- const sorted = [...values].sort((a, b) => a - b);
1692
- return sorted[Math.max(0, Math.ceil(sorted.length * 0.95) - 1)] ?? 0;
1693
- }
1694
- function stats(entries, damage) {
1695
- const evidence = entries.filter((entry) => entry.status !== "dry-run");
1696
- const successfulRuns = evidence.filter((entry) => entry.status === "success");
1697
- const quality = successfulRuns.filter((entry) => typeof entry.verificationScore === "number");
1698
- const appliedRunIds = new Set(successfulRuns.filter((entry) => typeof entry.runId === "string" && entry.runId.length >= 8).map((entry) => entry.runId));
1699
- const observedScores = new Map;
1700
- for (const observation of damage) {
1701
- if (!observation.runId || !appliedRunIds.has(observation.runId) || typeof observation.damageScore !== "number" || !Number.isFinite(observation.damageScore))
1702
- continue;
1703
- observedScores.set(observation.runId, Math.max(observedScores.get(observation.runId) ?? 0, Math.max(0, Math.min(100, observation.damageScore))));
1704
- }
1705
- const damaging = [...observedScores.values()].filter((score) => score > 0).length;
1706
- return {
1707
- runs: entries.length,
1708
- appliedRuns: evidence.length,
1709
- successRate: evidence.length ? successfulRuns.length / evidence.length : 1,
1710
- avgQuality: quality.length ? quality.reduce((sum, entry) => sum + (entry.verificationScore ?? 0), 0) / quality.length : null,
1711
- qualityCoverage: evidence.length ? quality.length / evidence.length : 0,
1712
- p95LatencyMs: p95(evidence.map((entry) => entry.durationMs ?? entry.avgLatency).filter(Number.isFinite)),
1713
- avgTokens: evidence.length ? evidence.reduce((sum, entry) => sum + entry.totalInput + entry.totalCacheHit + (entry.totalCacheWrite ?? 0) + entry.totalOutput, 0) / evidence.length : 0,
1714
- fallbackRate: evidence.length ? evidence.filter((entry) => entry.method === "heuristic" || Array.isArray(entry.providerRoutes) && entry.providerRoutes.some((route) => route.successes < route.calls)).length / evidence.length : 0,
1715
- damageRate: observedScores.size ? damaging / observedScores.size : 0,
1716
- damageCoverage: successfulRuns.length ? observedScores.size / successfulRuns.length : 0
1717
- };
1718
- }
1719
- function roundStats(value) {
1720
- return {
1721
- ...value,
1722
- successRate: Math.round(value.successRate * 1000) / 1000,
1723
- avgQuality: value.avgQuality == null ? null : Math.round(value.avgQuality * 10) / 10,
1724
- qualityCoverage: Math.round(value.qualityCoverage * 1000) / 1000,
1725
- p95LatencyMs: Math.round(value.p95LatencyMs),
1726
- avgTokens: Math.round(value.avgTokens),
1727
- fallbackRate: Math.round(value.fallbackRate * 1000) / 1000,
1728
- damageRate: Math.round(value.damageRate * 1000) / 1000,
1729
- damageCoverage: Math.round(value.damageCoverage * 1000) / 1000
1730
- };
1731
- }
1732
- function assessCanary(entries, damageEntries, options) {
1733
- const minCanaryRuns = Math.max(5, options.minCanaryRuns ?? 20);
1734
- const canaryEntries = entries.filter((entry) => entry.metricsSchemaVersion === 2 && entry.version === options.version && entry.releaseChannel === "canary").slice(-Math.max(100, minCanaryRuns));
1735
- const baselineEntries = entries.filter((entry) => entry.metricsSchemaVersion === 2 && (entry.releaseChannel ?? "stable") === "stable").slice(-(options.baselineRuns ?? Math.max(50, minCanaryRuns * 2)));
1736
- const baseline = stats(baselineEntries, damageEntries);
1737
- const canary = stats(canaryEntries, damageEntries);
1738
- const triggers = [];
1739
- const failureBaseline = 1 - baseline.successRate;
1740
- const failureCanary = 1 - canary.successRate;
1741
- if (canary.appliedRuns >= 3 && (failureCanary > 0.050001 || failureCanary - failureBaseline >= 0.050001)) {
1742
- triggers.push({
1743
- metric: "failure-rate",
1744
- baseline: failureBaseline,
1745
- canary: failureCanary,
1746
- threshold: failureCanary > 0.050001 ? ">5% absolute" : "+5pp regression"
1747
- });
1748
- }
1749
- if (canary.avgQuality != null && (canary.avgQuality < 85 || baseline.avgQuality != null && baseline.avgQuality - canary.avgQuality >= 5)) {
1750
- triggers.push({
1751
- metric: "quality",
1752
- baseline: baseline.avgQuality ?? 0,
1753
- canary: canary.avgQuality,
1754
- threshold: canary.avgQuality < 85 ? "<85 absolute" : "-5 points"
1755
- });
1756
- }
1757
- if (baseline.p95LatencyMs >= 1000 && canary.p95LatencyMs >= baseline.p95LatencyMs * 1.5) {
1758
- triggers.push({ metric: "latency", baseline: baseline.p95LatencyMs, canary: canary.p95LatencyMs, threshold: "+50% p95" });
1759
- }
1760
- if (baseline.avgTokens >= 1000 && canary.avgTokens >= baseline.avgTokens * 1.5) {
1761
- triggers.push({ metric: "tokens", baseline: baseline.avgTokens, canary: canary.avgTokens, threshold: "+50%" });
1762
- }
1763
- if (canary.fallbackRate - baseline.fallbackRate >= 0.1) {
1764
- triggers.push({ metric: "fallback", baseline: baseline.fallbackRate, canary: canary.fallbackRate, threshold: "+10pp" });
1765
- }
1766
- if (canary.damageRate - baseline.damageRate >= 0.1) {
1767
- triggers.push({ metric: "damage", baseline: baseline.damageRate, canary: canary.damageRate, threshold: "+10pp" });
1768
- }
1769
- const canarySampleAdequacy = Math.min(1, canary.appliedRuns / minCanaryRuns);
1770
- const baselineSampleAdequacy = Math.min(1, baseline.appliedRuns / Math.max(20, minCanaryRuns));
1771
- const dataConfidence = Math.round(100 * (canarySampleAdequacy * 0.25 + baselineSampleAdequacy * 0.15 + canary.qualityCoverage * canarySampleAdequacy * 0.2 + canary.damageCoverage * canarySampleAdequacy * 0.2 + baseline.damageCoverage * baselineSampleAdequacy * 0.2));
1772
- const reasons = [];
1773
- let decision = "hold";
1774
- if (triggers.length && canary.appliedRuns >= 3) {
1775
- decision = "rollback";
1776
- reasons.push(...triggers.map((trigger) => trigger.metric + " crossed " + trigger.threshold));
1777
- } else if (canary.appliedRuns < minCanaryRuns) {
1778
- reasons.push("need " + (minCanaryRuns - canary.appliedRuns) + " more canary runs with applied outcomes");
1779
- } else if (baseline.appliedRuns < Math.max(20, minCanaryRuns)) {
1780
- reasons.push("stable baseline is too small");
1781
- } else if (canary.qualityCoverage < 0.7) {
1782
- reasons.push("schema-v2 quality coverage is below 70%");
1783
- } else if (canary.damageCoverage < 0.7) {
1784
- reasons.push("correlated canary damage-observation coverage is below 70%");
1785
- } else if (baseline.damageCoverage < 0.7) {
1786
- reasons.push("correlated stable damage-observation coverage is below 70%");
1787
- } else if ((canary.avgQuality ?? 0) < 85) {
1788
- reasons.push("absolute verifier quality is below 85");
1789
- } else if (canary.successRate < 0.949999) {
1790
- reasons.push("absolute success rate is below 95%");
1791
- } else {
1792
- decision = "promote";
1793
- reasons.push("sample, absolute quality, reliability, latency, token, fallback, and damage gates passed");
1794
- }
1795
- return {
1796
- version: options.version,
1797
- decision,
1798
- dataConfidence,
1799
- baseline: roundStats(baseline),
1800
- canary: roundStats(canary),
1801
- triggers,
1802
- reasons
1803
- };
1804
- }
1805
- function safeMetricLabel(value, fallback) {
1806
- if (typeof value !== "string" || !/^[\w./:@+-]{1,160}$/.test(value))
1807
- return fallback;
1808
- return value;
1809
- }
1810
- var TELEMETRY_FAILURE_KINDS = new Set([
1811
- "cancelled",
1812
- "timeout",
1813
- "rate-limit",
1814
- "authentication",
1815
- "budget",
1816
- "output-limit",
1817
- "provider",
1818
- "persistence",
1819
- "validation",
1820
- "verification",
1821
- "yield",
1822
- "internal"
1823
- ]);
1824
- function isTelemetryFailureKind(value) {
1825
- return typeof value === "string" && TELEMETRY_FAILURE_KINDS.has(value);
1826
- }
1827
- function buildPrivacySafeTelemetry(entries, damageEntries, options) {
1828
- const groups = new Map;
1829
- const failures = {};
1830
- for (const entry of entries) {
1831
- const version = safeMetricLabel(entry.version, "legacy");
1832
- const channel = entry.releaseChannel === "canary" ? "canary" : "stable";
1833
- const provider = safeMetricLabel(entry.provider, "unknown");
1834
- const rawModel = safeMetricLabel(entry.model, "unknown");
1835
- const model = rawModel.startsWith(provider + "/") ? rawModel.slice(provider.length + 1) : rawModel;
1836
- const key = [version, channel, provider, model].join("\x00");
1837
- const group = groups.get(key) ?? {
1838
- version,
1839
- channel,
1840
- provider,
1841
- model,
1842
- runs: 0,
1843
- successes: 0,
1844
- quality: 0,
1845
- qualityRuns: 0,
1846
- latency: 0,
1847
- input: 0,
1848
- output: 0
1849
- };
1850
- group.runs++;
1851
- if (entry.status === "success" || entry.status === "dry-run")
1852
- group.successes++;
1853
- if (entry.metricsSchemaVersion === 2 && typeof entry.verificationScore === "number") {
1854
- group.quality += entry.verificationScore;
1855
- group.qualityRuns++;
1856
- }
1857
- group.latency += entry.avgLatency;
1858
- group.input += entry.totalInput + entry.totalCacheHit + (entry.totalCacheWrite ?? 0);
1859
- group.output += entry.totalOutput;
1860
- groups.set(key, group);
1861
- if (isTelemetryFailureKind(entry.failureKind)) {
1862
- failures[entry.failureKind] = (failures[entry.failureKind] ?? 0) + 1;
1863
- }
1864
- }
1865
- const aggregates = [...groups.values()].map((group) => ({
1866
- version: group.version,
1867
- channel: group.channel,
1868
- provider: group.provider,
1869
- model: group.model,
1870
- runs: group.runs,
1871
- successes: group.successes,
1872
- avgQuality: group.qualityRuns ? Math.round(group.quality / group.qualityRuns * 10) / 10 : null,
1873
- avgLatencyMs: group.runs ? Math.round(group.latency / group.runs) : 0,
1874
- inputTokens: group.input,
1875
- outputTokens: group.output
1876
- })).sort((a, b) => b.runs - a.runs || a.version.localeCompare(b.version));
1877
- return {
1878
- generatedAt: new Date().toISOString(),
1879
- totalRuns: entries.length,
1880
- aggregates,
1881
- failures,
1882
- canary: assessCanary(entries, damageEntries, options),
1883
- privacy: "aggregate-only; no session ids, project ids, prompts, summaries, paths, or error text"
1884
- };
1885
- }
1886
- function formatPrivacySafeTelemetry(report) {
1887
- const lines = [
1888
- "# Smart Compact Telemetry",
1889
- "",
1890
- "Privacy: " + report.privacy + ".",
1891
- "",
1892
- "| Version | Channel | Provider/model | Runs | Success | Quality | Latency | Input | Output |",
1893
- "|---|---|---|---:|---:|---:|---:|---:|---:|"
1894
- ];
1895
- for (const item of report.aggregates) {
1896
- lines.push("| " + item.version + " | " + item.channel + " | " + item.provider + "/" + item.model + " | " + item.runs + " | " + item.successes + "/" + item.runs + " | " + (item.avgQuality == null ? "n/a" : item.avgQuality.toFixed(1)) + " | " + item.avgLatencyMs + "ms | " + item.inputTokens + " | " + item.outputTokens + " |");
1897
- }
1898
- lines.push("", "## Canary: " + report.canary.decision.toUpperCase() + " (data confidence " + report.canary.dataConfidence + "%)", "");
1899
- const baseline = report.canary.baseline;
1900
- const canary = report.canary.canary;
1901
- lines.push("| Gate | Stable baseline | Canary |", "|---|---:|---:|", "| Runs (total/applied) | " + baseline.runs + "/" + baseline.appliedRuns + " | " + canary.runs + "/" + canary.appliedRuns + " |", "| Success | " + Math.round(baseline.successRate * 100) + "% | " + Math.round(canary.successRate * 100) + "% |", "| Verify quality | " + (baseline.avgQuality ?? "n/a") + " | " + (canary.avgQuality ?? "n/a") + " |", "| p95 duration | " + baseline.p95LatencyMs + "ms | " + canary.p95LatencyMs + "ms |", "| Avg tokens | " + baseline.avgTokens + " | " + canary.avgTokens + " |", "| Fallback | " + Math.round(baseline.fallbackRate * 100) + "% | " + Math.round(canary.fallbackRate * 100) + "% |", "| Damage | " + Math.round(baseline.damageRate * 100) + "% | " + Math.round(canary.damageRate * 100) + "% |", "| Damage observed | " + Math.round(baseline.damageCoverage * 100) + "% | " + Math.round(canary.damageCoverage * 100) + "% |", "");
1902
- for (const reason of report.canary.reasons)
1903
- lines.push("- " + reason);
1904
- const failureText = Object.entries(report.failures).map(([kind, count]) => kind + "=" + count).join(", ");
1905
- lines.push("", "Failures: " + (failureText || "none classified"));
1906
- return lines.join(`
1907
- `);
1908
- }
1909
-
1910
- // src/utils/cache.ts
1911
- var INTERNAL_PHASES = new Set([
1912
- "explore-retry",
1913
- "explore-direct",
1914
- "single-pass",
1915
- "batch",
1916
- "assemble",
1917
- "patch"
1918
- ]);
1919
- var SEGMENTATION_PHASES = new Set([
1920
- "probe",
1921
- "explore",
1922
- "explore-loop",
1923
- "explore-retry",
1924
- "explore-direct"
1925
- ]);
1926
- function readMetricsLog(limit = 100) {
1927
- try {
1928
- const logPath = metricsLogFile();
1929
- if (!fs2.existsSync(logPath))
1930
- return [];
1931
- const stat = fs2.statSync(logPath);
1932
- const TAIL_CHUNK = 64 * 1024;
1933
- const wantBytes = Math.min(stat.size, Math.max(TAIL_CHUNK, limit * 8 * 512));
1934
- const startPos = Math.max(0, stat.size - wantBytes);
1935
- const fd = fs2.openSync(logPath, "r");
1936
- try {
1937
- const buf = Buffer.alloc(wantBytes);
1938
- const bytesRead = fs2.readSync(fd, buf, 0, wantBytes, startPos);
1939
- let text = buf.subarray(0, bytesRead).toString("utf8");
1940
- if (startPos > 0) {
1941
- const nl = text.indexOf(`
1942
- `);
1943
- if (nl >= 0)
1944
- text = text.slice(nl + 1);
1945
- }
1946
- const lines = text.split(`
1947
- `).filter(Boolean);
1948
- const entries = [];
1949
- for (const line of lines) {
1950
- try {
1951
- entries.push(JSON.parse(line));
1952
- } catch {
1953
- warn("Skipping corrupt compact metrics line");
1954
- }
1955
- }
1956
- return entries.slice(-limit);
1957
- } finally {
1958
- fs2.closeSync(fd);
1959
- }
1960
- } catch (e) {
1961
- warn("readMetricsLog failed", e);
1962
- return [];
1963
- }
1964
- }
1965
-
1966
- // src/phases/verify.ts
1967
- var HIGH_RISK_OUTCOME_RE = /(?:\ball\s+tests?\s+(?:pass|passed|passing)\b|\btests?\s+(?:pass|passed|passing)\b|\b(?:build|deployment|migration)\s+(?:completed|succeeded|passed|successful)\b|\b(?:deployed|published|released)\b|\b(?:bug|issue|error)\s+(?:fixed|resolved)\b|\bno\s+(?:errors?|failures?)\b|\bcompleted successfully\b|\btestler?\s+(?:ge\u00E7ti|ba\u015Far\u0131l\u0131)\b|\bba\u015Far\u0131yla\s+(?:tamamland\u0131|da\u011F\u0131t\u0131ld\u0131|yay\u0131nland\u0131)\b|\b(?:deploy edildi|yay\u0131nland\u0131|hata yok)\b)/iu;
1968
- var NEGATED_OUTCOME_RE = /\b(?:not|never|pending|failed|failing|unresolved|hen\u00FCz|de\u011Fil|ba\u015Far\u0131s\u0131z)\b/iu;
1969
- var NONE_BLOCKER_VALUE_RE = /^(?:none|no blockers?|yok)\s*(?:recorded|known)?[.!]?$/i;
1970
- var BULLET_NONE_BLOCKER_RE = /^(?:[-*+]|\d+[.)])\s+(?:none|no blockers?|yok)\s*(?:recorded|known)?[.!]?$/i;
1971
- var PATH_PLACEHOLDER_RE = /^(?:none|none recorded|no blockers?|yok)[.!]?$/i;
1972
- function noneBlockerLineIndexes(lines) {
1973
- const indexes = new Set;
1974
- const nonEmpty = lines.map((line, index) => ({ index, text: line.trim() })).filter((item) => item.text);
1975
- for (const item of nonEmpty) {
1976
- if (BULLET_NONE_BLOCKER_RE.test(item.text))
1977
- indexes.add(item.index);
1978
- }
1979
- if (nonEmpty.length === 1 && NONE_BLOCKER_VALUE_RE.test(nonEmpty[0].text)) {
1980
- indexes.add(nonEmpty[0].index);
1981
- }
1982
- return indexes;
1983
- }
1984
- function collectListedPaths(body, expectedPaths) {
1985
- const values = new Set;
1986
- const encodedValues = new Set;
1987
- for (const line of body.split(`
1988
- `)) {
1989
- const raw = line.replace(/^\s*(?:[-*+]|\d+[.)])\s+/, "").trim();
1990
- if (!raw)
1991
- continue;
1992
- if (raw.startsWith('"')) {
1993
- try {
1994
- const decoded = JSON.parse(raw);
1995
- if (typeof decoded === "string") {
1996
- values.add(decoded);
1997
- encodedValues.add(decoded);
1998
- continue;
1999
- }
2000
- } catch {}
2001
- }
2002
- values.add(raw);
2003
- if (expectedPaths.has(raw))
2004
- continue;
2005
- const unwrapped = raw.startsWith("`") && raw.endsWith("`") ? raw.slice(1, -1) : raw;
2006
- const unchecked = unwrapped.replace(/^\[[ x]\]\s+/i, "");
2007
- if (expectedPaths.has(unchecked))
2008
- values.add(unchecked);
2009
- }
2010
- return {
2011
- values,
2012
- encodedValues,
2013
- normalizedValues: new Set(Array.from(values, normalizePath))
2014
- };
2015
- }
2016
- function decodePathDisplay(display) {
2017
- try {
2018
- const decoded = JSON.parse(display);
2019
- return typeof decoded === "string" ? decoded : display;
2020
- } catch {
2021
- return display;
2022
- }
2023
- }
2024
- function hasListedPath(listed, file, display, normalizedOwners) {
2025
- const decodedDisplay = decodePathDisplay(display);
2026
- if (listed.encodedValues.has(decodedDisplay))
2027
- return true;
2028
- if (PATH_PLACEHOLDER_RE.test(file))
2029
- return false;
2030
- if (listed.values.has(file))
2031
- return true;
2032
- for (const candidate of [file, decodedDisplay]) {
2033
- const normalized = normalizePath(candidate);
2034
- if (normalizedOwners.get(normalized) === 1 && listed.normalizedValues.has(normalized))
2035
- return true;
2036
- }
2037
- return false;
2038
- }
2039
- function outcomeClaims(summary, pathEvidence) {
2040
- const pathLines = new Set(Array.from(pathEvidence, ([path, display]) => [path, display, "`" + path + "`"]).flat());
2041
- return Array.from(new Set(summary.split(/\r?\n/).map((line) => line.replace(/^\s*(?:[-*+]|\d+[.)])\s+/, "").replace(/^\[[ x]\]\s+/i, "").trim()).filter((line) => line.length > 0 && !line.startsWith("#") && !pathLines.has(line)).filter((line) => HIGH_RISK_OUTCOME_RE.test(line)).filter((line) => /\bno\s+(?:errors?|failures?)\b/i.test(line) || !NEGATED_OUTCOME_RE.test(line)))).slice(0, 12);
2042
- }
2043
- function classifyOutcomeClaim(claim) {
2044
- const lower = claim.toLowerCase();
2045
- if (/\btests?\b|\btestler?\b/.test(lower))
2046
- return "test";
2047
- if (/\bbuild\b|\bcompil(?:e|ed|ation)\b|\btypecheck\b/.test(lower))
2048
- return "build";
2049
- if (/\bdeploy(?:ed|ment)?\b|\bpublish(?:ed)?\b|\breleas(?:e|ed)\b/.test(lower))
2050
- return "release";
2051
- if (/\bbug\b|\bissue\b|\berror\b|\bfail(?:ed|ure)?\b|\bhata\b/.test(lower))
2052
- return "error";
2053
- if (/\bfile\b|\bdosya\b/.test(lower))
2054
- return "file";
2055
- return "generic";
2056
- }
2057
- var successfulToolEvidenceCache = new WeakMap;
2058
- var sourceTextCache = new WeakMap;
2059
- function sourceSupportsFileReference(ref, messages) {
2060
- let texts = sourceTextCache.get(messages);
2061
- if (!texts) {
2062
- texts = messages.map((message) => extractText(message.content).replace(/\\/g, "/").toLowerCase());
2063
- sourceTextCache.set(messages, texts);
2064
- }
2065
- const needle = ref.replace(/\\/g, "/").replace(/^\.\//, "").toLowerCase();
2066
- if (!needle)
2067
- return false;
2068
- for (const text of texts) {
2069
- let index = text.indexOf(needle);
2070
- while (index >= 0) {
2071
- const before = text[index - 1] ?? "";
2072
- const after = text[index + needle.length] ?? "";
2073
- if ((!before || !/[\w.-]/.test(before)) && (!after || !/[\w.-]/.test(after)))
2074
- return true;
2075
- index = text.indexOf(needle, index + 1);
2076
- }
2077
- }
2078
- return false;
2079
- }
2080
- function successfulToolEvidence(messages) {
2081
- const cached = successfulToolEvidenceCache.get(messages);
2082
- if (cached)
2083
- return cached;
2084
- const toolCalls = buildToolCallIndex(messages);
2085
- const evidence = [];
2086
- for (const message of messages) {
2087
- if (message.role !== "toolResult" || message.isError)
2088
- continue;
2089
- const call = toolCalls.get(message.toolCallId ?? "");
2090
- if (!call)
2091
- continue;
2092
- const result = extractText(message.content).slice(0, 8000);
2093
- if (!result.trim() || LIKELY_ERROR_RE.test(result))
2094
- continue;
2095
- const command = [call.arguments.command, call.arguments.cmd, call.arguments.script].find((value) => typeof value === "string") ?? "";
2096
- evidence.push({
2097
- name: normalizeToolName(call.name),
2098
- operation: classifyToolOperation(call.arguments, call.name),
2099
- command,
2100
- path: extractToolPath(call.arguments),
2101
- result
2102
- });
2103
- }
2104
- successfulToolEvidenceCache.set(messages, evidence);
2105
- return evidence;
2106
- }
2107
- function successfulToolSupportsClaim(claim, tools, extraction) {
2108
- const shape = semanticShape(claim);
2109
- const category = classifyOutcomeClaim(claim);
2110
- if (category === "error" && extraction.errors.some((error) => error.resolved && hasSemanticEvidence(claim, error.message)))
2111
- return true;
2112
- if (category === "file" && extraction.modifiedFiles.some((file) => claim.toLowerCase().includes(file.path.toLowerCase())))
2113
- return true;
2114
- for (const tool of tools) {
2115
- const operationText = tool.name + " " + tool.command;
2116
- const operationSupports = category === "test" ? /\b(?:test|tests|pytest|jest|vitest|mocha|rspec)\b/i.test(operationText) : category === "build" ? /\b(?:build|compile|typecheck|tsc|check)\b/i.test(operationText) : category === "release" ? /\b(?:deploy|publish|release)\b/i.test(operationText) : category === "file" ? tool.operation === "mutate" || tool.operation === "delete" : category === "error" ? tool.operation === "execute" || tool.operation === "mutate" || tool.operation === "delete" : tool.operation !== "read" && tool.operation !== "search" && tool.operation !== "list";
2117
- if (!operationSupports)
2118
- continue;
2119
- if (hasSemanticEvidence(claim, tool.result))
2120
- return true;
2121
- const lower = tool.result.toLowerCase();
2122
- if (category === "test" && /\b\d+\s+(?:tests?\s+)?pass(?:ed)?\b/.test(lower) && !/\b(?:fail(?:ed|ures?)?|errors?)\s*[:=]?\s*[1-9]\d*\b/.test(lower))
2123
- return true;
2124
- if (category === "build" && /\b(?:succeeded|successful|passed|exit(?:ed)?\s+(?:code\s+)?0)\b/.test(lower))
2125
- return true;
2126
- if (category === "release" && /\b(?:succeeded|successful|completed|published|deployed|released)\b/.test(lower))
2127
- return true;
2128
- if (category === "error" && shape.concepts.length > 0 && /\b(?:fixed|resolved|passed|succeeded|successful)\b/.test(lower) && hasSemanticEvidence(claim, tool.result))
2129
- return true;
2130
- }
2131
- return false;
2132
- }
2133
- var NEGATION_MARKERS = new Set([
2134
- "no",
2135
- "not",
2136
- "never",
2137
- "without",
2138
- "avoid",
2139
- "forbidden",
2140
- "prohibit",
2141
- "de\u011Fil",
2142
- "asla",
2143
- "olmadan",
2144
- "yasak",
2145
- "hay\u0131r"
2146
- ]);
2147
- var CONDITION_MARKERS = new Set([
2148
- "only",
2149
- "after",
2150
- "before",
2151
- "with",
2152
- "requir",
2153
- "until",
2154
- "sadece",
2155
- "sonra",
2156
- "\xF6nce",
2157
- "gerekli",
2158
- "gerektirir"
2159
- ]);
2160
- var POLARITY_INVERTING_GUARDS = new Set([
2161
- "skip",
2162
- "skipp",
2163
- "forget",
2164
- "forgett",
2165
- "omit",
2166
- "omitt",
2167
- "neglect",
2168
- "fail",
2169
- "avoid"
2170
- ]);
2171
- var SEMANTIC_STOP = new Set([
2172
- "the",
2173
- "and",
2174
- "that",
2175
- "this",
2176
- "with",
2177
- "from",
2178
- "into",
2179
- "must",
2180
- "should",
2181
- "only",
2182
- "after",
2183
- "before",
2184
- "without",
2185
- "never",
2186
- "not",
2187
- "does",
2188
- "have",
2189
- "i\xE7in",
2190
- "ile",
2191
- "sonra",
2192
- "\xF6nce",
2193
- "sadece",
2194
- "asla",
2195
- "de\u011Fil",
2196
- "olmadan"
2197
- ]);
2198
- var TR_SUFFIXES = [
2199
- "lar\u0131",
2200
- "leri",
2201
- "\u0131n\u0131n",
2202
- "inin",
2203
- "unun",
2204
- "\xFCn\xFCn",
2205
- "\u0131nda",
2206
- "inde",
2207
- "unda",
2208
- "\xFCnde",
2209
- "m\u0131\u015F",
2210
- "mi\u015F",
2211
- "mu\u015F",
2212
- "m\xFC\u015F",
2213
- "lar",
2214
- "ler",
2215
- "\u0131n\u0131",
2216
- "ini",
2217
- "unu",
2218
- "\xFCn\xFC",
2219
- "\u0131na",
2220
- "ine",
2221
- "una",
2222
- "\xFCne",
2223
- "dan",
2224
- "den",
2225
- "tan",
2226
- "ten",
2227
- "d\u0131r",
2228
- "dir",
2229
- "dur",
2230
- "d\xFCr",
2231
- "t\u0131r",
2232
- "tir",
2233
- "tur",
2234
- "t\xFCr",
2235
- "yor",
2236
- "mak",
2237
- "mek",
2238
- "da",
2239
- "de",
2240
- "ta",
2241
- "te",
2242
- "d\u0131",
2243
- "di",
2244
- "du",
2245
- "d\xFC",
2246
- "t\u0131",
2247
- "ti",
2248
- "tu",
2249
- "t\xFC",
2250
- "\u0131n",
2251
- "in",
2252
- "un",
2253
- "\xFCn",
2254
- "sa",
2255
- "se"
2256
- ];
2257
- function stemToken(token) {
2258
- const lower = token.toLocaleLowerCase();
2259
- for (const suffix of TR_SUFFIXES) {
2260
- if (lower.length >= 4 + suffix.length && lower.endsWith(suffix)) {
2261
- return lower.slice(0, -suffix.length);
2262
- }
2263
- }
2264
- if (lower.length > 6 && lower.endsWith("ing"))
2265
- return lower.slice(0, -3);
2266
- if (lower.length > 5 && lower.endsWith("ed"))
2267
- return lower.slice(0, -2);
2268
- if (lower.length > 5 && lower.endsWith("es"))
2269
- return lower.slice(0, -2);
2270
- if (lower.length > 4 && lower.endsWith("s"))
2271
- return lower.slice(0, -1);
2272
- return lower;
2273
- }
2274
- function semanticTokens(text) {
2275
- return (text.normalize("NFKC").match(/[\p{L}\p{N}_-]+/gu) ?? []).map(stemToken).filter((token) => token.length > 2 || NEGATION_MARKERS.has(token));
2276
- }
2277
- var semanticShapeCache = new Map;
2278
- var semanticFragmentCache = new Map;
2279
- function semanticFragments(text) {
2280
- const cached = lruGet(semanticFragmentCache, text);
2281
- if (cached)
2282
- return cached;
2283
- const fragments = Array.from(new Set(text.split(/\r?\n/).flatMap((line) => [line, ...line.split(/[.;]/)]).map((part) => part.replace(/^\s*[-*\d.)]+\s*/, "").trim()).filter(Boolean))).map(semanticTokens);
2284
- lruSet(semanticFragmentCache, text, fragments, 256);
2285
- return fragments;
2286
- }
2287
- function hasNearbyMarker(tokens, anchor, markers) {
2288
- return tokens.some((token, index) => token === anchor && tokens.slice(Math.max(0, index - 2), index + 3).some((near) => markers.has(near)));
2289
- }
2290
- function hasEffectiveTargetNegation(tokens, anchor) {
2291
- return tokens.some((token, anchorIndex) => {
2292
- if (token !== anchor)
2293
- return false;
2294
- const nearbyStart = Math.max(0, anchorIndex - 2);
2295
- const nearbyNegations = tokens.slice(nearbyStart, anchorIndex + 3).map((near, offset) => NEGATION_MARKERS.has(near) && !(near === "without" && nearbyStart + offset > anchorIndex) ? nearbyStart + offset : -1).filter((index) => index >= 0);
2296
- const governingStart = Math.max(0, anchorIndex - 3);
2297
- const preceding = tokens.slice(governingStart, anchorIndex);
2298
- const nearbyGuards = preceding.map((near, offset) => POLARITY_INVERTING_GUARDS.has(near) ? governingStart + offset : -1).filter((index) => index >= 0);
2299
- const governingIndex = preceding.findIndex((near, offset) => NEGATION_MARKERS.has(near) && POLARITY_INVERTING_GUARDS.has(preceding[offset + 1] ?? ""));
2300
- if (governingIndex < 0)
2301
- return nearbyNegations.length > 0 || nearbyGuards.length > 0;
2302
- const absoluteGoverningIndex = governingStart + governingIndex;
2303
- const guardIndex = absoluteGoverningIndex + 1;
2304
- const nested = tokens.slice(guardIndex + 1, anchorIndex).some((inner) => NEGATION_MARKERS.has(inner) || POLARITY_INVERTING_GUARDS.has(inner));
2305
- return nested || nearbyNegations.some((index) => index !== absoluteGoverningIndex && index !== guardIndex) || nearbyGuards.some((index) => index !== guardIndex);
2306
- });
2307
- }
2308
- function semanticShape(source) {
2309
- const cached = lruGet(semanticShapeCache, source);
2310
- if (cached)
2311
- return cached;
2312
- const sourceTokens = semanticTokens(source);
2313
- const concepts = Array.from(new Set(sourceTokens.filter((token) => !/^\d+$/.test(token) && !SEMANTIC_STOP.has(token) && !NEGATION_MARKERS.has(token) && !CONDITION_MARKERS.has(token))));
2314
- const negative = sourceTokens.some((token) => NEGATION_MARKERS.has(token));
2315
- const conditional = sourceTokens.some((token) => CONDITION_MARKERS.has(token));
2316
- const anchor = concepts.find((concept) => hasNearbyMarker(sourceTokens, concept, NEGATION_MARKERS)) ?? concepts[0] ?? "";
2317
- const shape = { sourceTokens, concepts, anchor, negative, conditional };
2318
- lruSet(semanticShapeCache, source, shape, 512);
2319
- return shape;
2320
- }
2321
- function hasSemanticEvidence(source, target) {
2322
- const { sourceTokens, concepts, anchor, negative, conditional } = semanticShape(source);
2323
- if (!concepts.length)
2324
- return true;
2325
- const required = Math.min(concepts.length, Math.max(1, Math.ceil(concepts.length * 0.6)));
2326
- return semanticFragments(target).some((tokens) => {
2327
- const overlap = concepts.filter((concept) => tokens.includes(concept)).length;
2328
- if (overlap < required)
2329
- return false;
2330
- const targetNegative = hasEffectiveTargetNegation(tokens, anchor);
2331
- if (negative && !targetNegative) {
2332
- const conditionalRestatement = sourceTokens.includes("without") && tokens.some((token) => CONDITION_MARKERS.has(token)) && overlap >= Math.min(2, concepts.length);
2333
- if (!conditionalRestatement)
2334
- return false;
2335
- }
2336
- if (!negative && targetNegative)
2337
- return false;
2338
- if (conditional && !negative && !tokens.some((token) => CONDITION_MARKERS.has(token)))
2339
- return false;
2340
- return true;
2341
- });
2342
- }
2343
- function hasSemanticContradiction(source, target) {
2344
- const sourceFragments = new Set(semanticFragments(source).map((tokens) => tokens.join(" ")));
2345
- const { sourceTokens, concepts, anchor, negative, conditional } = semanticShape(source);
2346
- if (!anchor)
2347
- return false;
2348
- const required = Math.min(concepts.length, Math.max(1, Math.ceil(concepts.length * 0.6)));
2349
- return semanticFragments(target).some((tokens) => {
2350
- if (!tokens.includes(anchor) || sourceFragments.has(tokens.join(" ")))
2351
- return false;
2352
- const overlap = concepts.filter((concept) => tokens.includes(concept)).length;
2353
- if (overlap < required)
2354
- return false;
2355
- const targetNegative = hasEffectiveTargetNegation(tokens, anchor);
2356
- if (negative && !targetNegative) {
2357
- const validConditional = sourceTokens.includes("without") && tokens.some((token) => CONDITION_MARKERS.has(token)) && overlap >= Math.min(2, concepts.length);
2358
- return !validConditional;
2359
- }
2360
- if (!negative && targetNegative)
2361
- return true;
2362
- return conditional && !negative && !tokens.some((token) => CONDITION_MARKERS.has(token));
2363
- });
2364
- }
2365
- function addGap(accumulator, gap, penalty) {
2366
- accumulator.gaps.push(gap);
2367
- accumulator.score -= penalty;
2368
- }
2369
- function uniqueByText(items, text) {
2370
- const seen = new Set;
2371
- return items.filter((item) => {
2372
- const key = text(item).toLowerCase().replace(/\s+/g, " ").trim();
2373
- if (!key || seen.has(key))
2374
- return false;
2375
- seen.add(key);
2376
- return true;
2377
- });
2378
- }
2379
- function collectVerificationEvidence(extraction, continuity, evidence) {
2380
- const unresolved = uniqueByText([
2381
- ...extraction.errors.flatMap((error) => !error.resolved ? [{ message: error.message }] : []),
2382
- ...(continuity?.unresolvedErrors ?? []).map((error) => ({
2383
- message: error.message
2384
- }))
2385
- ], (item) => item.message);
2386
- const resolved = uniqueByText([
2387
- ...extraction.errors.flatMap((error) => error.resolved ? [{ message: error.message }] : []),
2388
- ...(continuity?.resolvedErrors ?? []).map((error) => ({
2389
- message: error.message
2390
- }))
2391
- ], (item) => item.message).slice(-5);
2392
- const steeringConstraints = [];
2393
- if (evidence.steering?.focus?.trim()) {
2394
- steeringConstraints.push({
2395
- text: "Preserve detail about: " + evidence.steering.focus
2396
- });
2397
- }
2398
- if (evidence.steering?.note?.trim()) {
2399
- steeringConstraints.push({ text: evidence.steering.note });
2400
- }
2401
- const retired = new Set([...continuity?.factOverrides ?? [], ...evidence.factOverrides ?? []].filter((item) => item.kind === "constraint" && item.status !== "active").map((item) => item.summaryKey));
2402
- const constraints = uniqueByText([
2403
- ...extraction.constraints.flatMap((item) => item.confidence >= 0.8 ? [{ text: item.text }] : []),
2404
- ...(continuity?.constraints ?? []).flatMap((item) => item.confidence >= 0.8 ? [{ text: item.text }] : []),
2405
- ...steeringConstraints
2406
- ], (item) => item.text).filter((item) => !isNonLiveConstraintText(item.text) && !retired.has(normalizeFactKey(item.text)));
2407
- const decisions = uniqueByText([
2408
- ...extraction.decisions.flatMap((item) => item.type === "explicit" ? [{ summary: item.summary }] : []),
2409
- ...(continuity?.decisions ?? []).flatMap((item) => item.type === "explicit" ? [{ summary: item.summary }] : [])
2410
- ], (item) => item.summary);
2411
- return {
2412
- unresolved,
2413
- resolved,
2414
- constraints,
2415
- decisions,
2416
- goal: extraction.mainGoal ?? continuity?.goal ?? null
2417
- };
2418
- }
2419
- function verifyRequiredSections(parsed, accumulator) {
2420
- const required = [
2421
- { kind: "goal", penalty: 5 },
2422
- { kind: "progress", penalty: 5 },
2423
- { kind: "critical-context", penalty: 3 }
2424
- ];
2425
- for (const item of required) {
2426
- if (!findSection(parsed, item.kind)) {
2427
- addGap(accumulator, { kind: "missing-section", section: item.kind }, item.penalty);
2428
- }
2429
- }
2430
- }
2431
- function verifyPathCoverage(parsed, extraction, continuity, evidence, accumulator) {
2432
- const modified = extraction.modifiedFiles.map((file) => file.path);
2433
- const read = extraction.readFiles;
2434
- const deleted = Array.from(new Set([...extraction.deletedFiles, ...continuity?.deletedFiles ?? []]));
2435
- const required = [...modified, ...read, ...deleted];
2436
- const expected = new Set(required);
2437
- const rendered = buildSummaryPathEvidence(required, evidence.summaryBudgetTokens);
2438
- const ownerSets = new Map;
2439
- for (const file of required) {
2440
- const display = rendered.get(file);
2441
- for (const candidate of [
2442
- file,
2443
- ...display ? [decodePathDisplay(display)] : []
2444
- ]) {
2445
- const normalized = normalizePath(candidate);
2446
- const owners = ownerSets.get(normalized) ?? new Set;
2447
- owners.add(file);
2448
- ownerSets.set(normalized, owners);
2449
- }
2450
- }
2451
- const ownerCounts = new Map(Array.from(ownerSets, ([file, owners]) => [file, owners.size]));
2452
- const listed = (kind) => collectListedPaths(findSection(parsed, kind)?.body ?? "", expected);
2453
- const modifiedListed = listed("files-modified");
2454
- const readListed = listed("files-read");
2455
- const deletedListed = listed("files-deleted");
2456
- for (const file of modified) {
2457
- const display = rendered.get(file);
2458
- if (display && !hasListedPath(modifiedListed, file, display, ownerCounts)) {
2459
- addGap(accumulator, { kind: "missing-file", path: file }, 5);
2460
- }
2461
- }
2462
- for (const file of read) {
2463
- const display = rendered.get(file);
2464
- if (display && !hasListedPath(readListed, file, display, ownerCounts)) {
2465
- addGap(accumulator, { kind: "missing-read-file", path: file }, 5);
2466
- }
2467
- }
2468
- for (const file of deleted) {
2469
- const display = rendered.get(file);
2470
- if (display && !hasListedPath(deletedListed, file, display, ownerCounts)) {
2471
- addGap(accumulator, { kind: "missing-deleted-file", path: file }, 5);
2472
- }
2473
- }
2474
- return { modified, read, deleted, rendered };
2475
- }
2476
- function verifyErrorEvidence(normalizedSummary, collected, accumulator) {
2477
- for (const error of collected.unresolved) {
2478
- const snippet = summaryEvidenceLine(error.message, TRUNC.ERROR_SNIPPET).toLowerCase().replace(/\\/g, "/");
2479
- if (snippet.length > 5 && !normalizedSummary.includes(snippet)) {
2480
- addGap(accumulator, { kind: "missing-error", message: error.message }, 5);
2481
- }
2482
- }
2483
- for (const error of collected.resolved) {
2484
- const snippet = summaryEvidenceLine(error.message, TRUNC.ERROR_SNIPPET).toLowerCase().replace(/\\/g, "/");
2485
- if (snippet.length > 5 && !normalizedSummary.includes(snippet)) {
2486
- addGap(accumulator, { kind: "missing-error", message: error.message, resolved: true }, 2);
2487
- }
2488
- }
2489
- }
2490
- function verifySemanticCoverage(parsed, collected, accumulator) {
2491
- const constraintTarget = [
2492
- findSection(parsed, "constraints")?.body ?? "",
2493
- findSection(parsed, "critical-context")?.body ?? ""
2494
- ].join(`
2495
- `);
2496
- for (const constraint of collected.constraints) {
2497
- if (!hasSemanticEvidence(constraint.text, constraintTarget)) {
2498
- addGap(accumulator, { kind: "missing-constraint", text: constraint.text }, 8);
2499
- }
2500
- if (hasSemanticContradiction(constraint.text, constraintTarget)) {
2501
- addGap(accumulator, {
2502
- kind: "inconsistency",
2503
- detail: "semantic-contradiction: constraint contradicts " + constraint.text.slice(0, TRUNC.SNIPPET)
2504
- }, 20);
2505
- }
2506
- }
2507
- if (collected.goal) {
2508
- const goalTarget = findSection(parsed, "goal")?.body ?? "";
2509
- if (!hasSemanticEvidence(collected.goal, goalTarget)) {
2510
- addGap(accumulator, { kind: "missing-goal", goal: collected.goal }, 12);
2511
- }
2512
- if (hasSemanticContradiction(collected.goal, goalTarget)) {
2513
- addGap(accumulator, {
2514
- kind: "inconsistency",
2515
- detail: "semantic-contradiction: goal polarity or condition changed"
2516
- }, 20);
2517
- }
2518
- }
2519
- const decisionBody = findSection(parsed, "decisions")?.body ?? "";
2520
- for (const decision of collected.decisions) {
2521
- if (!hasSemanticEvidence(decision.summary, decisionBody)) {
2522
- addGap(accumulator, { kind: "missing-decision", summary: decision.summary }, 8);
2523
- }
2524
- if (hasSemanticContradiction(decision.summary, decisionBody)) {
2525
- addGap(accumulator, {
2526
- kind: "inconsistency",
2527
- detail: "semantic-contradiction: decision contradicts " + decision.summary.slice(0, TRUNC.SNIPPET)
2528
- }, 20);
2529
- }
2530
- }
2531
- }
2532
- function verifyFileReferences(summary, extraction, continuity, evidence, collected, paths, accumulator) {
2533
- const groundedEvidence = [
2534
- ...collected.unresolved.map((item) => item.message),
2535
- ...collected.resolved.map((item) => item.message),
2536
- ...collected.constraints.map((item) => item.text),
2537
- ...collected.decisions.map((item) => item.summary),
2538
- ...collected.goal ? [collected.goal] : [],
2539
- ...extraction.lastUserMessages,
2540
- ...extraction.timeline.map((item) => item.summary),
2541
- ...extraction.topics.map((item) => item.primaryFile ?? ""),
2542
- ...continuity?.openLoops.map((item) => item.summary) ?? [],
2543
- ...continuity?.criticalContext ?? []
2544
- ];
2545
- const groundedFiles = groundedEvidence.flatMap((value) => [
2546
- value,
2547
- summaryEvidenceLine(value, TRUNC.ERROR_SNIPPET),
2548
- summaryEvidenceLine(value, TRUNC.TOPIC_LABEL),
2549
- summaryEvidenceLine(value, TRUNC.PREVIEW),
2550
- summaryEvidenceLine(value, TRUNC.MESSAGE)
2551
- ]).flatMap(extractFileRefs);
2552
- const renderedPaths = Array.from(paths.rendered.values()).flatMap((line) => [
2553
- decodePathDisplay(line),
2554
- line.startsWith('"') && line.endsWith('"') ? line.slice(1, -1) : line
2555
- ]);
2556
- const knownFiles = Array.from(new Set([
2557
- ...paths.modified,
2558
- ...paths.read,
2559
- ...paths.deleted,
2560
- ...extraction.referencedFiles ?? [],
2561
- ...groundedFiles,
2562
- ...renderedPaths,
2563
- ...renderedPaths.flatMap(extractFileRefs),
2564
- ...continuity?.modifiedFiles ?? [],
2565
- ...continuity?.readFiles ?? [],
2566
- ...(continuity?.unresolvedErrors ?? []).flatMap((error) => error.files),
2567
- ...(continuity?.openLoops ?? []).flatMap((loop) => loop.files)
2568
- ]));
2569
- const knownFileIndex = buildKnownPathReferenceIndex(knownFiles);
2570
- for (const ref of new Set(extractFileRefs(summary))) {
2571
- const grounded = isKnownPathReferenceInIndex(ref, knownFileIndex) || Boolean(evidence.sourceMessages && sourceSupportsFileReference(ref, evidence.sourceMessages));
2572
- if (!grounded)
2573
- addGap(accumulator, { kind: "fabricated-file", ref }, 4);
2574
- }
2575
- }
2576
- function verifyProgressConsistency(parsed, extraction, collected, paths, accumulator) {
2577
- const progress = findSection(parsed, "progress");
2578
- if (!progress)
2579
- return;
2580
- const done = progress.body.match(/###\s*Done[\s\S]*?(?=###|$)/i)?.[0] ?? "";
2581
- const blocked = progress.body.match(/###\s*Blocked[\s\S]*?(?=###|$)/i)?.[0] ?? "";
2582
- if (collected.unresolved.length > 0 && noneBlockerLineIndexes(blocked.split(/\r?\n/).slice(1)).size > 0) {
2583
- addGap(accumulator, {
2584
- kind: "inconsistency",
2585
- detail: "blocked-none: Blocked says none despite unresolved errors"
2586
- }, 12);
2587
- }
2588
- const doneRefs = new Set(extractFileRefs(done).map(normalizePath));
2589
- const modifiedPathOwners = buildPathNeedleOwnershipIndex(paths.modified);
2590
- for (const file of extraction.modifiedFiles) {
2591
- const needles = buildUniquePathNeedlesFromIndex(file.path, modifiedPathOwners);
2592
- if (!needles.some((needle) => doneRefs.has(needle)))
2593
- continue;
2594
- const unresolved = collected.unresolved.find((error) => {
2595
- const firstLine = error.message.split(/\r?\n/, 1)[0] ?? "";
2596
- const refs = extractFileRefs(firstLine).map(normalizePath);
2597
- return needles.some((needle) => refs.includes(normalizePath(needle)));
2598
- });
2599
- if (unresolved) {
2600
- addGap(accumulator, {
2601
- kind: "inconsistency",
2602
- detail: file.path + " marked Done but has unresolved error"
2603
- }, 5);
2604
- }
2605
- }
2606
- }
2607
- function verifyOpenLoopsAndClaims(summary, parsed, paths, extraction, continuity, evidence, collected, accumulator) {
2608
- const unresolvedCount = collected.unresolved.length + (continuity?.openLoops.filter((loop) => loop.status !== "resolved").length ?? 0);
2609
- if (unresolvedCount >= 1 && !findSection(parsed, "open-loops") && !summary.toLowerCase().replace(/\\/g, "/").includes("unresolved")) {
2610
- addGap(accumulator, { kind: "missing-open-loops", unresolvedCount }, 5);
2611
- }
2612
- if (!evidence.sourceMessages)
2613
- return;
2614
- const tools = successfulToolEvidence(evidence.sourceMessages);
2615
- for (const claim of outcomeClaims(summary, paths.rendered)) {
2616
- if (!successfulToolSupportsClaim(claim, tools, extraction)) {
2617
- addGap(accumulator, { kind: "unsupported-claim", claim }, 20);
2618
- }
2619
- }
2620
- }
2621
- function verifySummary(summary, extraction, continuity = null, evidence = {}) {
2622
- const parsed = parseSummary(summary);
2623
- const accumulator = { gaps: [], score: 100 };
2624
- const collected = collectVerificationEvidence(extraction, continuity, evidence);
2625
- verifyRequiredSections(parsed, accumulator);
2626
- const paths = verifyPathCoverage(parsed, extraction, continuity, evidence, accumulator);
2627
- verifyErrorEvidence(summary.toLowerCase().replace(/\\/g, "/").replace(/\s+/g, " "), collected, accumulator);
2628
- verifySemanticCoverage(parsed, collected, accumulator);
2629
- verifyFileReferences(summary, extraction, continuity, evidence, collected, paths, accumulator);
2630
- verifyProgressConsistency(parsed, extraction, collected, paths, accumulator);
2631
- verifyOpenLoopsAndClaims(summary, parsed, paths, extraction, continuity, evidence, collected, accumulator);
2632
- const score = Math.max(0, accumulator.score);
2633
- return {
2634
- ok: accumulator.gaps.length === 0 && score >= 85,
2635
- gaps: accumulator.gaps,
2636
- score
2637
- };
2638
- }
2639
-
2640
- // scripts/provider-scenario-eval.ts
2641
- var scenarios = [
2642
- {
2643
- name: "implementation",
2644
- transcript: `User goal: finish JWT key rotation without new dependencies.
2645
- Decision: reuse node:crypto and preserve the existing AuthService API.
2646
- Modified files: src/auth.ts and test/auth.test.ts.
2647
- Unresolved error: expiry boundary test still fails by one second.
2648
- Latest request: fix the test, run the release gate, and do not publish.`,
2649
- extraction: {
2650
- mainGoal: "Finish JWT key rotation",
2651
- messageCount: 12,
2652
- modifiedFiles: [
2653
- { path: "src/auth.ts", toolCalls: 2, lastModifiedIndex: 7 },
2654
- { path: "test/auth.test.ts", toolCalls: 1, lastModifiedIndex: 9 }
2655
- ],
2656
- readFiles: ["package.json"],
2657
- deletedFiles: [],
2658
- errors: [
2659
- {
2660
- index: 10,
2661
- tool: "test",
2662
- message: "expiry boundary test still fails by one second",
2663
- retryAttempted: true,
2664
- resolved: false
2665
- }
2666
- ],
2667
- decisions: [
2668
- {
2669
- index: 4,
2670
- type: "explicit",
2671
- summary: "Reuse node:crypto and preserve the AuthService API"
2672
- }
2673
- ],
2674
- constraints: [
2675
- {
2676
- index: 1,
2677
- text: "Do not add dependencies",
2678
- category: "prohibition",
2679
- confidence: 1
2680
- }
2681
- ],
2682
- topics: [],
2683
- timeline: [],
2684
- lastUserMessages: [
2685
- "Fix the test, run the release gate, and do not publish"
2686
- ],
2687
- lastErrors: []
2688
- }
2689
- },
2690
- {
2691
- name: "debugging",
2692
- transcript: `Goal: stop duplicate background jobs.
2693
- Read src/queue.ts and src/worker.ts; modified src/queue.ts.
2694
- Root cause found: retry scheduling happens before the idempotency key is committed.
2695
- Decision: commit the key first; keep concurrency at two.
2696
- The network timeout was resolved. Open loop: add a regression check for two simultaneous sessions.`,
2697
- extraction: {
2698
- mainGoal: "Stop duplicate background jobs",
2699
- messageCount: 18,
2700
- modifiedFiles: [
2701
- { path: "src/queue.ts", toolCalls: 2, lastModifiedIndex: 13 }
2702
- ],
2703
- readFiles: ["src/queue.ts", "src/worker.ts"],
2704
- deletedFiles: [],
2705
- errors: [],
2706
- decisions: [
2707
- {
2708
- index: 9,
2709
- type: "explicit",
2710
- summary: "Commit the idempotency key before retry scheduling"
2711
- }
2712
- ],
2713
- constraints: [
2714
- {
2715
- index: 10,
2716
- text: "Keep concurrency at two",
2717
- category: "requirement",
2718
- confidence: 1
2719
- }
2720
- ],
2721
- topics: [],
2722
- timeline: [
2723
- {
2724
- index: 16,
2725
- event: "open-loop",
2726
- summary: "Add a simultaneous-session regression check"
2727
- }
2728
- ],
2729
- lastUserMessages: ["Add the regression check next"],
2730
- lastErrors: []
2731
- }
2732
- },
2733
- {
2734
- name: "continuity",
2735
- transcript: `Goal: prepare v8 without publishing.
2736
- Completed milestones: scoped continuity and session locking.
2737
- In progress: provider evaluation. Next: telemetry canary, then dashboard confidence.
2738
- Constraint: selected model must never change automatically; wildcard Pi peer dependencies remain.
2739
- Critical paths: src/app/run-smart-compact.ts, src/infra/llm-client.ts, scripts/eval-gate.ts.`,
2740
- extraction: {
2741
- mainGoal: "Prepare v8 without publishing",
2742
- messageCount: 30,
2743
- modifiedFiles: [
2744
- {
2745
- path: "src/app/run-smart-compact.ts",
2746
- toolCalls: 2,
2747
- lastModifiedIndex: 20
2748
- }
2749
- ],
2750
- readFiles: ["src/infra/llm-client.ts", "scripts/eval-gate.ts"],
2751
- deletedFiles: [],
2752
- errors: [],
2753
- decisions: [
2754
- {
2755
- index: 5,
2756
- type: "explicit",
2757
- summary: "Do not change the selected model automatically"
2758
- }
2759
- ],
2760
- constraints: [
2761
- {
2762
- index: 2,
2763
- text: "Do not publish",
2764
- category: "prohibition",
2765
- confidence: 1
2766
- },
2767
- {
2768
- index: 6,
2769
- text: "Keep Pi peer dependencies as wildcards",
2770
- category: "requirement",
2771
- confidence: 1
2772
- }
2773
- ],
2774
- topics: [],
2775
- timeline: [
2776
- {
2777
- index: 28,
2778
- event: "open-loop",
2779
- summary: "Implement telemetry canary and dashboard confidence"
2780
- }
2781
- ],
2782
- lastUserMessages: [
2783
- "Continue provider evaluation, then telemetry and dashboard confidence"
2784
- ],
2785
- lastErrors: []
2786
- }
2787
- }
2788
- ];
2789
- var modelsArg = process.argv.find((arg) => arg.startsWith("--models="));
2790
- if (!process.argv.includes("--live") || !modelsArg) {
2791
- console.error("Live provider evaluation makes paid API calls. Use --live --models=provider/model,...");
2792
- process.exit(1);
2793
- }
2794
- var requestedModels = modelsArg.slice("--models=".length).split(",").map((value) => value.trim()).filter(Boolean);
2795
- if (!requestedModels.length || requestedModels.length > 8) {
2796
- console.error("Choose 1-8 models");
2797
- process.exit(1);
2798
- }
2799
- var home2 = process.env.HOME ?? "";
2800
- var runtime = await ModelRuntime.create({
2801
- authPath: home2 + "/.pi/agent/auth.json",
2802
- modelsPath: home2 + "/.pi/agent/models.json",
2803
- modelsStorePath: home2 + "/.pi/agent/models-store.json",
2804
- allowModelNetwork: false
2805
- });
2806
- var registry = new ModelRegistry(runtime);
2807
- var models = requestedModels.map((label) => {
2808
- const slash = label.indexOf("/");
2809
- const model = slash > 0 ? registry.find(label.slice(0, slash), label.slice(slash + 1)) : undefined;
2810
- if (!model)
2811
- throw new Error("Unavailable model: " + label);
2812
- return model;
2813
- });
2814
- var systemPrompt = `Produce a faithful coding-session compaction summary. Use exactly these H2 sections:
2815
- ## Goal
2816
- ## Constraints & Preferences
2817
- ## Progress (with H3 Done, In Progress, Blocked)
2818
- ## Key Decisions
2819
- ## Files Modified
2820
- ## Files Read
2821
- ## Next Steps
2822
- ## Critical Context
2823
- Do not invent files, outcomes, or resolved work. Preserve explicit constraints, unresolved errors, and next work.`;
2824
- var results = [];
2825
- for (const model of models) {
2826
- const label = model.provider + "/" + model.id;
2827
- const auth = await registry.getApiKeyAndHeaders(model);
2828
- if (!auth.ok || !auth.apiKey) {
2829
- for (const scenario of scenarios)
2830
- results.push({
2831
- model: label,
2832
- scenario: scenario.name,
2833
- score: 0,
2834
- latencyMs: 0,
2835
- input: 0,
2836
- output: 0,
2837
- error: "auth unavailable"
2838
- });
2839
- continue;
2840
- }
2841
- for (const scenario of scenarios) {
2842
- console.error("Evaluating " + label + " / " + scenario.name);
2843
- const controller = new AbortController;
2844
- const timeout = setTimeout(() => controller.abort("provider-eval-timeout"), 60000);
2845
- const start = Date.now();
2846
- try {
2847
- const response = await rawLlmClient.complete(model, {
2848
- systemPrompt,
2849
- messages: [
2850
- {
2851
- role: "user",
2852
- content: [{ type: "text", text: scenario.transcript }],
2853
- timestamp: Date.now()
2854
- }
2855
- ]
2856
- }, {
2857
- apiKey: auth.apiKey,
2858
- headers: auth.headers,
2859
- maxTokens: 1500,
2860
- reasoning: "minimal",
2861
- codexWatchdogMs: 60000,
2862
- signal: controller.signal
2863
- });
2864
- const summary = response.content.flatMap((item) => item.type === "text" ? [item.text] : []).join(`
2865
- `).trim();
2866
- results.push({
2867
- model: label,
2868
- scenario: scenario.name,
2869
- score: verifySummary(summary, scenario.extraction).score,
2870
- latencyMs: Date.now() - start,
2871
- input: response.usage?.input ?? 0,
2872
- output: response.usage?.output ?? 0
2873
- });
2874
- } catch (error) {
2875
- results.push({
2876
- model: label,
2877
- scenario: scenario.name,
2878
- score: 0,
2879
- latencyMs: Date.now() - start,
2880
- input: 0,
2881
- output: 0,
2882
- error: error instanceof Error ? error.message.slice(0, 160) : String(error).slice(0, 160)
2883
- });
2884
- } finally {
2885
- clearTimeout(timeout);
2886
- }
2887
- }
2888
- }
2889
- function tableCell(value) {
2890
- return String(value).replace(/\\/g, "\\\\").replace(/\|/g, "\\|").replace(/\r?\n/g, " ");
2891
- }
2892
- console.log(`# Live Provider Scenario Matrix
2893
- `);
2894
- console.log("| Model | Scenario | Verify | Latency | Input | Output | Status |");
2895
- console.log("|---|---|---:|---:|---:|---:|---|");
2896
- for (const result of results) {
2897
- console.log("| " + tableCell(result.model) + " | " + tableCell(result.scenario) + " | " + result.score + " | " + result.latencyMs + "ms | " + result.input + " | " + result.output + " | " + (result.error ? "error: " + tableCell(result.error) : "ok") + " |");
2898
- }
2899
- console.log(`
2900
- Advisory only: stage routes stay on the selected model unless explicitly configured.`);