@gotcos/glasses-server 6.36.11 → 6.36.12
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +60 -0
- package/package.json +2 -2
- package/server/lib/agent-session-search.ts +118 -40
- package/server/lib/agent-session-store.ts +23 -1
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,50 @@
|
|
|
1
1
|
## Unreleased
|
|
2
2
|
|
|
3
|
+
## 6.36.12
|
|
4
|
+
- **Claude and Codex now rank candidates by recency before spending the budget.** Both
|
|
5
|
+
walked in raw `readdir` order, so which sessions were reachable came down to filesystem
|
|
6
|
+
layout — `collectCursorDocs` had ranked for a while and these two were the inconsistent
|
|
7
|
+
ones. Statting every candidate first costs ~41ms across 2,118 files. Cursor also charged
|
|
8
|
+
its budget before the keep-warm filter, the same defect, now fixed.
|
|
9
|
+
- **The reach claim in the module docstring was never true and is now measured.** It said
|
|
10
|
+
"older chats stay findable". Ranked and measured on this machine, search reaches roughly
|
|
11
|
+
24 days of Claude, 89 days of Codex and 19 days of Cursor, because once candidates are
|
|
12
|
+
ordered by recency the per-provider doc budget IS the horizon. `EXAMINE_MULTIPLE` was
|
|
13
|
+
re-swept and raised to 12 — the point where examining more files stops finding anything
|
|
14
|
+
older and the doc budget takes over. A sampled older stratum was considered and
|
|
15
|
+
rejected: partial coverage makes a miss uninterpretable, and a search that silently
|
|
16
|
+
samples cannot tell you whether something is absent or merely unsampled.
|
|
17
|
+
- **Embedding batches go out together instead of one after another.** The loop awaited
|
|
18
|
+
each batch in turn, so cost was a round trip per 64 docs and grew as the collector
|
|
19
|
+
returned more — 389 docs is 7 serialized trips at the old size. Now 128 per request,
|
|
20
|
+
all in flight at once. Structural; not measured end to end, because this harness has no
|
|
21
|
+
OpenAI key and the running server was left alone.
|
|
22
|
+
- **Session search was 68% scaffolding, and the budget was the reason.** Of the 1,296
|
|
23
|
+
Claude transcripts on this machine the collector indexed 41, and 28 of those 41 were
|
|
24
|
+
machine prompts — 22 Slack Bridge proxy, 4 reply-with, 2 slack_search_users. Roughly 13
|
|
25
|
+
real conversations were searchable, which is why searching an exact session title
|
|
26
|
+
returned nothing. Two changes, which do not work apart: `isKeepWarmSessionTitle` now
|
|
27
|
+
recognises the machine families by anchored prefix, and both the Claude and Codex
|
|
28
|
+
collectors run that filter — and the Codex `thread_source === 'subagent'` check — BEFORE
|
|
29
|
+
charging the doc budget rather than after. Measured against the real corpus: 41 indexed
|
|
30
|
+
Claude docs with 28 junk becomes **79 indexed with 0 junk**.
|
|
31
|
+
- **The budget now counts docs kept, not files opened.** That is the whole fix: a run of
|
|
32
|
+
machine transcripts used to consume the 134-file allowance and return nothing, so the
|
|
33
|
+
newest real transcript on disk was never reached. A second `examined` ceiling
|
|
34
|
+
(`EXAMINE_MULTIPLE`, 5x the doc budget) stops a pathological corpus walking all 1,296
|
|
35
|
+
files, and the expensive transcript-body read is deferred until a file is being kept.
|
|
36
|
+
- **This also removes rows from COS Control's session LIST.** The predicate is shared by
|
|
37
|
+
all four collectors and the list path, so keep-warm and Slack Bridge entries stop
|
|
38
|
+
appearing there too. That is intended, not a side effect to fix.
|
|
39
|
+
- **Control may not see any of this yet.** The search route's median is ~2.27s against
|
|
40
|
+
Control's 2s client timeout, of which ~1.7s is two sequential embedding round trips;
|
|
41
|
+
this work adds ~350ms on top. Until that timeout and `EMBED_BATCH` are addressed,
|
|
42
|
+
Control falls back to its local scanner and reports a fabricated `server_too_old`.
|
|
43
|
+
- **The scan budget had no test coverage at all.** `collectAgentSessionSearchDocs` was
|
|
44
|
+
never called by any test and its `cap` was never exercised, so the branch that spends
|
|
45
|
+
the budget had never run. 21 tests added, each verified to FAIL against the previous
|
|
46
|
+
code before being kept.
|
|
47
|
+
|
|
3
48
|
## 6.36.11
|
|
4
49
|
- **Fences now record WHY, so the population can be measured before anything resolves
|
|
5
50
|
automatically.** Two plans designed an automatic fence resolver and both were rejected —
|
|
@@ -54,6 +99,21 @@
|
|
|
54
99
|
default install (`COS_THREAD_FENCE_DURABLE` unset) nothing is written to disk at all;
|
|
55
100
|
and reading the distribution means reading `thread-fences.json` or the server log —
|
|
56
101
|
there is no UI for it.
|
|
102
|
+
- **A live session was reporting itself hours idle.** Separate from the fence work above.
|
|
103
|
+
`liveClaudeRows` builds a row's `modified` from the peer registry's `lastActiveAt`, which
|
|
104
|
+
tracks the REGISTRY record and not the transcript — so a session that is actively writing
|
|
105
|
+
keeps reporting whenever the registry last moved. Measured on three live sessions
|
|
106
|
+
2026-08-18: the wire said 55.3m / 407.7m / 435.0m old while their transcripts had been
|
|
107
|
+
written 0.1m / 0.2m / 5.1m earlier. Under-reporting by up to 7.2 hours. Shipped in
|
|
108
|
+
66dff88; `enrichLiveClaude` already resolves the transcript path and reads the file twice,
|
|
109
|
+
so the true mtime costs one `stat`. A resolved file that fails to stat keeps the
|
|
110
|
+
heartbeat; a session with no transcript at all (2 of 6 measured) returns early on the
|
|
111
|
+
existing guard. Prerequisite for any surface that renders a real date — without it,
|
|
112
|
+
showing the timestamp displays an actively-writing session as seven hours stale.
|
|
113
|
+
Coverage: `enrichLiveClaude` had NO execution coverage before this (every existing test
|
|
114
|
+
passes an empty live array). Two tests now drive it through `listAgentSessions`. Three
|
|
115
|
+
mutations, two caught; the third (the stat-failure fallback) SURVIVES because the
|
|
116
|
+
missing-file guard returns first, so that branch is unreached. The code says so.
|
|
57
117
|
|
|
58
118
|
## 6.36.10
|
|
59
119
|
- **A fenced thread had no exit and left no trace.** An ambiguous delivery fences the
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@gotcos/glasses-server",
|
|
3
|
-
"version": "6.36.
|
|
4
|
-
"description": "COS Glasses
|
|
3
|
+
"version": "6.36.12",
|
|
4
|
+
"description": "COS Glasses — self-hosted AI heads-up-display server for Even G2 smart glasses, powered by Claude Code, Codex, or Cursor Agent CLI",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"bin": {
|
|
7
7
|
"glasses-server": "bin/cli.cjs",
|
|
@@ -4,7 +4,18 @@
|
|
|
4
4
|
* Keyword: local title / sidebar name / first prompt / transcript scan. No model.
|
|
5
5
|
* Semantic: one OpenAI query embedding scored against those same texts.
|
|
6
6
|
* Sessions have no Qdrant collection — this is not meeting or memory search.
|
|
7
|
-
*
|
|
7
|
+
*
|
|
8
|
+
* REACH, measured 2026-08-18 rather than claimed. The 7-day list window does not apply,
|
|
9
|
+
* but this is not unbounded either: each provider gets a doc budget of MAX_SCAN_FILES/3,
|
|
10
|
+
* and once candidates are ranked by recency that budget IS the horizon. On this machine
|
|
11
|
+
* that is roughly 24 days of Claude, 89 days of Codex and 19 days of Cursor. An older
|
|
12
|
+
* chat is reachable only if its provider has not filled its budget with newer ones.
|
|
13
|
+
*
|
|
14
|
+
* This previously read "older chats stay findable", which was written before ranking and
|
|
15
|
+
* was never measured. Widening it means a bigger budget, which costs embedding time on
|
|
16
|
+
* the route -- not a deeper examine ceiling, which saturates. A sampled older stratum was
|
|
17
|
+
* considered and rejected: partial coverage makes a miss uninterpretable, and a search
|
|
18
|
+
* that silently samples cannot tell you whether a thing is absent or merely unsampled.
|
|
8
19
|
*/
|
|
9
20
|
|
|
10
21
|
import { join } from 'node:path'
|
|
@@ -46,11 +57,51 @@ import {
|
|
|
46
57
|
|
|
47
58
|
const HEAD_TEXT = 8_000
|
|
48
59
|
const MAX_SCAN_FILES = 400
|
|
60
|
+
|
|
61
|
+
/**
|
|
62
|
+
* How many files a provider may OPEN per search, as a multiple of the docs it may KEEP.
|
|
63
|
+
*
|
|
64
|
+
* Before the filter moved ahead of the decrement, `remaining` meant "files read" -- so a
|
|
65
|
+
* run of machine transcripts consumed the whole allowance and returned nothing. Now it
|
|
66
|
+
* means "docs kept", which on its own would let a pathological corpus walk all 1,296
|
|
67
|
+
* files. This is the ceiling that stops it.
|
|
68
|
+
*
|
|
69
|
+
* Swept 2026-08-18 against the real corpus (1,296 Claude transcripts, 822 Codex
|
|
70
|
+
* rollouts). Docs kept vs collector wall time, median of 5 warm runs:
|
|
71
|
+
*
|
|
72
|
+
* Re-swept after recency ranking landed, since ranking changes which files the ceiling
|
|
73
|
+
* is spent on -- the newest region of the corpus is the densest in machine transcripts:
|
|
74
|
+
*
|
|
75
|
+
* x5 329 docs 74 claude 11.4d window 1287 ms
|
|
76
|
+
* x8 389 docs 134 claude 21.3d window 1585 ms
|
|
77
|
+
* x12 389 docs 134 claude 24.5d window 1697 ms <- chosen
|
|
78
|
+
* x16 389 docs 134 claude 24.5d window 1637 ms
|
|
79
|
+
*
|
|
80
|
+
* x12 is where the ceiling stops being the constraint and the 134-doc budget takes over:
|
|
81
|
+
* past it, examining more files finds nothing older. That is the principled stopping
|
|
82
|
+
* point -- raise until the other limit binds, then stop.
|
|
83
|
+
*/
|
|
84
|
+
const EXAMINE_MULTIPLE = 12
|
|
85
|
+
|
|
86
|
+
/**
|
|
87
|
+
* `remaining` counts docs KEPT. `examined` counts files OPENED. They are different
|
|
88
|
+
* numbers because most files on this machine are filtered out after their title is read.
|
|
89
|
+
*/
|
|
90
|
+
type ScanBudget = { remaining: number; examined: number }
|
|
49
91
|
const SEMANTIC_DOC_CAP = 120
|
|
50
92
|
const SEMANTIC_MIN = 0.28
|
|
51
93
|
const EMBED_MODEL = 'text-embedding-3-small'
|
|
52
94
|
const EMBED_TIMEOUT_MS = 12_000
|
|
53
|
-
|
|
95
|
+
/**
|
|
96
|
+
* 128 inputs per request, and the requests go out TOGETHER.
|
|
97
|
+
*
|
|
98
|
+
* This loop used to `await` each batch in turn, so the embedding cost was a round trip
|
|
99
|
+
* per 64 docs -- about 1.7s of a 2.3s route, and it got worse as the collector started
|
|
100
|
+
* returning more docs (389 docs is 7 serialized trips at the old batch size, 4 at this
|
|
101
|
+
* one). Serialization was the expensive part, not the batch size, so both changed:
|
|
102
|
+
* fewer requests, and one round trip of latency instead of N.
|
|
103
|
+
*/
|
|
104
|
+
const EMBED_BATCH = 128
|
|
54
105
|
|
|
55
106
|
export interface AgentSessionSearchHit extends AgentSessionRow {
|
|
56
107
|
snippet: string
|
|
@@ -139,7 +190,9 @@ export async function defaultEmbedTexts(texts: string[]): Promise<number[][] | {
|
|
|
139
190
|
if (!key) return { reason: 'no_session_embeddings' }
|
|
140
191
|
if (texts.length === 0) return []
|
|
141
192
|
const vectors: number[][] = new Array(texts.length)
|
|
142
|
-
|
|
193
|
+
const offsets: number[] = []
|
|
194
|
+
for (let offset = 0; offset < texts.length; offset += EMBED_BATCH) offsets.push(offset)
|
|
195
|
+
const failed = await Promise.all(offsets.map(async offset => {
|
|
143
196
|
const slice = texts.slice(offset, offset + EMBED_BATCH).map(text => text.slice(0, HEAD_TEXT))
|
|
144
197
|
try {
|
|
145
198
|
const response = await fetch('https://api.openai.com/v1/embeddings', {
|
|
@@ -151,17 +204,19 @@ export async function defaultEmbedTexts(texts: string[]): Promise<number[][] | {
|
|
|
151
204
|
body: JSON.stringify({ model: EMBED_MODEL, input: slice }),
|
|
152
205
|
signal: AbortSignal.timeout(EMBED_TIMEOUT_MS),
|
|
153
206
|
})
|
|
154
|
-
if (!response.ok) return
|
|
207
|
+
if (!response.ok) return true
|
|
155
208
|
const parsed = await response.json() as { data?: Array<{ embedding?: number[]; index?: number }> }
|
|
156
209
|
const rows = Array.isArray(parsed.data) ? parsed.data : []
|
|
157
210
|
for (const row of rows) {
|
|
158
211
|
const index = typeof row.index === 'number' ? row.index : 0
|
|
159
212
|
if (Array.isArray(row.embedding)) vectors[offset + index] = row.embedding
|
|
160
213
|
}
|
|
214
|
+
return false
|
|
161
215
|
} catch {
|
|
162
|
-
return
|
|
216
|
+
return true
|
|
163
217
|
}
|
|
164
|
-
}
|
|
218
|
+
}))
|
|
219
|
+
if (failed.some(Boolean)) return { reason: 'embeddings_unreachable' }
|
|
165
220
|
if (vectors.some(row => !Array.isArray(row) || row.length === 0)) return { reason: 'embeddings_unreachable' }
|
|
166
221
|
return vectors
|
|
167
222
|
}
|
|
@@ -180,40 +235,53 @@ function pushDoc(docs: SearchDoc[], next: SearchDoc) {
|
|
|
180
235
|
}
|
|
181
236
|
}
|
|
182
237
|
|
|
183
|
-
async function collectClaudeDocs(roots: AgentSessionRoots, docs: SearchDoc[], budget:
|
|
238
|
+
async function collectClaudeDocs(roots: AgentSessionRoots, docs: SearchDoc[], budget: ScanBudget) {
|
|
184
239
|
const starred = await loadClaudeStarredIds(roots.claudeDesktopConfig)
|
|
240
|
+
// Stat every candidate and rank by recency BEFORE spending anything, the shape
|
|
241
|
+
// `collectCursorDocs` already used. Claude and Codex were the inconsistent ones: they
|
|
242
|
+
// walked in raw `readdir` order, so which sessions were reachable came down to
|
|
243
|
+
// filesystem layout and the newest transcript on disk was routinely never opened.
|
|
244
|
+
// Statting is not the expensive part -- measured at 41ms across 2,118 candidates.
|
|
245
|
+
const candidates: Array<{ file: string; native: string; folder: string; mtimeMs: number; birthtimeMs: number }> = []
|
|
185
246
|
for (const folder of await dirents(roots.claudeProjects)) {
|
|
186
247
|
const dir = join(roots.claudeProjects, folder)
|
|
187
248
|
for (const name of await dirents(dir)) {
|
|
188
|
-
if (!CLAUDE_UUID_JSONL.test(name)
|
|
249
|
+
if (!CLAUDE_UUID_JSONL.test(name)) continue
|
|
189
250
|
const file = join(dir, name)
|
|
190
251
|
const st = await fileStat(file)
|
|
191
252
|
if (!st?.isFile) continue
|
|
192
|
-
|
|
193
|
-
const native = name.slice(0, -6)
|
|
194
|
-
const custom = await lastCustomTitle(file)
|
|
195
|
-
const first = await firstClaudeUserTitle(file)
|
|
196
|
-
const users = await transcriptHaystack('claude', file)
|
|
197
|
-
const title = custom || first || 'Claude session'
|
|
198
|
-
if (isKeepWarmSessionTitle(title)) continue
|
|
199
|
-
const haystack = clip(`${title}\n${custom || ''}\n${first || ''}\n${workspaceLabel(folder)}\n${users}`)
|
|
200
|
-
pushDoc(docs, {
|
|
201
|
-
title,
|
|
202
|
-
haystack,
|
|
203
|
-
row: {
|
|
204
|
-
session_id: native,
|
|
205
|
-
provider: 'claude',
|
|
206
|
-
display_label: title,
|
|
207
|
-
project: workspaceLabel(folder),
|
|
208
|
-
modified: isoFromMtime(st.mtimeMs),
|
|
209
|
-
created: isoFromMtime(st.birthtimeMs),
|
|
210
|
-
alive: false,
|
|
211
|
-
state: 'recent',
|
|
212
|
-
pinned: starred.has(native.toLowerCase()),
|
|
213
|
-
},
|
|
214
|
-
})
|
|
253
|
+
candidates.push({ file, native: name.slice(0, -6), folder, mtimeMs: st.mtimeMs, birthtimeMs: st.birthtimeMs })
|
|
215
254
|
}
|
|
216
255
|
}
|
|
256
|
+
candidates.sort((a, b) => b.mtimeMs - a.mtimeMs)
|
|
257
|
+
for (const candidate of candidates) {
|
|
258
|
+
if (budget.remaining <= 0 || budget.examined <= 0) break
|
|
259
|
+
budget.examined -= 1
|
|
260
|
+
const custom = await lastCustomTitle(candidate.file)
|
|
261
|
+
const first = await firstClaudeUserTitle(candidate.file)
|
|
262
|
+
const title = custom || first || 'Claude session'
|
|
263
|
+
// Both filters run BEFORE the doc budget is charged, and the haystack -- by far the
|
|
264
|
+
// most expensive read here -- runs only for a file we are actually keeping.
|
|
265
|
+
if (isKeepWarmSessionTitle(title)) continue
|
|
266
|
+
budget.remaining -= 1
|
|
267
|
+
const users = await transcriptHaystack('claude', candidate.file)
|
|
268
|
+
const haystack = clip(`${title}\n${custom || ''}\n${first || ''}\n${workspaceLabel(candidate.folder)}\n${users}`)
|
|
269
|
+
pushDoc(docs, {
|
|
270
|
+
title,
|
|
271
|
+
haystack,
|
|
272
|
+
row: {
|
|
273
|
+
session_id: candidate.native,
|
|
274
|
+
provider: 'claude',
|
|
275
|
+
display_label: title,
|
|
276
|
+
project: workspaceLabel(candidate.folder),
|
|
277
|
+
modified: isoFromMtime(candidate.mtimeMs),
|
|
278
|
+
created: isoFromMtime(candidate.birthtimeMs),
|
|
279
|
+
alive: false,
|
|
280
|
+
state: 'recent',
|
|
281
|
+
pinned: starred.has(candidate.native.toLowerCase()),
|
|
282
|
+
},
|
|
283
|
+
})
|
|
284
|
+
}
|
|
217
285
|
for (const starredId of starred) {
|
|
218
286
|
if (docs.some(doc => doc.row.provider === 'claude' && doc.row.session_id.toLowerCase() === starredId)) continue
|
|
219
287
|
if (budget.remaining <= 0) break
|
|
@@ -221,10 +289,10 @@ async function collectClaudeDocs(roots: AgentSessionRoots, docs: SearchDoc[], bu
|
|
|
221
289
|
if (!desktop) continue
|
|
222
290
|
const st = await fileStat(desktop)
|
|
223
291
|
if (!st?.isFile) continue
|
|
224
|
-
budget.remaining -= 1
|
|
225
292
|
const head = peekClaudeDesktopHead(await readWindow(desktop, false))
|
|
226
293
|
const title = head.title || 'Claude session'
|
|
227
294
|
if (isKeepWarmSessionTitle(title)) continue
|
|
295
|
+
budget.remaining -= 1
|
|
228
296
|
pushDoc(docs, {
|
|
229
297
|
title,
|
|
230
298
|
haystack: clip(`${title}\n${head.cwd}\n${workspaceLabel(head.cwd)}`),
|
|
@@ -243,23 +311,31 @@ async function collectClaudeDocs(roots: AgentSessionRoots, docs: SearchDoc[], bu
|
|
|
243
311
|
}
|
|
244
312
|
}
|
|
245
313
|
|
|
246
|
-
async function collectCodexDocs(roots: AgentSessionRoots, docs: SearchDoc[], budget:
|
|
314
|
+
async function collectCodexDocs(roots: AgentSessionRoots, docs: SearchDoc[], budget: ScanBudget) {
|
|
247
315
|
const names = await loadCodexThreadNames(roots.codexSessions)
|
|
248
316
|
const pinned = await loadCodexPinnedIds(roots.codexSessions)
|
|
317
|
+
const candidates: Array<{ file: string; mtimeMs: number; birthtimeMs: number }> = []
|
|
249
318
|
for (const file of await listCodexJsonlFiles(roots.codexSessions)) {
|
|
250
|
-
if (budget.remaining <= 0) break
|
|
251
319
|
const st = await fileStat(file)
|
|
252
320
|
if (!st?.isFile) continue
|
|
253
|
-
|
|
321
|
+
candidates.push({ file, mtimeMs: st.mtimeMs, birthtimeMs: st.birthtimeMs })
|
|
322
|
+
}
|
|
323
|
+
candidates.sort((a, b) => b.mtimeMs - a.mtimeMs)
|
|
324
|
+
for (const { file, mtimeMs, birthtimeMs } of candidates) {
|
|
325
|
+
if (budget.remaining <= 0 || budget.examined <= 0) break
|
|
326
|
+
budget.examined -= 1
|
|
254
327
|
const meta = await peekCodexMeta(file)
|
|
328
|
+
// Subagent rollouts are the Codex analogue of the keep-warm transcripts: 427 of 822
|
|
329
|
+
// on this machine per the 2026-08-18 census. Filtered before the doc budget is charged.
|
|
255
330
|
if (!meta || meta.subagent) continue
|
|
256
331
|
const name = file.split('/').pop() || file
|
|
257
332
|
const native = meta.id || name.slice(0, -6)
|
|
258
333
|
const thread = names.get(native) || ''
|
|
259
334
|
const title = thread || meta.title || 'Codex session'
|
|
260
335
|
if (isKeepWarmSessionTitle(title)) continue
|
|
336
|
+
budget.remaining -= 1
|
|
261
337
|
const users = await transcriptHaystack('codex', file)
|
|
262
|
-
const created = meta.created || createdFromCodexFilename(name) || isoFromMtime(
|
|
338
|
+
const created = meta.created || createdFromCodexFilename(name) || isoFromMtime(birthtimeMs)
|
|
263
339
|
pushDoc(docs, {
|
|
264
340
|
title,
|
|
265
341
|
haystack: clip(`${title}\n${thread}\n${meta.title}\n${meta.cwd}\n${users}`),
|
|
@@ -268,7 +344,7 @@ async function collectCodexDocs(roots: AgentSessionRoots, docs: SearchDoc[], bud
|
|
|
268
344
|
provider: 'codex',
|
|
269
345
|
display_label: title,
|
|
270
346
|
project: workspaceLabel(meta.cwd),
|
|
271
|
-
modified: isoFromMtime(
|
|
347
|
+
modified: isoFromMtime(mtimeMs),
|
|
272
348
|
created,
|
|
273
349
|
alive: false,
|
|
274
350
|
state: 'recent',
|
|
@@ -305,11 +381,12 @@ async function collectCursorDocs(
|
|
|
305
381
|
const ranked = [...byId.entries()].sort((a, b) => b[1].mtimeMs - a[1].mtimeMs)
|
|
306
382
|
for (const [sessionDir, candidate] of ranked) {
|
|
307
383
|
if (budget.remaining <= 0) break
|
|
308
|
-
budget.remaining -= 1
|
|
309
384
|
const sidebar = composerNames.get(sessionDir) || ''
|
|
310
385
|
const users = await transcriptHaystack('cursor', candidate.file)
|
|
311
386
|
const title = sidebar || users.split('\n')[0] || 'Cursor session'
|
|
387
|
+
// Cursor already ranked, but it charged the budget before this filter too.
|
|
312
388
|
if (isKeepWarmSessionTitle(title)) continue
|
|
389
|
+
budget.remaining -= 1
|
|
313
390
|
pushDoc(docs, {
|
|
314
391
|
title,
|
|
315
392
|
haystack: clip(`${title}\n${sidebar}\n${candidate.project}\n${users}`),
|
|
@@ -335,8 +412,9 @@ export async function collectAgentSessionSearchDocs(
|
|
|
335
412
|
): Promise<SearchDoc[]> {
|
|
336
413
|
const docs: SearchDoc[] = []
|
|
337
414
|
const share = Math.max(1, Math.ceil(Math.min(cap, MAX_SCAN_FILES) / 3))
|
|
338
|
-
|
|
339
|
-
await
|
|
415
|
+
const examined = share * EXAMINE_MULTIPLE
|
|
416
|
+
await collectClaudeDocs(roots, docs, { remaining: share, examined })
|
|
417
|
+
await collectCodexDocs(roots, docs, { remaining: share, examined })
|
|
340
418
|
await collectCursorDocs(roots, docs, { remaining: share }, now)
|
|
341
419
|
docs.sort((a, b) => (b.row.modified || '').localeCompare(a.row.modified || ''))
|
|
342
420
|
return docs.slice(0, Math.max(1, Math.min(cap, MAX_SCAN_FILES)))
|
|
@@ -97,10 +97,32 @@ export function isScratchCursorProject(folder: string): boolean {
|
|
|
97
97
|
|
|
98
98
|
export const isSkippedCursorFolder = isScratchCursorProject
|
|
99
99
|
|
|
100
|
+
/**
|
|
101
|
+
* Titles that belong to a MACHINE, not to a conversation Miles had.
|
|
102
|
+
*
|
|
103
|
+
* Measured 2026-08-18 by driving the real collector over the 1,296 Claude transcripts on
|
|
104
|
+
* this machine: of the 41 Claude docs it indexed, 28 were machine prompts -- 22 Slack
|
|
105
|
+
* Bridge proxy, 4 reply-with, 2 slack_search_users. Two thirds of Claude search was
|
|
106
|
+
* answering with scaffolding.
|
|
107
|
+
*
|
|
108
|
+
* Every `you are ...` prefix here is ANCHORED to a specific known caller. A bare
|
|
109
|
+
* `startsWith('you are')` would also swallow any human sentence beginning that way
|
|
110
|
+
* ("You are right that ..."), and the anchored pair covers the measured volume without
|
|
111
|
+
* that risk.
|
|
112
|
+
*
|
|
113
|
+
* Shared by all four search collectors AND the session-list path, so a title added here
|
|
114
|
+
* also stops appearing as a row in COS Control's session list. That is the intent.
|
|
115
|
+
*/
|
|
100
116
|
export function isKeepWarmSessionTitle(title: string): boolean {
|
|
101
117
|
const t = title.trim().toLowerCase()
|
|
102
118
|
if (t === 'ready') return true
|
|
103
|
-
|
|
119
|
+
if (t.startsWith('this is an automated local readiness check')) return true
|
|
120
|
+
if (t.startsWith('you are the cos slack bridge proxy')) return true
|
|
121
|
+
if (t.startsWith('you are a post-processing editor')) return true
|
|
122
|
+
if (t.startsWith('call mcp__claude_ai_slack__slack_search_users')) return true
|
|
123
|
+
if (/^reply with (exactly|the single word)\b/.test(t)) return true
|
|
124
|
+
if (t === 'say ok' || t === 'say: ok' || t === 'reply ok') return true
|
|
125
|
+
return false
|
|
104
126
|
}
|
|
105
127
|
|
|
106
128
|
/**
|