@remnic/core 9.3.645 → 9.3.647
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/access-cli.js +18 -17
- package/dist/access-cli.js.map +1 -1
- package/dist/access-http.js +13 -13
- package/dist/access-mcp.js +10 -10
- package/dist/access-schema.js +3 -3
- package/dist/access-service.js +8 -8
- package/dist/adapters/index.js +4 -4
- package/dist/adapters/registry.js +2 -2
- package/dist/{capsule-crypto-YO5QJ6L3.js → capsule-crypto-GWVG7LGC.js} +2 -2
- package/dist/{chunk-6YLJZ2EU.js → chunk-3D6L7CEP.js} +92 -29
- package/dist/chunk-3D6L7CEP.js.map +1 -0
- package/dist/{chunk-3SF7QNKF.js → chunk-6BNFVP7Y.js} +2 -2
- package/dist/{chunk-AGNBY3VG.js → chunk-APJQ6UEA.js} +4 -4
- package/dist/{chunk-HRXCPKN4.js → chunk-APWJRJFW.js} +2 -2
- package/dist/{chunk-VDX2J7OX.js → chunk-BUKK5SWA.js} +2 -2
- package/dist/{chunk-T22TZJWJ.js → chunk-DC66QVL2.js} +2 -2
- package/dist/chunk-EVWIEEKZ.js +315 -0
- package/dist/chunk-EVWIEEKZ.js.map +1 -0
- package/dist/{chunk-IZXOO6NP.js → chunk-FAYDM5WD.js} +2 -2
- package/dist/{chunk-Y2XKJJEI.js → chunk-IMWFHBG2.js} +10 -33
- package/dist/chunk-IMWFHBG2.js.map +1 -0
- package/dist/{chunk-W5ISYSIL.js → chunk-L7W5YW6Y.js} +4 -4
- package/dist/{chunk-DGNQRNLL.js → chunk-NT5TINK5.js} +2 -2
- package/dist/{chunk-BXLOS5AJ.js → chunk-OWHERGF2.js} +2 -2
- package/dist/{chunk-CDHRUNBX.js → chunk-QKE4LHNR.js} +2 -2
- package/dist/{chunk-BTRL6GG7.js → chunk-RAELB5NX.js} +5 -5
- package/dist/{chunk-2QSZNTDO.js → chunk-RKNJBZ55.js} +4 -4
- package/dist/chunk-S4DDLTPX.js +140 -0
- package/dist/chunk-S4DDLTPX.js.map +1 -0
- package/dist/{chunk-MBT3ASYV.js → chunk-TA4LQ5SR.js} +12 -12
- package/dist/{chunk-KVUG7FF7.js → chunk-U55D5UD5.js} +7 -7
- package/dist/{chunk-DQEMWVMT.js → chunk-UVYI6VIX.js} +1 -1
- package/dist/{chunk-D2B22JDF.js → chunk-WEPMT6SC.js} +7 -7
- package/dist/{chunk-IHG27THY.js → chunk-XUGQQPGO.js} +77 -61
- package/dist/chunk-XUGQQPGO.js.map +1 -0
- package/dist/{chunk-YOXCJP3A.js → chunk-ZPPFKVSD.js} +3 -3
- package/dist/cli.js +30 -28
- package/dist/{first-start-migration-FF7YFGRP.js → first-start-migration-PG5HBC3K.js} +4 -4
- package/dist/index.d.ts +2 -0
- package/dist/index.js +77 -63
- package/dist/index.js.map +1 -1
- package/dist/lcm/engine.js +3 -3
- package/dist/lcm/index.js +11 -11
- package/dist/namespaces/migrate.js +5 -5
- package/dist/namespaces/search.js +4 -4
- package/dist/operator-toolkit.js +6 -6
- package/dist/orchestrator.js +14 -13
- package/dist/resume-bundles.js +3 -2
- package/dist/schemas.d.ts +22 -22
- package/dist/search/factory.js +3 -3
- package/dist/search/index.js +7 -7
- package/dist/session-identity.d.ts +118 -0
- package/dist/session-identity.js +21 -0
- package/dist/session-identity.js.map +1 -0
- package/dist/session-transcript-migration.d.ts +65 -0
- package/dist/session-transcript-migration.js +16 -0
- package/dist/session-transcript-migration.js.map +1 -0
- package/dist/summarizer.js +2 -1
- package/dist/transcript.d.ts +18 -6
- package/dist/transcript.js +2 -1
- package/dist/transfer/backup.js +2 -2
- package/dist/transfer/capsule-export.js +2 -2
- package/dist/transfer/capsule-import.js +2 -2
- package/dist/transfer/import-sqlite.js +2 -2
- package/dist/transfer/types.d.ts +12 -12
- package/package.json +1 -1
- package/src/cli.ts +79 -0
- package/src/index.ts +18 -0
- package/src/session-identity.test.ts +130 -0
- package/src/session-identity.ts +281 -0
- package/src/session-transcript-migration.test.ts +350 -0
- package/src/session-transcript-migration.ts +527 -0
- package/src/summarizer.ts +15 -35
- package/src/transcript-session-identity.test.ts +424 -0
- package/src/transcript.test.ts +110 -0
- package/src/transcript.ts +121 -73
- package/dist/chunk-6YLJZ2EU.js.map +0 -1
- package/dist/chunk-IHG27THY.js.map +0 -1
- package/dist/chunk-Y2XKJJEI.js.map +0 -1
- /package/dist/{capsule-crypto-YO5QJ6L3.js.map → capsule-crypto-GWVG7LGC.js.map} +0 -0
- /package/dist/{chunk-3SF7QNKF.js.map → chunk-6BNFVP7Y.js.map} +0 -0
- /package/dist/{chunk-AGNBY3VG.js.map → chunk-APJQ6UEA.js.map} +0 -0
- /package/dist/{chunk-HRXCPKN4.js.map → chunk-APWJRJFW.js.map} +0 -0
- /package/dist/{chunk-VDX2J7OX.js.map → chunk-BUKK5SWA.js.map} +0 -0
- /package/dist/{chunk-T22TZJWJ.js.map → chunk-DC66QVL2.js.map} +0 -0
- /package/dist/{chunk-IZXOO6NP.js.map → chunk-FAYDM5WD.js.map} +0 -0
- /package/dist/{chunk-W5ISYSIL.js.map → chunk-L7W5YW6Y.js.map} +0 -0
- /package/dist/{chunk-DGNQRNLL.js.map → chunk-NT5TINK5.js.map} +0 -0
- /package/dist/{chunk-BXLOS5AJ.js.map → chunk-OWHERGF2.js.map} +0 -0
- /package/dist/{chunk-CDHRUNBX.js.map → chunk-QKE4LHNR.js.map} +0 -0
- /package/dist/{chunk-BTRL6GG7.js.map → chunk-RAELB5NX.js.map} +0 -0
- /package/dist/{chunk-2QSZNTDO.js.map → chunk-RKNJBZ55.js.map} +0 -0
- /package/dist/{chunk-MBT3ASYV.js.map → chunk-TA4LQ5SR.js.map} +0 -0
- /package/dist/{chunk-KVUG7FF7.js.map → chunk-U55D5UD5.js.map} +0 -0
- /package/dist/{chunk-DQEMWVMT.js.map → chunk-UVYI6VIX.js.map} +0 -0
- /package/dist/{chunk-D2B22JDF.js.map → chunk-WEPMT6SC.js.map} +0 -0
- /package/dist/{chunk-YOXCJP3A.js.map → chunk-ZPPFKVSD.js.map} +0 -0
- /package/dist/{first-start-migration-FF7YFGRP.js.map → first-start-migration-PG5HBC3K.js.map} +0 -0
|
@@ -0,0 +1,527 @@
|
|
|
1
|
+
import { mkdir, readFile, readdir, rename, stat, unlink, writeFile } from "node:fs/promises";
|
|
2
|
+
import path from "node:path";
|
|
3
|
+
import { log } from "./logger.js";
|
|
4
|
+
import { displayErrorDetail } from "./runtime/better-sqlite.js";
|
|
5
|
+
import { parseSessionIdentity, sessionStoragePaths } from "./session-identity.js";
|
|
6
|
+
import { resolveSafeStoragePath } from "./storage-paths.js";
|
|
7
|
+
import type { TranscriptEntry } from "./types.js";
|
|
8
|
+
|
|
9
|
+
/**
|
|
10
|
+
* Lossless migration of legacy `transcripts/other/default/*.jsonl` files into
|
|
11
|
+
* first-class `transcripts/session/<hash>/*.jsonl` directories (issue #1496).
|
|
12
|
+
*
|
|
13
|
+
* Older builds routed every non-legacy session key (e.g. `pi-geek:abc123`)
|
|
14
|
+
* into the shared `other/default` directory on first write. That conflated
|
|
15
|
+
* distinct standalone-client sessions into one transcript directory. This
|
|
16
|
+
* migration splits those mixed files back out, grouping entries by
|
|
17
|
+
* `entry.sessionKey` and re-homing each distinct session under its own
|
|
18
|
+
* deterministic hashed directory.
|
|
19
|
+
*
|
|
20
|
+
* Safety contract (rules #54, #51, #18, #35):
|
|
21
|
+
* - Dry-run is the default; nothing is written without `apply: true`.
|
|
22
|
+
* - Writes go to a temp file then atomic-rename (never delete-before-write).
|
|
23
|
+
* - JSONL line ordering and byte content are preserved per session.
|
|
24
|
+
* - Idempotent: re-running after apply finds nothing left to migrate.
|
|
25
|
+
* - A manifest is returned (and written under `state/` on apply) as an audit
|
|
26
|
+
* trail.
|
|
27
|
+
*/
|
|
28
|
+
|
|
29
|
+
/**
|
|
30
|
+
* Channel type that NEW writes use for first-class arbitrary keys. Subdirs of
|
|
31
|
+
* this type whose `<id>` is a canonical `storagePathHash` (16 lowercase hex)
|
|
32
|
+
* are already homed — every line there is, by definition, already in its
|
|
33
|
+
* destination directory, so they are never migration sources.
|
|
34
|
+
*/
|
|
35
|
+
const FIRST_CLASS_CHANNEL_TYPE = "session";
|
|
36
|
+
|
|
37
|
+
/**
|
|
38
|
+
* Matches a canonical `session/<hash>` channel id produced by
|
|
39
|
+
* {@link storagePathHash} (16 lowercase hex chars). Used to tell a NEW
|
|
40
|
+
* first-class directory (skip) apart from a LEGACY `session/<name>` directory.
|
|
41
|
+
*
|
|
42
|
+
* Pre-#1496 the OLD `parts.length >= 3` parser stored a key whose 3rd colon
|
|
43
|
+
* segment was literally `session` (e.g. `foo:bar:session:baz`) under
|
|
44
|
+
* `transcripts/session/baz` — a non-hash id. Those legacy dirs MUST still be
|
|
45
|
+
* scanned and split, while genuine `session/<hash>` data is left untouched
|
|
46
|
+
* (codex review on PR #1496 / PR #1504). The hex length is fixed because
|
|
47
|
+
* `storagePathHash` slices the sha256 hex digest to 16 chars.
|
|
48
|
+
*/
|
|
49
|
+
const CANONICAL_SESSION_HASH_RE = /^[0-9a-f]{16}$/;
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* True when `<type>/<id>` is a canonical first-class `session/<hash>` directory
|
|
53
|
+
* (NEW write target) rather than a legacy stranded directory. Defensive: also
|
|
54
|
+
* confirms the id is exactly what `storagePathHash` would produce for some
|
|
55
|
+
* value by shape — a 16-hex string — so we never misclassify a legacy
|
|
56
|
+
* `session/<name>` dir as homed and skip migrating it.
|
|
57
|
+
*/
|
|
58
|
+
function isCanonicalSessionDir(channelType: string, channelId: string): boolean {
|
|
59
|
+
return channelType === FIRST_CLASS_CHANNEL_TYPE && CANONICAL_SESSION_HASH_RE.test(channelId);
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
export interface SessionMigrationEntryGroup {
|
|
63
|
+
/** The distinct session key these lines belong to. */
|
|
64
|
+
sessionKey: string;
|
|
65
|
+
/** Whether this key is a recognized legacy `agent:<id>:...` shape. */
|
|
66
|
+
legacy: boolean;
|
|
67
|
+
/** Destination directory (relative to the transcripts root). */
|
|
68
|
+
destDir: string;
|
|
69
|
+
/** Number of JSONL entries that will move. */
|
|
70
|
+
entryCount: number;
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
export interface SessionMigrationFilePlan {
|
|
74
|
+
/** Source file path (relative to transcripts root). */
|
|
75
|
+
sourceRelPath: string;
|
|
76
|
+
/** The date-stamped file name (e.g. `2026-06-29.jsonl`). */
|
|
77
|
+
fileName: string;
|
|
78
|
+
/** Distinct session groups discovered inside this source file. */
|
|
79
|
+
groups: SessionMigrationEntryGroup[];
|
|
80
|
+
/** Lines that could not be parsed as JSON or lacked a sessionKey. */
|
|
81
|
+
unmovableLines: number;
|
|
82
|
+
/**
|
|
83
|
+
* True when EVERY entry in the file already belongs in the source directory
|
|
84
|
+
* (nothing to do). Such files are skipped.
|
|
85
|
+
*/
|
|
86
|
+
alreadyHomed: boolean;
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
export interface SessionMigrationPlan {
|
|
90
|
+
generatedAt: string;
|
|
91
|
+
dryRun: boolean;
|
|
92
|
+
transcriptsDir: string;
|
|
93
|
+
/** Per-file plans that have at least one entry to move. */
|
|
94
|
+
files: SessionMigrationFilePlan[];
|
|
95
|
+
/** Distinct destination session keys across all files. */
|
|
96
|
+
distinctSessions: number;
|
|
97
|
+
/** Total entries that will be re-homed. */
|
|
98
|
+
movedEntries: number;
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
export interface SessionMigrationResult {
|
|
102
|
+
plan: SessionMigrationPlan;
|
|
103
|
+
applied: boolean;
|
|
104
|
+
filesRewritten: number;
|
|
105
|
+
filesRemoved: number;
|
|
106
|
+
errors: string[];
|
|
107
|
+
manifestPath?: string;
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
export interface MigrateSessionTranscriptsOptions {
|
|
111
|
+
/** The memory directory root (transcripts live under `<memoryDir>/transcripts`). */
|
|
112
|
+
memoryDir: string;
|
|
113
|
+
/** When true, perform the migration; otherwise produce a dry-run plan only. */
|
|
114
|
+
apply?: boolean;
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
const DATE_FILE_RE = /^\d{4}-\d{2}-\d{2}\.jsonl$/;
|
|
118
|
+
|
|
119
|
+
function isDateStampedJsonl(name: string): boolean {
|
|
120
|
+
return DATE_FILE_RE.test(name);
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
/**
|
|
124
|
+
* Candidate source directories that older builds may have used to conflate or
|
|
125
|
+
* misroute arbitrary sessions. We scan `transcripts/<type>/<id>/` two levels
|
|
126
|
+
* deep and consider any directory EXCEPT the first-class `session/<hash>` tree
|
|
127
|
+
* (whose contents are already homed). This covers:
|
|
128
|
+
*
|
|
129
|
+
* - the shared `other/default` fallback every arbitrary key once landed in;
|
|
130
|
+
* - the OLD `parts.length >= 3` parser's directories (e.g. `foo:bar:baz` →
|
|
131
|
+
* `baz/default`, `foo:bar:baz:qux` → `baz/qux`), which the pre-#1496 build
|
|
132
|
+
* wrote even for non-`agent:` keys (Thread B / codex review on PR #1504).
|
|
133
|
+
*
|
|
134
|
+
* `planFile` is the actual gate: a line is only moved when its session key now
|
|
135
|
+
* resolves to a DIFFERENT directory than the one it sits in. Legacy
|
|
136
|
+
* `agent:<id>:...` keys still resolve to their original channel directory, so
|
|
137
|
+
* scanning `main/default`, `discord/<chan>`, etc. is a safe no-op for them
|
|
138
|
+
* (rule #39 — identical routing across paths). This keeps the migration
|
|
139
|
+
* lossless and idempotent regardless of which directory data was stranded in.
|
|
140
|
+
*/
|
|
141
|
+
async function listFallbackSourceFiles(
|
|
142
|
+
transcriptsDir: string
|
|
143
|
+
): Promise<Array<{ relPath: string; fileName: string; absPath: string }>> {
|
|
144
|
+
const out: Array<{ relPath: string; fileName: string; absPath: string }> = [];
|
|
145
|
+
let typeEntries: Array<{ name: string; isDirectory(): boolean }>;
|
|
146
|
+
try {
|
|
147
|
+
typeEntries = await readdir(transcriptsDir, { withFileTypes: true });
|
|
148
|
+
} catch {
|
|
149
|
+
return out;
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
for (const typeEnt of typeEntries) {
|
|
153
|
+
if (!typeEnt.isDirectory()) continue;
|
|
154
|
+
|
|
155
|
+
const typeDir = path.join(transcriptsDir, typeEnt.name);
|
|
156
|
+
let idEntries: Array<{ name: string; isDirectory(): boolean }>;
|
|
157
|
+
try {
|
|
158
|
+
idEntries = await readdir(typeDir, { withFileTypes: true });
|
|
159
|
+
} catch {
|
|
160
|
+
continue;
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
for (const idEnt of idEntries) {
|
|
164
|
+
if (!idEnt.isDirectory()) continue;
|
|
165
|
+
|
|
166
|
+
// Only canonical `session/<hash>` dirs are already homed — skip those.
|
|
167
|
+
// A LEGACY `session/<name>` dir (e.g. `foo:bar:session:baz` →
|
|
168
|
+
// `session/baz` under the OLD parser) is a real migration source and
|
|
169
|
+
// must be scanned/split (codex review on PR #1504). `planFile`'s
|
|
170
|
+
// destination gate still guarantees losslessness/idempotence.
|
|
171
|
+
if (isCanonicalSessionDir(typeEnt.name, idEnt.name)) continue;
|
|
172
|
+
|
|
173
|
+
const chanDir = path.join(typeDir, idEnt.name);
|
|
174
|
+
let files: string[];
|
|
175
|
+
try {
|
|
176
|
+
files = (await readdir(chanDir)).filter((f) => f.endsWith(".jsonl")).sort();
|
|
177
|
+
} catch {
|
|
178
|
+
continue;
|
|
179
|
+
}
|
|
180
|
+
for (const fileName of files) {
|
|
181
|
+
out.push({
|
|
182
|
+
relPath: path.join(typeEnt.name, idEnt.name, fileName),
|
|
183
|
+
fileName,
|
|
184
|
+
absPath: path.join(chanDir, fileName),
|
|
185
|
+
});
|
|
186
|
+
}
|
|
187
|
+
}
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
return out.sort((a, b) => a.relPath.localeCompare(b.relPath));
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
interface ParsedSourceLine {
|
|
194
|
+
sessionKey?: string;
|
|
195
|
+
/** The raw JSONL text (without trailing newline). */
|
|
196
|
+
raw: string;
|
|
197
|
+
}
|
|
198
|
+
|
|
199
|
+
/**
|
|
200
|
+
* Parse a source JSONL file, preserving line order. Lines that are not valid
|
|
201
|
+
* JSON or lack a string `sessionKey` are tracked as unmovable and left in the
|
|
202
|
+
* source file so no data is lost (rule #18 — validate parse result type).
|
|
203
|
+
*/
|
|
204
|
+
function parseSourceLines(content: string): ParsedSourceLine[] {
|
|
205
|
+
const out: ParsedSourceLine[] = [];
|
|
206
|
+
for (const line of content.split("\n")) {
|
|
207
|
+
if (!line.trim()) continue;
|
|
208
|
+
let sessionKey: string | undefined;
|
|
209
|
+
try {
|
|
210
|
+
const parsed: unknown = JSON.parse(line);
|
|
211
|
+
if (parsed && typeof parsed === "object") {
|
|
212
|
+
const sk = (parsed as { sessionKey?: unknown }).sessionKey;
|
|
213
|
+
if (typeof sk === "string" && sk.length > 0) {
|
|
214
|
+
sessionKey = sk;
|
|
215
|
+
}
|
|
216
|
+
}
|
|
217
|
+
} catch {
|
|
218
|
+
// Unparseable line — keep it in place.
|
|
219
|
+
}
|
|
220
|
+
out.push({ sessionKey, raw: line });
|
|
221
|
+
}
|
|
222
|
+
return out;
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
/**
|
|
226
|
+
* Build a per-file migration plan. A line is "movable" when its session key
|
|
227
|
+
* resolves to a destination directory different from the source directory.
|
|
228
|
+
*/
|
|
229
|
+
function planFile(sourceRelPath: string, fileName: string, lines: ParsedSourceLine[]): SessionMigrationFilePlan {
|
|
230
|
+
const sourceDir = path.dirname(sourceRelPath);
|
|
231
|
+
const groups = new Map<string, { legacy: boolean; destDir: string; count: number }>();
|
|
232
|
+
let unmovableLines = 0;
|
|
233
|
+
let movableCount = 0;
|
|
234
|
+
|
|
235
|
+
for (const line of lines) {
|
|
236
|
+
if (!line.sessionKey) {
|
|
237
|
+
unmovableLines += 1;
|
|
238
|
+
continue;
|
|
239
|
+
}
|
|
240
|
+
const identity = parseSessionIdentity(line.sessionKey);
|
|
241
|
+
const paths = sessionStoragePaths(line.sessionKey);
|
|
242
|
+
// Only move when the destination directory differs from the source.
|
|
243
|
+
if (paths.dir === sourceDir) {
|
|
244
|
+
continue;
|
|
245
|
+
}
|
|
246
|
+
movableCount += 1;
|
|
247
|
+
const existing = groups.get(line.sessionKey);
|
|
248
|
+
if (existing) {
|
|
249
|
+
existing.count += 1;
|
|
250
|
+
} else {
|
|
251
|
+
groups.set(line.sessionKey, {
|
|
252
|
+
legacy: identity.legacy,
|
|
253
|
+
destDir: paths.dir,
|
|
254
|
+
count: 1,
|
|
255
|
+
});
|
|
256
|
+
}
|
|
257
|
+
}
|
|
258
|
+
|
|
259
|
+
const groupList: SessionMigrationEntryGroup[] = [...groups.entries()]
|
|
260
|
+
.map(([sessionKey, info]) => ({
|
|
261
|
+
sessionKey,
|
|
262
|
+
legacy: info.legacy,
|
|
263
|
+
destDir: info.destDir,
|
|
264
|
+
entryCount: info.count,
|
|
265
|
+
}))
|
|
266
|
+
.sort((a, b) => a.sessionKey.localeCompare(b.sessionKey));
|
|
267
|
+
|
|
268
|
+
return {
|
|
269
|
+
sourceRelPath,
|
|
270
|
+
fileName,
|
|
271
|
+
groups: groupList,
|
|
272
|
+
unmovableLines,
|
|
273
|
+
alreadyHomed: movableCount === 0,
|
|
274
|
+
};
|
|
275
|
+
}
|
|
276
|
+
|
|
277
|
+
/**
|
|
278
|
+
* Compute the migration plan without modifying anything.
|
|
279
|
+
*/
|
|
280
|
+
export async function planSessionTranscriptMigration(
|
|
281
|
+
options: MigrateSessionTranscriptsOptions
|
|
282
|
+
): Promise<SessionMigrationPlan> {
|
|
283
|
+
const transcriptsDir = path.join(options.memoryDir, "transcripts");
|
|
284
|
+
const sources = await listFallbackSourceFiles(transcriptsDir);
|
|
285
|
+
const files: SessionMigrationFilePlan[] = [];
|
|
286
|
+
const distinctSessions = new Set<string>();
|
|
287
|
+
let movedEntries = 0;
|
|
288
|
+
|
|
289
|
+
for (const source of sources) {
|
|
290
|
+
if (!isDateStampedJsonl(source.fileName)) continue;
|
|
291
|
+
let content: string;
|
|
292
|
+
try {
|
|
293
|
+
content = await readFile(source.absPath, "utf-8");
|
|
294
|
+
} catch {
|
|
295
|
+
continue;
|
|
296
|
+
}
|
|
297
|
+
const lines = parseSourceLines(content);
|
|
298
|
+
const filePlan = planFile(source.relPath, source.fileName, lines);
|
|
299
|
+
if (filePlan.alreadyHomed) continue;
|
|
300
|
+
|
|
301
|
+
for (const group of filePlan.groups) {
|
|
302
|
+
distinctSessions.add(group.sessionKey);
|
|
303
|
+
movedEntries += group.entryCount;
|
|
304
|
+
}
|
|
305
|
+
files.push(filePlan);
|
|
306
|
+
}
|
|
307
|
+
|
|
308
|
+
return {
|
|
309
|
+
generatedAt: new Date().toISOString(),
|
|
310
|
+
dryRun: options.apply !== true,
|
|
311
|
+
transcriptsDir,
|
|
312
|
+
files,
|
|
313
|
+
distinctSessions: distinctSessions.size,
|
|
314
|
+
movedEntries,
|
|
315
|
+
};
|
|
316
|
+
}
|
|
317
|
+
|
|
318
|
+
/**
|
|
319
|
+
* Append `lines` (raw JSONL strings) to `<destDir>/<fileName>` using a
|
|
320
|
+
* write-then-rename merge so a crash mid-migration cannot corrupt or truncate
|
|
321
|
+
* the destination (rule #54).
|
|
322
|
+
*
|
|
323
|
+
* Existing destination content is preserved and de-duplicated by raw line so
|
|
324
|
+
* re-running the migration is idempotent.
|
|
325
|
+
*/
|
|
326
|
+
async function mergeAppendLines(
|
|
327
|
+
transcriptsDir: string,
|
|
328
|
+
destDir: string,
|
|
329
|
+
fileName: string,
|
|
330
|
+
newLines: string[]
|
|
331
|
+
): Promise<void> {
|
|
332
|
+
const destChannelDir = await resolveSafeStoragePath(transcriptsDir, destDir);
|
|
333
|
+
await mkdir(destChannelDir, { recursive: true });
|
|
334
|
+
const destPath = await resolveSafeStoragePath(transcriptsDir, destDir, fileName);
|
|
335
|
+
|
|
336
|
+
const existing: string[] = [];
|
|
337
|
+
const seen = new Set<string>();
|
|
338
|
+
try {
|
|
339
|
+
const raw = await readFile(destPath, "utf-8");
|
|
340
|
+
for (const line of raw.split("\n")) {
|
|
341
|
+
if (!line.trim()) continue;
|
|
342
|
+
existing.push(line);
|
|
343
|
+
seen.add(line);
|
|
344
|
+
}
|
|
345
|
+
} catch {
|
|
346
|
+
// No existing destination file — fine.
|
|
347
|
+
}
|
|
348
|
+
|
|
349
|
+
const merged = [...existing];
|
|
350
|
+
for (const line of newLines) {
|
|
351
|
+
if (seen.has(line)) continue;
|
|
352
|
+
merged.push(line);
|
|
353
|
+
seen.add(line);
|
|
354
|
+
}
|
|
355
|
+
|
|
356
|
+
const body = merged.length > 0 ? `${merged.join("\n")}\n` : "";
|
|
357
|
+
const tmpPath = await resolveSafeStoragePath(
|
|
358
|
+
transcriptsDir,
|
|
359
|
+
destDir,
|
|
360
|
+
`${fileName}.migrate-${process.pid}-${Date.now()}.tmp`
|
|
361
|
+
);
|
|
362
|
+
await writeFile(tmpPath, body, "utf-8");
|
|
363
|
+
await rename(tmpPath, destPath);
|
|
364
|
+
}
|
|
365
|
+
|
|
366
|
+
/**
|
|
367
|
+
* Rewrite the source file to retain ONLY the lines that were not moved
|
|
368
|
+
* (unparseable lines and any that already belong in the source dir). Uses
|
|
369
|
+
* write-then-rename; deletes the source only when nothing remains.
|
|
370
|
+
*/
|
|
371
|
+
async function rewriteSourceRetainingUnmoved(
|
|
372
|
+
transcriptsDir: string,
|
|
373
|
+
sourceRelPath: string,
|
|
374
|
+
retainedLines: string[]
|
|
375
|
+
): Promise<{ removed: boolean }> {
|
|
376
|
+
const sourcePath = await resolveSafeStoragePath(transcriptsDir, sourceRelPath);
|
|
377
|
+
if (retainedLines.length === 0) {
|
|
378
|
+
// Atomic-rename into a sibling tombstone then unlink, so a crash leaves the
|
|
379
|
+
// original intact (never delete-before-confirm; rule #54). Simpler and
|
|
380
|
+
// equally safe: unlink only after destinations are confirmed written.
|
|
381
|
+
try {
|
|
382
|
+
await unlink(sourcePath);
|
|
383
|
+
} catch (err) {
|
|
384
|
+
const code =
|
|
385
|
+
err && typeof err === "object" && "code" in err ? String((err as { code?: unknown }).code ?? "") : "";
|
|
386
|
+
if (code !== "ENOENT") throw err;
|
|
387
|
+
}
|
|
388
|
+
return { removed: true };
|
|
389
|
+
}
|
|
390
|
+
|
|
391
|
+
const body = `${retainedLines.join("\n")}\n`;
|
|
392
|
+
const sourceDir = path.dirname(sourceRelPath);
|
|
393
|
+
const tmpPath = await resolveSafeStoragePath(
|
|
394
|
+
transcriptsDir,
|
|
395
|
+
sourceDir,
|
|
396
|
+
`${path.basename(sourceRelPath)}.migrate-${process.pid}-${Date.now()}.tmp`
|
|
397
|
+
);
|
|
398
|
+
await writeFile(tmpPath, body, "utf-8");
|
|
399
|
+
await rename(tmpPath, sourcePath);
|
|
400
|
+
return { removed: false };
|
|
401
|
+
}
|
|
402
|
+
|
|
403
|
+
/**
|
|
404
|
+
* Run the migration. Dry-run by default; pass `apply: true` to mutate.
|
|
405
|
+
*/
|
|
406
|
+
export async function migrateSessionTranscripts(
|
|
407
|
+
options: MigrateSessionTranscriptsOptions
|
|
408
|
+
): Promise<SessionMigrationResult> {
|
|
409
|
+
const plan = await planSessionTranscriptMigration(options);
|
|
410
|
+
const apply = options.apply === true;
|
|
411
|
+
|
|
412
|
+
if (!apply) {
|
|
413
|
+
return {
|
|
414
|
+
plan,
|
|
415
|
+
applied: false,
|
|
416
|
+
filesRewritten: 0,
|
|
417
|
+
filesRemoved: 0,
|
|
418
|
+
errors: [],
|
|
419
|
+
};
|
|
420
|
+
}
|
|
421
|
+
|
|
422
|
+
const transcriptsDir = plan.transcriptsDir;
|
|
423
|
+
const errors: string[] = [];
|
|
424
|
+
let filesRewritten = 0;
|
|
425
|
+
let filesRemoved = 0;
|
|
426
|
+
|
|
427
|
+
for (const filePlan of plan.files) {
|
|
428
|
+
try {
|
|
429
|
+
const sourcePath = await resolveSafeStoragePath(transcriptsDir, filePlan.sourceRelPath);
|
|
430
|
+
const content = await readFile(sourcePath, "utf-8");
|
|
431
|
+
const lines = parseSourceLines(content);
|
|
432
|
+
const sourceDir = path.dirname(filePlan.sourceRelPath);
|
|
433
|
+
|
|
434
|
+
// Bucket lines by destination, preserving order. Retain unmovable lines
|
|
435
|
+
// and any line already homed in the source dir.
|
|
436
|
+
const byDest = new Map<string, string[]>();
|
|
437
|
+
const retained: string[] = [];
|
|
438
|
+
for (const line of lines) {
|
|
439
|
+
if (!line.sessionKey) {
|
|
440
|
+
retained.push(line.raw);
|
|
441
|
+
continue;
|
|
442
|
+
}
|
|
443
|
+
const dest = sessionStoragePaths(line.sessionKey).dir;
|
|
444
|
+
if (dest === sourceDir) {
|
|
445
|
+
retained.push(line.raw);
|
|
446
|
+
continue;
|
|
447
|
+
}
|
|
448
|
+
const bucket = byDest.get(dest) ?? [];
|
|
449
|
+
bucket.push(line.raw);
|
|
450
|
+
byDest.set(dest, bucket);
|
|
451
|
+
}
|
|
452
|
+
|
|
453
|
+
// 1. Write destinations FIRST (write-then-rename, idempotent merge).
|
|
454
|
+
for (const [destDir, destLines] of byDest.entries()) {
|
|
455
|
+
await mergeAppendLines(transcriptsDir, destDir, filePlan.fileName, destLines);
|
|
456
|
+
}
|
|
457
|
+
|
|
458
|
+
// 2. Only after all destinations are confirmed, rewrite/remove source.
|
|
459
|
+
const { removed } = await rewriteSourceRetainingUnmoved(transcriptsDir, filePlan.sourceRelPath, retained);
|
|
460
|
+
if (removed) {
|
|
461
|
+
filesRemoved += 1;
|
|
462
|
+
} else {
|
|
463
|
+
filesRewritten += 1;
|
|
464
|
+
}
|
|
465
|
+
} catch (err) {
|
|
466
|
+
// The `errors` array is surfaced to operators via the CLI
|
|
467
|
+
// (`sessions migrate-transcripts --apply`) AND persisted in the audit
|
|
468
|
+
// manifest. Raw `err.message`/`String(err)` can leak filesystem paths or
|
|
469
|
+
// stack detail, so route operator-facing strings through the shared
|
|
470
|
+
// `displayErrorDetail()` sanitizer (cursor review on PR #1504, rule #51).
|
|
471
|
+
// The full error still goes to the operator-only debug log.
|
|
472
|
+
const detail = displayErrorDetail(err);
|
|
473
|
+
errors.push(
|
|
474
|
+
`Failed to migrate ${filePlan.sourceRelPath}${detail ? `: ${detail}` : ""}`,
|
|
475
|
+
);
|
|
476
|
+
log.error(`session transcript migration failed for ${filePlan.sourceRelPath}:`, err);
|
|
477
|
+
}
|
|
478
|
+
}
|
|
479
|
+
|
|
480
|
+
let manifestPath: string | undefined;
|
|
481
|
+
try {
|
|
482
|
+
manifestPath = await writeManifest(options.memoryDir, {
|
|
483
|
+
plan,
|
|
484
|
+
applied: true,
|
|
485
|
+
filesRewritten,
|
|
486
|
+
filesRemoved,
|
|
487
|
+
errors,
|
|
488
|
+
});
|
|
489
|
+
} catch (err) {
|
|
490
|
+
// Manifest is best-effort audit; do not fail the migration over it.
|
|
491
|
+
log.debug(`failed to write session migration manifest: ${err}`);
|
|
492
|
+
}
|
|
493
|
+
|
|
494
|
+
return {
|
|
495
|
+
plan,
|
|
496
|
+
applied: true,
|
|
497
|
+
filesRewritten,
|
|
498
|
+
filesRemoved,
|
|
499
|
+
errors,
|
|
500
|
+
manifestPath,
|
|
501
|
+
};
|
|
502
|
+
}
|
|
503
|
+
|
|
504
|
+
async function writeManifest(memoryDir: string, result: Omit<SessionMigrationResult, "manifestPath">): Promise<string> {
|
|
505
|
+
const auditDir = path.join(memoryDir, "state", "session-migration");
|
|
506
|
+
await mkdir(auditDir, { recursive: true });
|
|
507
|
+
const stamp = new Date().toISOString().replace(/[:.]/g, "-");
|
|
508
|
+
const manifestPath = path.join(auditDir, `migrate-transcripts-${stamp}.json`);
|
|
509
|
+
await writeFile(manifestPath, JSON.stringify(result, null, 2), "utf-8");
|
|
510
|
+
return manifestPath;
|
|
511
|
+
}
|
|
512
|
+
|
|
513
|
+
/** Best-effort byte-size summary for reporting (used by CLI output). */
|
|
514
|
+
export async function summarizeMigrationSources(memoryDir: string): Promise<{ files: number; bytes: number }> {
|
|
515
|
+
const transcriptsDir = path.join(memoryDir, "transcripts");
|
|
516
|
+
const sources = await listFallbackSourceFiles(transcriptsDir);
|
|
517
|
+
let bytes = 0;
|
|
518
|
+
let files = 0;
|
|
519
|
+
for (const source of sources) {
|
|
520
|
+
const info = await stat(source.absPath).catch(() => null);
|
|
521
|
+
if (info?.isFile()) {
|
|
522
|
+
bytes += info.size;
|
|
523
|
+
files += 1;
|
|
524
|
+
}
|
|
525
|
+
}
|
|
526
|
+
return { files, bytes };
|
|
527
|
+
}
|
package/src/summarizer.ts
CHANGED
|
@@ -11,11 +11,9 @@ import type { TranscriptManager } from "./transcript.js";
|
|
|
11
11
|
import { readSummarySnapshot, upsertSummarySnapshot, writeSummarySnapshot } from "./summary-snapshot.js";
|
|
12
12
|
import {
|
|
13
13
|
encodeStoragePathSegment,
|
|
14
|
-
encodeStoragePathSegmentWithHash,
|
|
15
|
-
isSafeLegacyPathSegment,
|
|
16
14
|
resolveSafeStoragePath,
|
|
17
|
-
storagePathHash,
|
|
18
15
|
} from "./storage-paths.js";
|
|
16
|
+
import { sessionStoragePaths } from "./session-identity.js";
|
|
19
17
|
|
|
20
18
|
// Schema for LLM summary output
|
|
21
19
|
const HourlySummarySchema = z.object({
|
|
@@ -763,45 +761,25 @@ Respond with valid JSON matching this schema:
|
|
|
763
761
|
startTime: Date,
|
|
764
762
|
endTime: Date
|
|
765
763
|
): Promise<TranscriptEntry[]> {
|
|
766
|
-
|
|
767
|
-
|
|
768
|
-
|
|
769
|
-
|
|
770
|
-
|
|
771
|
-
|
|
772
|
-
if (channelType === "main") {
|
|
773
|
-
channelId = "default";
|
|
774
|
-
} else if (channelType === "discord" && parts.length >= 5 && parts[3] === "channel") {
|
|
775
|
-
channelId = parts[4];
|
|
776
|
-
} else if (channelType === "slack" && parts.length >= 5 && parts[3] === "channel") {
|
|
777
|
-
channelId = parts[4];
|
|
778
|
-
} else if (channelType === "cron" && parts.length >= 4) {
|
|
779
|
-
channelId = parts[3];
|
|
780
|
-
} else if (parts.length >= 4) {
|
|
781
|
-
channelId = parts[3];
|
|
782
|
-
}
|
|
783
|
-
}
|
|
764
|
+
// Shared session-identity layer (issue #1496, rule #22). Arbitrary keys
|
|
765
|
+
// route to `session/<hash>`; legacy `agent:<id>:...` keep their paths. The
|
|
766
|
+
// shared helper also supplies the read-back-only candidate dirs (mirrors
|
|
767
|
+
// TranscriptManager) so legacy `other/default` AND old `parts.length >= 3`
|
|
768
|
+
// data stays discoverable here too.
|
|
769
|
+
const paths = sessionStoragePaths(sessionKey);
|
|
784
770
|
|
|
785
771
|
try {
|
|
786
772
|
const transcriptRoot = path.join(this.config.memoryDir, "transcripts");
|
|
787
|
-
const encodedDir = path.join(
|
|
788
|
-
encodeStoragePathSegment(channelType),
|
|
789
|
-
encodeStoragePathSegment(channelId),
|
|
790
|
-
);
|
|
791
|
-
const alternateDir = path.join(
|
|
792
|
-
encodeStoragePathSegmentWithHash(channelType),
|
|
793
|
-
`${encodeStoragePathSegmentWithHash(channelId)}--session-${storagePathHash(sessionKey)}`,
|
|
794
|
-
);
|
|
795
|
-
const legacyDir =
|
|
796
|
-
isSafeLegacyPathSegment(channelType) && isSafeLegacyPathSegment(channelId)
|
|
797
|
-
? path.join(channelType, channelId)
|
|
798
|
-
: undefined;
|
|
799
773
|
const candidateDirs = new Set(
|
|
800
|
-
[
|
|
774
|
+
[paths.dir, paths.alternateDir, paths.legacyDir, ...paths.readbackDirs].filter(
|
|
801
775
|
(dir): dir is string => typeof dir === "string" && dir.length > 0,
|
|
802
776
|
),
|
|
803
777
|
);
|
|
804
778
|
const entries: TranscriptEntry[] = [];
|
|
779
|
+
// Dedup identical raw rows that a partially-applied migration may have
|
|
780
|
+
// left in both the primary dir and a read-back dir (issue #1496, cursor
|
|
781
|
+
// review). Exact raw JSONL line is the stable identity.
|
|
782
|
+
const seenRawLines = new Set<string>();
|
|
805
783
|
|
|
806
784
|
// Read all daily transcript files in the directory
|
|
807
785
|
for (const candidateDir of candidateDirs) {
|
|
@@ -828,8 +806,10 @@ Respond with valid JSON matching this schema:
|
|
|
828
806
|
if (
|
|
829
807
|
entry.sessionKey === sessionKey &&
|
|
830
808
|
entryTime >= startTime.getTime() &&
|
|
831
|
-
entryTime < endTime.getTime()
|
|
809
|
+
entryTime < endTime.getTime() &&
|
|
810
|
+
!seenRawLines.has(line)
|
|
832
811
|
) {
|
|
812
|
+
seenRawLines.add(line);
|
|
833
813
|
entries.push(entry);
|
|
834
814
|
}
|
|
835
815
|
} catch {
|