mindforge-cc 11.9.1 → 11.9.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.agent/CLAUDE.md +37 -13
- package/.agent/hooks/mindforge-block-no-verify.js +61 -13
- package/.agent/hooks/mindforge-config-protection.js +82 -3
- package/.agent/hooks/mindforge-context-monitor.js +1 -1
- package/.agent/hooks/mindforge-workflow-guard.js +2 -2
- package/.agent/hooks/run-with-flags.js +190 -20
- package/.agent/mindforge/browse.md +2 -2
- package/.agent/mindforge/checkpoint.md +1 -1
- package/.agent/mindforge/consult.md +1 -1
- package/.agent/mindforge/cost-report.md +1 -1
- package/.agent/mindforge/harness-audit.md +1 -1
- package/.agent/mindforge/orch-add-feature.md +1 -1
- package/.agent/mindforge/orch-build-mvp.md +1 -1
- package/.agent/mindforge/orch-change-feature.md +1 -1
- package/.agent/mindforge/orch-fix-defect.md +1 -1
- package/.agent/mindforge/orch-refine-code.md +1 -1
- package/.agent/mindforge/qa.md +2 -2
- package/.claude/CLAUDE.md +37 -13
- package/.claude/commands/mindforge/browse.md +2 -2
- package/.claude/commands/mindforge/checkpoint.md +1 -1
- package/.claude/commands/mindforge/consult.md +1 -1
- package/.claude/commands/mindforge/cost-report.md +1 -1
- package/.claude/commands/mindforge/harness-audit.md +1 -1
- package/.claude/commands/mindforge/orch-add-feature.md +1 -1
- package/.claude/commands/mindforge/orch-build-mvp.md +1 -1
- package/.claude/commands/mindforge/orch-change-feature.md +1 -1
- package/.claude/commands/mindforge/orch-fix-defect.md +1 -1
- package/.claude/commands/mindforge/orch-refine-code.md +1 -1
- package/.claude/commands/mindforge/qa.md +2 -2
- package/.mindforge/MINDFORGE-SCHEMA.json +126 -13
- package/.mindforge/config.json +4 -4
- package/.mindforge/engine/autonomous/headless-adapter.md +2 -2
- package/.mindforge/engine/cost-tracking/router.md +1 -1
- package/.mindforge/engine/cost-tracking/token-ledger.md +21 -24
- package/.mindforge/engine/temporal-protocol.md +2 -2
- package/.mindforge/governance/change-classifier.md +20 -4
- package/.mindforge/memory/sync-manifest.json +1 -1
- package/.mindforge/metrics/METRICS-SCHEMA.md +13 -4
- package/.mindforge/personas/cost-optimizer.md +2 -2
- package/.mindforge/personas/multi-model-bridge.md +1 -1
- package/.mindforge/skills/agent-architecture-audit/SKILL.md +2 -2
- package/.mindforge/skills/cost-aware-routing/SKILL.md +3 -3
- package/.mindforge/skills/multi-llm-consult/SKILL.md +2 -2
- package/.mindforge/skills/orch-pipeline/SKILL.md +4 -4
- package/CHANGELOG.md +402 -0
- package/MINDFORGE.md +13 -6
- package/README.md +51 -2
- package/RELEASENOTES.md +55 -2
- package/SECURITY.md +22 -3
- package/bin/autonomous/audit-writer.js +48 -33
- package/bin/autonomous/auto-runner.js +65 -2
- package/bin/change-classifier.js +151 -16
- package/bin/dashboard/api-router.js +28 -47
- package/bin/dashboard/error-response.js +44 -0
- package/bin/dashboard/frontend/app.js +429 -0
- package/bin/dashboard/frontend/index.html +14 -390
- package/bin/dashboard/metrics-aggregator.js +75 -30
- package/bin/dashboard/revops-api.js +12 -2
- package/bin/dashboard/server.js +245 -6
- package/bin/dashboard/sse-bridge.js +11 -8
- package/bin/dashboard/temporal-api.js +11 -5
- package/bin/engine/remediation-engine.js +12 -1
- package/bin/engine/sre-manager.js +1 -1
- package/bin/engine/temporal-cli.js +56 -6
- package/bin/engine/temporal-hub.js +41 -9
- package/bin/engine/verification-runner.js +134 -17
- package/bin/engine/verify-cli.js +25 -7
- package/bin/eval/eval-harness.js +212 -1
- package/bin/eval/golden-set-retrieval.json +9 -0
- package/bin/governance/approval-record.js +147 -0
- package/bin/governance/approve.js +12 -7
- package/bin/governance/policy-engine.js +41 -3
- package/bin/governance/policy-gate-hardened.js +36 -1
- package/bin/governance/verify-approvals.js +163 -0
- package/bin/harness-audit.js +224 -10
- package/bin/hindsight-injector.js +8 -2
- package/bin/hooks/instinct-capture-hook.js +19 -5
- package/bin/install.js +63 -3
- package/bin/installer/harness-adapter-compliance.js +339 -28
- package/bin/installer/hook-registration.js +504 -0
- package/bin/installer-core.js +451 -63
- package/bin/learning/instinct-cli.js +14 -24
- package/bin/memory/knowledge-capture.js +23 -3
- package/bin/memory/knowledge-graph.js +70 -31
- package/bin/memory/vector-hub.js +500 -44
- package/bin/migrations/0.6.0-to-1.0.0.js +30 -25
- package/bin/migrations/1.0.0-to-2.0.0.js +22 -23
- package/bin/mindforge-cli.js +110 -17
- package/bin/models/cost-tracker.js +126 -29
- package/bin/models/model-client.js +6 -1
- package/bin/models/model-router.js +28 -7
- package/bin/models/usage-record.js +71 -0
- package/bin/revops/debt-monitor.js +57 -13
- package/bin/security/trust-gate-hook.js +50 -6
- package/bin/skill-validator.js +6 -1
- package/bin/skills-builder/skill-scorer.js +46 -6
- package/bin/updater/self-update.js +6 -1
- package/bin/updater/version-comparator.js +21 -1
- package/bin/utils/file-lock.js +106 -0
- package/bin/utils/mindforge-params.js +124 -0
- package/bin/utils/mindforge-version.js +99 -0
- package/bin/utils/redact-secrets.js +106 -0
- package/bin/validate-config.js +75 -17
- package/bin/wizard/setup-wizard.js +4 -1
- package/bin/wizard/theme.js +9 -1
- package/changelogs/index.json +11 -9
- package/changelogs/v11.9.2.md +209 -0
- package/changelogs/v11.9.3.md +195 -0
- package/docs/References/config-reference.md +76 -14
- package/docs/References/sdk-api.md +1 -1
- package/docs/Templates/Codebase/architecture.md +1 -1
- package/docs/commands-reference.md +4 -5
- package/docs/faq.md +25 -5
- package/docs/getting-started.md +3 -3
- package/docs/sdk-reference.md +15 -7
- package/docs/troubleshooting.md +10 -6
- package/docs/user-guide.md +14 -14
- package/examples/sdk-integration/README.md +1 -1
- package/package.json +10 -4
- package/subagents/.claude-plugin/marketplace.json +1 -1
- package/bin/dashboard/approval-handler.js +0 -136
package/bin/memory/vector-hub.js
CHANGED
|
@@ -8,6 +8,136 @@
|
|
|
8
8
|
const crypto = require('crypto');
|
|
9
9
|
const path = require('path');
|
|
10
10
|
const fs = require('fs');
|
|
11
|
+
const { withFileLock } = require('../utils/file-lock');
|
|
12
|
+
|
|
13
|
+
// ── FTS retrieval (FTS-01) ───────────────────────────────────────────────────
|
|
14
|
+
// A MATCH argument is an FTS *query expression*, not a literal. Wrapping the
|
|
15
|
+
// whole user query in double quotes made it ONE phrase, so a multi-word query
|
|
16
|
+
// only matched documents holding those exact ADJACENT words. Default behaviour
|
|
17
|
+
// is now search-box semantics: tokenise, drop FTS operator keywords, quote each
|
|
18
|
+
// term (which neutralises every FTS metacharacter) and OR the results together.
|
|
19
|
+
// Callers that genuinely need adjacency pass { phrase: true }.
|
|
20
|
+
//
|
|
21
|
+
// The OR alone is NOT the fix. FTS4 has no ranker, so the first `limit` rows
|
|
22
|
+
// are docid order over documents containing ANY term ("how", "the", "work") and
|
|
23
|
+
// measured mean recall@10 stays 0.0000. Ranking is what recovers it: tf-idf
|
|
24
|
+
// scored in JS from FTS4's matchinfo('pcnx') blob. Measured via
|
|
25
|
+
// `npm run eval:retrieval` on the repo doc corpus (517 docs, 10 golden
|
|
26
|
+
// queries): recall@10 0.0000 -> 0.6417, nDCG@10 0.0000 -> 0.5698.
|
|
27
|
+
const FTS_OPERATOR_WORDS = new Set(['and', 'or', 'not', 'near']);
|
|
28
|
+
const FTS_MAX_TERMS = 32;
|
|
29
|
+
const FTS_DEFAULT_LIMIT = 10;
|
|
30
|
+
const FTS_MAX_LIMIT = 100;
|
|
31
|
+
// Candidate rows pulled PER TERM before ranking. Bounds the work done for a very
|
|
32
|
+
// common term; see _rankedFtsSearch for why the pool is per term and not per
|
|
33
|
+
// query.
|
|
34
|
+
const FTS_RANK_POOL = 2000;
|
|
35
|
+
// BM25-style term-frequency saturation. matchinfo('pcnx') carries no document
|
|
36
|
+
// length, so a raw term count would systematically favour long documents;
|
|
37
|
+
// saturating tf at tf/(tf + k) keeps a repeated term ranked above a single
|
|
38
|
+
// occurrence without letting file size dominate the score.
|
|
39
|
+
const FTS_TF_SATURATION = 1.2;
|
|
40
|
+
// The only tables a ranked search may touch, with each FTS table's per-column
|
|
41
|
+
// relevance weights in DECLARED column order. _rankedFtsSearch interpolates
|
|
42
|
+
// these identifiers into SQL (SQLite cannot bind an identifier), so they are
|
|
43
|
+
// allowlisted here rather than trusted — no caller string can reach that SQL.
|
|
44
|
+
// A hit in the identifier column is a title match and is worth more than one in
|
|
45
|
+
// the body; traces_search declares `id` notindexed so it can never produce a hit
|
|
46
|
+
// at all, and weight 0 documents that rather than implying otherwise.
|
|
47
|
+
const FTS_SEARCH_TABLES = {
|
|
48
|
+
traces: { ftsTable: 'traces_search', columnWeights: [0, 1, 1, 1] }, // id, trace_id, content, agent
|
|
49
|
+
knowledge: { ftsTable: 'knowledge_search', columnWeights: [3, 1, 1] }, // id, content, tags
|
|
50
|
+
};
|
|
51
|
+
|
|
52
|
+
/**
|
|
53
|
+
* Tokenise a raw user query into individual FTS4 MATCH expressions.
|
|
54
|
+
* @param {string} rawQuery - MUST be a string; anything else is a caller bug
|
|
55
|
+
* @param {{phrase?: boolean}} [opts] - phrase:true yields a single
|
|
56
|
+
* exact-adjacency phrase expression instead of one expression per term
|
|
57
|
+
* @returns {string[]} MATCH expressions; empty when the query holds no term
|
|
58
|
+
* @throws {TypeError} when rawQuery is not a string
|
|
59
|
+
*/
|
|
60
|
+
function buildFtsTerms(rawQuery, opts = {}) {
|
|
61
|
+
// Reject a non-string LOUDLY. Stringifying would turn an accidentally-passed
|
|
62
|
+
// options object into "[object Object]" and silently search for that literal —
|
|
63
|
+
// a wrong answer that looks like a working search. bin/engine/
|
|
64
|
+
// remediation-engine.js did exactly that. Failing at the boundary is honest.
|
|
65
|
+
if (typeof rawQuery !== 'string') {
|
|
66
|
+
const got = rawQuery === null ? 'null' : typeof rawQuery;
|
|
67
|
+
throw new TypeError(`[VectorHub] search query must be a string, received ${got}`);
|
|
68
|
+
}
|
|
69
|
+
if (opts.phrase) {
|
|
70
|
+
// FTS query syntax has no escape for a double quote inside a phrase, so
|
|
71
|
+
// strip them rather than emit an unparseable expression.
|
|
72
|
+
const phrase = rawQuery.replace(/"/g, ' ').trim();
|
|
73
|
+
return phrase ? [`"${phrase}"`] : [];
|
|
74
|
+
}
|
|
75
|
+
const terms = [];
|
|
76
|
+
const seen = new Set();
|
|
77
|
+
// Split on every character that is not a letter, digit or underscore: the
|
|
78
|
+
// resulting tokens cannot contain an FTS metacharacter, so quoting is total.
|
|
79
|
+
for (const token of rawQuery.split(/[^\p{L}\p{N}_]+/u)) {
|
|
80
|
+
if (!token) continue;
|
|
81
|
+
const lower = token.toLowerCase();
|
|
82
|
+
if (FTS_OPERATOR_WORDS.has(lower) || seen.has(lower)) continue;
|
|
83
|
+
seen.add(lower);
|
|
84
|
+
terms.push(`"${token}"`);
|
|
85
|
+
if (terms.length >= FTS_MAX_TERMS) break;
|
|
86
|
+
}
|
|
87
|
+
return terms;
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
/**
|
|
91
|
+
* Clamp a caller-supplied row limit into a sane bounded integer.
|
|
92
|
+
* @param {*} limit
|
|
93
|
+
* @returns {number} an integer in [1, FTS_MAX_LIMIT]
|
|
94
|
+
*/
|
|
95
|
+
function clampFtsLimit(limit) {
|
|
96
|
+
const n = parseInt(limit, 10);
|
|
97
|
+
if (!Number.isFinite(n) || n < 1) return FTS_DEFAULT_LIMIT;
|
|
98
|
+
return Math.min(n, FTS_MAX_LIMIT);
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
/**
|
|
102
|
+
* Score one FTS4 matchinfo('pcnx') blob as a weighted tf-idf sum over every
|
|
103
|
+
* phrase/column pair.
|
|
104
|
+
*
|
|
105
|
+
* Blob layout (little-endian uint32): [p, c, n, then 3 values per (phrase,
|
|
106
|
+
* column) pair — hits in THIS row, hits in ALL rows, number of documents with at
|
|
107
|
+
* least one hit]. tf is saturated (see FTS_TF_SATURATION) because 'pcnx' carries
|
|
108
|
+
* no document length to normalise by; idf is the BM25 form clamped at 0, so a
|
|
109
|
+
* term present in nearly every document can never drive a score negative.
|
|
110
|
+
* @param {Uint8Array|Buffer|null} blob - value of matchinfo(<table>, 'pcnx')
|
|
111
|
+
* @param {number[]} [columnWeights] - per-column multipliers in declared column
|
|
112
|
+
* order; missing entries default to 1
|
|
113
|
+
* @returns {number} non-negative relevance score; 0 when the blob is unusable
|
|
114
|
+
*/
|
|
115
|
+
function scoreMatchInfo(blob, columnWeights) {
|
|
116
|
+
if (!blob || typeof blob.length !== 'number' || blob.length < 12) return 0;
|
|
117
|
+
const buf = Buffer.from(blob);
|
|
118
|
+
const u32 = (i) => buf.readUInt32LE(i * 4);
|
|
119
|
+
const phrases = u32(0);
|
|
120
|
+
const columns = u32(1);
|
|
121
|
+
const totalRows = u32(2);
|
|
122
|
+
if (!phrases || !columns) return 0;
|
|
123
|
+
let score = 0;
|
|
124
|
+
for (let p = 0; p < phrases; p++) {
|
|
125
|
+
for (let c = 0; c < columns; c++) {
|
|
126
|
+
const base = 3 + 3 * (p * columns + c);
|
|
127
|
+
// Truncated blob: return what was scored rather than read past the end.
|
|
128
|
+
if ((base + 3) * 4 > buf.length) return score;
|
|
129
|
+
const hitsThisRow = u32(base);
|
|
130
|
+
if (!hitsThisRow) continue;
|
|
131
|
+
const weight = (columnWeights && columnWeights[c] !== undefined) ? columnWeights[c] : 1;
|
|
132
|
+
if (weight === 0) continue;
|
|
133
|
+
const docsWithHits = u32(base + 2);
|
|
134
|
+
const tf = hitsThisRow / (hitsThisRow + FTS_TF_SATURATION);
|
|
135
|
+
const idf = Math.max(0, Math.log((totalRows - docsWithHits + 0.5) / (docsWithHits + 0.5)));
|
|
136
|
+
score += weight * tf * idf;
|
|
137
|
+
}
|
|
138
|
+
}
|
|
139
|
+
return score;
|
|
140
|
+
}
|
|
11
141
|
|
|
12
142
|
/**
|
|
13
143
|
* VectorHub — Unified Persistence Layer
|
|
@@ -38,6 +168,9 @@ class VectorHub {
|
|
|
38
168
|
// deliberately bias toward flushing.
|
|
39
169
|
this._pendingSaves = 0;
|
|
40
170
|
this._exitGuardInstalled = false;
|
|
171
|
+
// null until a file is loaded or written. null means "no expectation", so a first write into a
|
|
172
|
+
// fresh directory is never treated as a conflict.
|
|
173
|
+
this._diskFingerprint = null;
|
|
41
174
|
}
|
|
42
175
|
|
|
43
176
|
_installExitGuard() {
|
|
@@ -45,10 +178,59 @@ class VectorHub {
|
|
|
45
178
|
this._exitGuardInstalled = true;
|
|
46
179
|
// 'exit' handlers can only run synchronous code — saveSync() fits exactly.
|
|
47
180
|
process.once('exit', () => {
|
|
48
|
-
if (this._db
|
|
181
|
+
if (!this._db) return;
|
|
182
|
+
// _pendingSaves === 0 means every scheduled save already ran commitDb, which consumes
|
|
183
|
+
// its tmp file by renaming it. So there is nothing outstanding and the in-memory DB is
|
|
184
|
+
// durable — the reap below is safe without a further write.
|
|
185
|
+
const durable = this._pendingSaves > 0 ? this.saveSync() : true;
|
|
186
|
+
if (durable) this._reapAbandonedExport();
|
|
49
187
|
});
|
|
50
188
|
}
|
|
51
189
|
|
|
190
|
+
/**
|
|
191
|
+
* Delete THIS process's abandoned tmp export.
|
|
192
|
+
*
|
|
193
|
+
* THE LEAK. save() is a two-step chain: writeTmpDurable() writes and fsyncs
|
|
194
|
+
* `<db>.tmp.<pid>.async`, then commitDb() renames it into place. A 'exit' handler can only
|
|
195
|
+
* run synchronous code, so a process that exits with a save in flight abandons the pending
|
|
196
|
+
* .then() — the microtask never runs, commitDb never renames, and the tmp file stays on disk
|
|
197
|
+
* forever. Nothing anywhere deleted it. Measured in this repository: 176 orphaned
|
|
198
|
+
* `celestial.db.tmp.<pid>.async` files totalling 1.8 GB, against a live database of 10.6 MB.
|
|
199
|
+
* Each orphan is a complete copy of the database, so the directory grew by ~11 MB per exit.
|
|
200
|
+
*
|
|
201
|
+
* Reproduced with a control arm: three runs that call process.exit() without close() leave
|
|
202
|
+
* three orphans, one per pid; three runs that await close() leave zero. Orphan size records
|
|
203
|
+
* how far the chain got — 0 bytes if the process died inside fs.open, a full export if the
|
|
204
|
+
* write landed and only the rename was lost.
|
|
205
|
+
*
|
|
206
|
+
* The trigger is ordinary, not exotic. nexus-tracer.js is the framework-wide tracing
|
|
207
|
+
* singleton, so any command that traces and then exits hits this window;
|
|
208
|
+
* bin/migrations/v9-unified-memory.js calls process.exit() and never calls close() at all.
|
|
209
|
+
*
|
|
210
|
+
* WHY THIS IS SAFE, AND WHY IT IS GATED. An earlier snapshot from the SAME process is a
|
|
211
|
+
* subset of what saveSync() just exported: saveSync() serialises the entire current
|
|
212
|
+
* in-memory database, which already contains every row the abandoned export held. So once
|
|
213
|
+
* persistence is confirmed the orphan is provably redundant.
|
|
214
|
+
*
|
|
215
|
+
* It is confirmed, not assumed. When saveSync() fails — commitDb throws, or the conflict
|
|
216
|
+
* sidecar could not be written — the abandoned export may be the only copy of those rows on
|
|
217
|
+
* disk, and deleting it would be exactly the silent data destruction commitDb refuses to
|
|
218
|
+
* commit. In that case the caller does NOT reap, and the file is deliberately left behind for
|
|
219
|
+
* a human. Accumulating a file on a failed write is the correct trade against destroying the
|
|
220
|
+
* last copy of it. Measured: zero `.sync` orphans exist in this repository, so that path does
|
|
221
|
+
* not fire in practice — the accumulation was entirely the async one.
|
|
222
|
+
*
|
|
223
|
+
* Both suffixes are swept because saveSync()'s own tmp leaks by the same argument if
|
|
224
|
+
* commitDb throws after the write; that arm is currently unobserved but not impossible.
|
|
225
|
+
*/
|
|
226
|
+
_reapAbandonedExport() {
|
|
227
|
+
for (const suffix of ['async', 'sync']) {
|
|
228
|
+
try {
|
|
229
|
+
fs.unlinkSync(`${this.dbPath}.tmp.${process.pid}.${suffix}`);
|
|
230
|
+
} catch { /* not present — the normal case, since commitDb usually consumes it */ }
|
|
231
|
+
}
|
|
232
|
+
}
|
|
233
|
+
|
|
52
234
|
_ensureDir() {
|
|
53
235
|
const dir = path.dirname(this.dbPath);
|
|
54
236
|
if (!fs.existsSync(dir)) {
|
|
@@ -91,6 +273,9 @@ class VectorHub {
|
|
|
91
273
|
if (fs.existsSync(this.dbPath)) {
|
|
92
274
|
const buffer = fs.readFileSync(this.dbPath);
|
|
93
275
|
this._db = new SQL.Database(buffer);
|
|
276
|
+
// What the file looked like when this process took its copy. Every later write compares
|
|
277
|
+
// against it, so a rewrite by another process is detected instead of silently clobbered.
|
|
278
|
+
this._diskFingerprint = dbFingerprint(this.dbPath);
|
|
94
279
|
} else {
|
|
95
280
|
this._db = new SQL.Database();
|
|
96
281
|
}
|
|
@@ -203,10 +388,22 @@ class VectorHub {
|
|
|
203
388
|
|
|
204
389
|
// ── FTS4 Virtual Tables (FTS4 is available in all sql.js builds) ────────
|
|
205
390
|
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
391
|
+
// FTS-01: traces_search must be keyed on the traces PRIMARY KEY (id), not
|
|
392
|
+
// trace_id. With a trace_id key every new span in a trace DELETEd its
|
|
393
|
+
// siblings' index rows, so only the last span per trace stayed searchable
|
|
394
|
+
// (live DB: 5,082 content-bearing traces, 2,827 index rows, 2,255 = 44.4%
|
|
395
|
+
// unsearchable). CREATE VIRTUAL TABLE IF NOT EXISTS cannot add a column to a
|
|
396
|
+
// database created before this fix, so detect the old signature and rebuild.
|
|
397
|
+
// The index is 100% derivable from `traces`, so the rebuild is lossless.
|
|
398
|
+
const ftsMigration = this._ensureTracesSearchSchema();
|
|
399
|
+
// Never let the rebuild be silent: it DROPS and recreates the index table, and a
|
|
400
|
+
// parity shortfall means rows silently stopped being searchable. Both are reported.
|
|
401
|
+
if (ftsMigration && ftsMigration.migrated && ftsMigration.expected > 0) {
|
|
402
|
+
console.log(`[VectorHub] traces_search migrated to the id-keyed schema; rebuilt ${ftsMigration.indexed}/${ftsMigration.expected} rows`);
|
|
403
|
+
}
|
|
404
|
+
if (ftsMigration && ftsMigration.migrated && ftsMigration.indexed !== ftsMigration.expected) {
|
|
405
|
+
console.warn(`[VectorHub] traces_search parity BROKEN after rebuild: indexed=${ftsMigration.indexed} expected=${ftsMigration.expected}`);
|
|
406
|
+
}
|
|
210
407
|
|
|
211
408
|
this._db.run(`
|
|
212
409
|
CREATE VIRTUAL TABLE IF NOT EXISTS knowledge_search
|
|
@@ -228,6 +425,120 @@ class VectorHub {
|
|
|
228
425
|
console.log(`[VectorHub] Initialized WASM SQLite persistence at ${this.dbPath}`);
|
|
229
426
|
}
|
|
230
427
|
|
|
428
|
+
/**
|
|
429
|
+
* Create traces_search with the correct (id-keyed) schema, migrating and
|
|
430
|
+
* repopulating an older trace_id-keyed table in place when one exists.
|
|
431
|
+
* Idempotent — the sqlite_master signature check makes a re-run a no-op.
|
|
432
|
+
* `id` is declared notindexed: it is an opaque UUID, contributes no useful
|
|
433
|
+
* search term, and leaving it unindexed keeps MATCH semantics identical to the
|
|
434
|
+
* pre-fix table (content, agent and trace_id all remain searchable).
|
|
435
|
+
* @returns {{migrated: boolean, indexed: number|null, expected: number|null}}
|
|
436
|
+
*/
|
|
437
|
+
_ensureTracesSearchSchema() {
|
|
438
|
+
const existing = this.query(
|
|
439
|
+
'SELECT sql FROM sqlite_master WHERE type = ? AND name = ?',
|
|
440
|
+
['table', 'traces_search']
|
|
441
|
+
);
|
|
442
|
+
const currentSql = existing.length ? String(existing[0].sql || '') : null;
|
|
443
|
+
if (currentSql !== null && /fts4\s*\(\s*id\s*,/i.test(currentSql)) {
|
|
444
|
+
return { migrated: false, indexed: null, expected: null };
|
|
445
|
+
}
|
|
446
|
+
|
|
447
|
+
if (currentSql !== null) {
|
|
448
|
+
this._db.run('DROP TABLE traces_search');
|
|
449
|
+
}
|
|
450
|
+
this._db.run(`
|
|
451
|
+
CREATE VIRTUAL TABLE traces_search
|
|
452
|
+
USING fts4(id, trace_id, content, agent, notindexed=id, tokenize=porter)
|
|
453
|
+
`);
|
|
454
|
+
return { migrated: true, ...this.rebuildTracesSearch() };
|
|
455
|
+
}
|
|
456
|
+
|
|
457
|
+
/**
|
|
458
|
+
* Rebuild traces_search from the `traces` base table.
|
|
459
|
+
*
|
|
460
|
+
* Explicitly re-runnable and idempotent: every index row is derivable from
|
|
461
|
+
* `traces`, so this clears the index and re-inserts exactly one row per
|
|
462
|
+
* content-bearing trace. Running it twice yields the same counts. This is the
|
|
463
|
+
* backfill that recovers rows lost to the old trace_id-keyed DELETE; no schema
|
|
464
|
+
* migration is required to run it once the schema is id-keyed.
|
|
465
|
+
* @returns {{indexed: number, expected: number}} post-rebuild row counts —
|
|
466
|
+
* equal when the index is complete
|
|
467
|
+
*/
|
|
468
|
+
rebuildTracesSearch() {
|
|
469
|
+
this._db.run('DELETE FROM traces_search');
|
|
470
|
+
this._db.run(
|
|
471
|
+
`INSERT INTO traces_search (id, trace_id, content, agent)
|
|
472
|
+
SELECT id, trace_id, content, agent
|
|
473
|
+
FROM traces
|
|
474
|
+
WHERE content IS NOT NULL AND content <> ?`,
|
|
475
|
+
['']
|
|
476
|
+
);
|
|
477
|
+
const indexed = this.query('SELECT COUNT(*) AS c FROM traces_search')[0].c;
|
|
478
|
+
const expected = this.query(
|
|
479
|
+
'SELECT COUNT(*) AS c FROM traces WHERE content IS NOT NULL AND content <> ?',
|
|
480
|
+
['']
|
|
481
|
+
)[0].c;
|
|
482
|
+
return { indexed, expected };
|
|
483
|
+
}
|
|
484
|
+
|
|
485
|
+
/**
|
|
486
|
+
* Run an FTS MATCH per term, then return the top `limit` base-table rows ranked
|
|
487
|
+
* by summed tf-idf.
|
|
488
|
+
*
|
|
489
|
+
* One MATCH per term rather than a single OR-joined MATCH: an OR-join returns
|
|
490
|
+
* its rows in docid order, so LIMIT cuts the candidate pool at the OLDEST
|
|
491
|
+
* FTS_RANK_POOL matches. Measured on the 5,082-row live trace index, a query of
|
|
492
|
+
* "celestial" (2,888 matches) plus a unique token could not retrieve the
|
|
493
|
+
* uniquely-matching document AT ALL, because 2,000 older rows filled the pool
|
|
494
|
+
* first — and since traces are append-only, the newest rows are always the
|
|
495
|
+
* first to be dropped. Scored per term the arithmetic is identical (matchinfo
|
|
496
|
+
* sums over phrases either way), but the pool can only be exhausted by a term's
|
|
497
|
+
* OWN document frequency, so a rare, discriminating term never loses.
|
|
498
|
+
* @param {string} baseTable - key of FTS_SEARCH_TABLES
|
|
499
|
+
* @param {string[]} terms - MATCH expressions from buildFtsTerms()
|
|
500
|
+
* @param {number} limit - clamped row limit
|
|
501
|
+
* @returns {Array<Object>} base-table rows, most relevant first
|
|
502
|
+
*/
|
|
503
|
+
_rankedFtsSearch(baseTable, terms, limit) {
|
|
504
|
+
const config = FTS_SEARCH_TABLES[baseTable];
|
|
505
|
+
if (!config) {
|
|
506
|
+
throw new Error(`[VectorHub] refusing to search unknown table: ${baseTable}`);
|
|
507
|
+
}
|
|
508
|
+
if (terms.length === 0) return [];
|
|
509
|
+
|
|
510
|
+
// id -> { score, order }; `order` is the first-seen position, used as an
|
|
511
|
+
// explicit tiebreak so the ranking never depends on sort stability.
|
|
512
|
+
const scored = new Map();
|
|
513
|
+
for (const term of terms) {
|
|
514
|
+
const rows = this.query(
|
|
515
|
+
`SELECT id, matchinfo(${config.ftsTable}, 'pcnx') AS mi
|
|
516
|
+
FROM ${config.ftsTable}
|
|
517
|
+
WHERE ${config.ftsTable} MATCH ?
|
|
518
|
+
LIMIT ?`,
|
|
519
|
+
[term, FTS_RANK_POOL]
|
|
520
|
+
);
|
|
521
|
+
for (const row of rows) {
|
|
522
|
+
const add = scoreMatchInfo(row.mi, config.columnWeights);
|
|
523
|
+
const prev = scored.get(row.id);
|
|
524
|
+
scored.set(row.id, prev
|
|
525
|
+
? { score: prev.score + add, order: prev.order }
|
|
526
|
+
: { score: add, order: scored.size });
|
|
527
|
+
}
|
|
528
|
+
}
|
|
529
|
+
|
|
530
|
+
const ids = [...scored.entries()]
|
|
531
|
+
.sort((a, b) => (b[1].score - a[1].score) || (a[1].order - b[1].order))
|
|
532
|
+
.slice(0, limit)
|
|
533
|
+
.map(([id]) => id);
|
|
534
|
+
if (ids.length === 0) return [];
|
|
535
|
+
|
|
536
|
+
const placeholders = ids.map(() => '?').join(', ');
|
|
537
|
+
const rows = this.query(`SELECT * FROM ${baseTable} WHERE id IN (${placeholders})`, ids);
|
|
538
|
+
const byId = new Map(rows.map(r => [r.id, r]));
|
|
539
|
+
return ids.map(id => byId.get(id)).filter(Boolean);
|
|
540
|
+
}
|
|
541
|
+
|
|
231
542
|
/**
|
|
232
543
|
* Persist the in-memory database to disk (UC-09).
|
|
233
544
|
*
|
|
@@ -258,11 +569,24 @@ class VectorHub {
|
|
|
258
569
|
// COMPLETED (success or failure). The exit guard fires saveSync() while any
|
|
259
570
|
// scheduled save is still outstanding — see _installExitGuard().
|
|
260
571
|
this._pendingSaves++;
|
|
261
|
-
this._saveChain = this._saveChain
|
|
572
|
+
this._saveChain = this._saveChain
|
|
573
|
+
// Write the tmp asynchronously, then commit it under the lock. The commit MUST be the last
|
|
574
|
+
// step — see commitDb's note on the check-then-write gap that made this guard bypassable.
|
|
575
|
+
.then(() => writeTmpDurable(dbPath, buffer))
|
|
576
|
+
.then((tmpPath) => {
|
|
577
|
+
const res = commitDb(dbPath, this._diskFingerprint, buffer, tmpPath);
|
|
578
|
+
if (res.ok) {
|
|
579
|
+
this._diskFingerprint = res.fingerprint;
|
|
580
|
+
this._pendingSaves--; // decrement ONLY on a durable write; see below
|
|
581
|
+
}
|
|
582
|
+
})
|
|
262
583
|
.catch((err) => {
|
|
584
|
+
// Left deliberately outstanding. _pendingSaves gates the exit guard, so decrementing after a
|
|
585
|
+
// FAILED save would make the guard skip its last-resort saveSync() and drop this batch — the
|
|
586
|
+
// exact loss the counter exists to prevent. withFileLock throws when it cannot acquire, so
|
|
587
|
+
// lock contention lands here, and leaving the count high is what makes the retry happen.
|
|
263
588
|
console.warn(`[VectorHub] Failed to save database: ${err.message}`);
|
|
264
|
-
})
|
|
265
|
-
.then(() => { this._pendingSaves--; });
|
|
589
|
+
});
|
|
266
590
|
return this._saveChain;
|
|
267
591
|
}
|
|
268
592
|
|
|
@@ -270,13 +594,18 @@ class VectorHub {
|
|
|
270
594
|
* Synchronous, crash-safe persistence — used only on shutdown to GUARANTEE
|
|
271
595
|
* no acknowledged write is lost if the process exits before the async save
|
|
272
596
|
* chain drains. Correctness over non-blocking here.
|
|
597
|
+
*
|
|
598
|
+
* @returns {boolean} true when the in-memory database is durably on disk — either committed
|
|
599
|
+
* over the live file or preserved in a conflict sidecar. false when it is NOT, which is the
|
|
600
|
+
* signal _reapAbandonedExport() needs: on false, an abandoned tmp export may hold the only
|
|
601
|
+
* copy of those rows and must be left alone.
|
|
273
602
|
*/
|
|
274
603
|
saveSync() {
|
|
275
|
-
if (!this._db) return;
|
|
604
|
+
if (!this._db) return false;
|
|
276
605
|
try {
|
|
277
606
|
this._ensureDir();
|
|
278
607
|
const buffer = Buffer.from(this._db.export());
|
|
279
|
-
const tmpPath = `${this.dbPath}.tmp.${process.pid}`;
|
|
608
|
+
const tmpPath = `${this.dbPath}.tmp.${process.pid}.sync`;
|
|
280
609
|
const fd = fs.openSync(tmpPath, 'w');
|
|
281
610
|
try {
|
|
282
611
|
fs.writeSync(fd, buffer);
|
|
@@ -284,13 +613,26 @@ class VectorHub {
|
|
|
284
613
|
} finally {
|
|
285
614
|
fs.closeSync(fd);
|
|
286
615
|
}
|
|
287
|
-
|
|
616
|
+
// Commit under the lock. Worst case this blocks the exit handler for withFileLock's bounded
|
|
617
|
+
// retry ceiling (~1-2s) and then throws into the catch below — bounded, logged, and far better
|
|
618
|
+
// than writing over a file another process just changed.
|
|
619
|
+
const res = commitDb(this.dbPath, this._diskFingerprint, buffer, tmpPath);
|
|
620
|
+
if (!res.ok) {
|
|
621
|
+
this._pendingSaves = 0; // the data is in the sidecar; retrying would clobber again
|
|
622
|
+
// A written sidecar IS durable persistence — the bytes are on disk under a name nothing
|
|
623
|
+
// else will touch. A NULL sidecar is not: commitDb unlinked the tmp and wrote nothing,
|
|
624
|
+
// so this export exists only in memory and is about to be lost with the process.
|
|
625
|
+
return res.sidecar !== null && res.sidecar !== undefined;
|
|
626
|
+
}
|
|
627
|
+
this._diskFingerprint = res.fingerprint;
|
|
288
628
|
// A sync export captures the full in-memory DB — a superset of anything the
|
|
289
629
|
// outstanding async saves would have written — so the pending work is now
|
|
290
630
|
// durably satisfied. Clearing the counter prevents a redundant second flush.
|
|
291
631
|
this._pendingSaves = 0;
|
|
632
|
+
return true;
|
|
292
633
|
} catch (err) {
|
|
293
634
|
console.warn(`[VectorHub] Failed to save database (sync): ${err.message}`);
|
|
635
|
+
return false;
|
|
294
636
|
}
|
|
295
637
|
}
|
|
296
638
|
|
|
@@ -396,12 +738,14 @@ class VectorHub {
|
|
|
396
738
|
[entry.id, entry.trace_id, entry.span_id, entry.event, entry.timestamp, entry.agent, entry.content, entry.metadata, entry.drift_score, entry.mesh_node_id]
|
|
397
739
|
);
|
|
398
740
|
|
|
399
|
-
// Update FTS index if content exists
|
|
741
|
+
// Update the FTS index if content exists. Keyed on entry.id (the traces
|
|
742
|
+
// PRIMARY KEY) — keying the DELETE on trace_id wiped every sibling span's
|
|
743
|
+
// index row, which is what made 44.4% of trace content unsearchable (FTS-01).
|
|
400
744
|
if (entry.content) {
|
|
401
|
-
this._db.run('DELETE FROM traces_search WHERE
|
|
745
|
+
this._db.run('DELETE FROM traces_search WHERE id = ?', [entry.id]);
|
|
402
746
|
this._db.run(
|
|
403
|
-
'INSERT INTO traces_search (trace_id, content, agent) VALUES (?, ?, ?)',
|
|
404
|
-
[entry.trace_id, entry.content, entry.agent]
|
|
747
|
+
'INSERT INTO traces_search (id, trace_id, content, agent) VALUES (?, ?, ?, ?)',
|
|
748
|
+
[entry.id, entry.trace_id, entry.content, entry.agent]
|
|
405
749
|
);
|
|
406
750
|
}
|
|
407
751
|
|
|
@@ -436,26 +780,34 @@ class VectorHub {
|
|
|
436
780
|
}
|
|
437
781
|
|
|
438
782
|
/**
|
|
439
|
-
* Full-text search for traces.
|
|
783
|
+
* Full-text search for traces, ranked by tf-idf.
|
|
784
|
+
*
|
|
785
|
+
* Multi-word queries are ORed across terms and ranked (see _rankedFtsSearch).
|
|
786
|
+
* FTS4 has no built-in ranker, so without the ranking step the result is docid
|
|
787
|
+
* order and mean recall@10 measures 0.0000.
|
|
788
|
+
* @param {string} rawQuery - the user's query text
|
|
789
|
+
* @param {{phrase?: boolean, limit?: number}} [opts]
|
|
790
|
+
* phrase:true searches for the exact adjacent word sequence (old behaviour);
|
|
791
|
+
* limit is clamped to [1, 100] and defaults to 10.
|
|
792
|
+
* @returns {Promise<Array<Object>>} trace rows, most relevant first
|
|
793
|
+
* @throws {TypeError} when rawQuery is not a string
|
|
440
794
|
*/
|
|
441
|
-
async searchTraces(rawQuery) {
|
|
442
|
-
|
|
443
|
-
|
|
444
|
-
|
|
445
|
-
|
|
446
|
-
FROM traces t
|
|
447
|
-
JOIN traces_search ts ON t.trace_id = ts.trace_id
|
|
448
|
-
WHERE traces_search MATCH ?
|
|
449
|
-
LIMIT 10`,
|
|
450
|
-
[ftsQuery]
|
|
795
|
+
async searchTraces(rawQuery, opts = {}) {
|
|
796
|
+
return this._rankedFtsSearch(
|
|
797
|
+
'traces',
|
|
798
|
+
buildFtsTerms(rawQuery, opts),
|
|
799
|
+
clampFtsLimit(opts.limit)
|
|
451
800
|
);
|
|
452
801
|
}
|
|
453
802
|
|
|
454
803
|
/**
|
|
455
804
|
* Full-text search for traces (alias for backward compat).
|
|
805
|
+
* @param {string} rawQuery
|
|
806
|
+
* @param {{phrase?: boolean, limit?: number}} [opts]
|
|
807
|
+
* @returns {Promise<Array<Object>>}
|
|
456
808
|
*/
|
|
457
|
-
async searchFTS(rawQuery) {
|
|
458
|
-
return this.searchTraces(rawQuery);
|
|
809
|
+
async searchFTS(rawQuery, opts = {}) {
|
|
810
|
+
return this.searchTraces(rawQuery, opts);
|
|
459
811
|
}
|
|
460
812
|
|
|
461
813
|
/**
|
|
@@ -512,16 +864,22 @@ class VectorHub {
|
|
|
512
864
|
return record.id;
|
|
513
865
|
}
|
|
514
866
|
|
|
515
|
-
|
|
516
|
-
|
|
517
|
-
|
|
518
|
-
|
|
519
|
-
|
|
520
|
-
|
|
521
|
-
|
|
522
|
-
|
|
523
|
-
|
|
524
|
-
|
|
867
|
+
/**
|
|
868
|
+
* Full-text search for knowledge entries, ranked by tf-idf (see searchTraces).
|
|
869
|
+
* @param {string} rawQuery
|
|
870
|
+
* @param {number|{phrase?: boolean, limit?: number}} [limitOrOpts] - a bare
|
|
871
|
+
* number is still accepted for backward compatibility
|
|
872
|
+
* @returns {Promise<Array<Object>>} knowledge rows, most relevant first
|
|
873
|
+
* @throws {TypeError} when rawQuery is not a string
|
|
874
|
+
*/
|
|
875
|
+
async searchKnowledge(rawQuery, limitOrOpts = {}) {
|
|
876
|
+
const opts = (limitOrOpts !== null && typeof limitOrOpts === 'object')
|
|
877
|
+
? limitOrOpts
|
|
878
|
+
: { limit: limitOrOpts };
|
|
879
|
+
return this._rankedFtsSearch(
|
|
880
|
+
'knowledge',
|
|
881
|
+
buildFtsTerms(rawQuery, opts),
|
|
882
|
+
clampFtsLimit(opts.limit)
|
|
525
883
|
);
|
|
526
884
|
}
|
|
527
885
|
|
|
@@ -569,9 +927,108 @@ class VectorHub {
|
|
|
569
927
|
// ── Durable async DB file write (UC-09) ───────────────────────────────────────
|
|
570
928
|
// Crash-safe: write to a tmp file, fsync, then atomically rename over the target.
|
|
571
929
|
// A crash mid-write leaves the previous good .db intact (rename is atomic on POSIX).
|
|
572
|
-
|
|
930
|
+
/**
|
|
931
|
+
* A cheap identity for the on-disk database: size plus mtime in ms.
|
|
932
|
+
*
|
|
933
|
+
* Not a digest — this runs before every save, and hashing a 10MB+ file on each write would be a real
|
|
934
|
+
* cost for a check whose only job is "did somebody else touch this?". size+mtimeMs changes on any
|
|
935
|
+
* rewrite by another process, which is exactly the event being detected.
|
|
936
|
+
*/
|
|
937
|
+
function dbFingerprint(dbPath) {
|
|
938
|
+
try {
|
|
939
|
+
const st = fs.statSync(dbPath);
|
|
940
|
+
return `${st.size}:${st.mtimeMs}`;
|
|
941
|
+
} catch {
|
|
942
|
+
return null; // absent is a legitimate state, not an error
|
|
943
|
+
}
|
|
944
|
+
}
|
|
945
|
+
|
|
946
|
+
/**
|
|
947
|
+
* Refuse to silently overwrite another process's writes.
|
|
948
|
+
*
|
|
949
|
+
* sql.js holds the whole database in memory and persists by exporting the ENTIRE file and renaming it
|
|
950
|
+
* over the path. Two processes each hold their own copy and each rewrite the whole file, so the last
|
|
951
|
+
* writer wins and the other's rows are gone. Measured: two concurrent writers, 15 acknowledged
|
|
952
|
+
* recordTrace() calls each, 30 expected — 15 on disk, ALL from writer A. Writer B's 15 acknowledged
|
|
953
|
+
* writes vanished, with no error on either side.
|
|
954
|
+
*
|
|
955
|
+
* A caveat on the corroborating evidence, because it was overstated. `.mindforge/` does carry
|
|
956
|
+
* orphaned `celestial.db.tmp.<pid>` files, and one IS a valid database with skills rows absent from
|
|
957
|
+
* the live file — but those rows are not lost user data. Audited all 171 non-empty orphans against
|
|
958
|
+
* the live database: 161 are strict subsets, 9 exceed it only on `traces_search_segdir` (an FTS5
|
|
959
|
+
* segment-directory count, an index-merge artifact rather than rows), and exactly one
|
|
960
|
+
* (`celestial.db.tmp.4027`, 63.5 MB, pre-dating the `.async`/`.sync` suffixes) holds 1,373 skill
|
|
961
|
+
* names the live file lacks — every one of them a `Synthesized Skill (mf-*) - ev_<hash>` row, the
|
|
962
|
+
* generated filler already slated for deletion. So the orphans corroborate that the clobber window
|
|
963
|
+
* was ENTERED; they are not evidence that anything worth keeping was destroyed.
|
|
964
|
+
*
|
|
965
|
+
* The cross-process loss itself is proven by the two-writer measurement above, which does not
|
|
966
|
+
* depend on this. Recorded because "there is a database on disk holding rows the live file lacks"
|
|
967
|
+
* reads as recoverable data loss, and here it is not.
|
|
968
|
+
*
|
|
969
|
+
* Making sql.js genuinely multi-process safe means replacing the driver — the whole-file export IS the
|
|
970
|
+
* problem, and locking alone does not fix it, because both processes loaded a stale copy before either
|
|
971
|
+
* saved. What is fixable now is the SILENCE. On a detected clobber the export goes to a
|
|
972
|
+
* `.conflict.<pid>` sidecar and the caller is told, so the data still exists and somebody knows.
|
|
973
|
+
* Refusing loudly beats destroying quietly — the same choice as auto-runner refusing to record a
|
|
974
|
+
* completion it cannot substantiate.
|
|
975
|
+
*
|
|
976
|
+
* WHY CHECK AND RENAME ARE ONE CRITICAL SECTION. A first version checked the fingerprint and then
|
|
977
|
+
* handed the buffer to an ASYNC writer. Measured with a deterministic stage — parent writes 3, spawns a
|
|
978
|
+
* child that writes 5 and exits, parent then closes:
|
|
979
|
+
*
|
|
980
|
+
* [parent] GUARD ...898 == ...898 <- check passes
|
|
981
|
+
* === child writes and saves === <- file is now ...249
|
|
982
|
+
* [parent] ASYNC-WROTE new=...309 <- writes its PRE-CHILD snapshot anyway
|
|
983
|
+
* live rows: P0,P1,P2,seed <- the child's 5 rows destroyed, silently
|
|
984
|
+
*
|
|
985
|
+
* spawnSync blocked the event loop, so the child fit entirely inside the gap between the check and the
|
|
986
|
+
* write. A check followed by a later write is not a guard — it is the same clobber with extra steps.
|
|
987
|
+
* So the fingerprint comparison and the atomic rename now happen together, synchronously, under the
|
|
988
|
+
* shared fail-closed lock. withFileLock rejects a thenable outright, which makes reintroducing that gap
|
|
989
|
+
* a loud TypeError rather than silent data loss.
|
|
990
|
+
*
|
|
991
|
+
* The expensive part (writing and fsyncing the tmp file) stays OUTSIDE the lock and may still be async:
|
|
992
|
+
* the tmp path is pid-scoped, so no other process can observe or touch it.
|
|
993
|
+
*
|
|
994
|
+
* @param {string} tmpPath a fully-written, fsynced tmp file to rename into place
|
|
995
|
+
* @returns {{ok: boolean, fingerprint?: string, sidecar?: string}}
|
|
996
|
+
*/
|
|
997
|
+
function commitDb(dbPath, expected, buffer, tmpPath) {
|
|
998
|
+
return withFileLock(dbPath, () => {
|
|
999
|
+
const actual = dbFingerprint(dbPath);
|
|
1000
|
+
if (expected === null || actual === null || actual === expected) {
|
|
1001
|
+
fs.renameSync(tmpPath, dbPath);
|
|
1002
|
+
return { ok: true, fingerprint: dbFingerprint(dbPath) };
|
|
1003
|
+
}
|
|
1004
|
+
|
|
1005
|
+
// Conflict. Keep this process's bytes and leave the other process's file untouched.
|
|
1006
|
+
const sidecar = `${dbPath}.conflict.${process.pid}.${buffer.length}`;
|
|
1007
|
+
try {
|
|
1008
|
+
fs.renameSync(tmpPath, sidecar); // already fsynced; a rename beats a second full write
|
|
1009
|
+
} catch (err) {
|
|
1010
|
+
console.error(`[VectorHub] CONFLICT and the sidecar could not be written: ${err.message}`);
|
|
1011
|
+
try { fs.unlinkSync(tmpPath); } catch { /* nothing further to do */ }
|
|
1012
|
+
return { ok: false, sidecar: null };
|
|
1013
|
+
}
|
|
1014
|
+
console.error(
|
|
1015
|
+
`[VectorHub] REFUSING TO OVERWRITE ${dbPath}: another process changed it since this one loaded it `
|
|
1016
|
+
+ `(expected ${expected}, found ${actual}). sql.js rewrites the whole file, so continuing would `
|
|
1017
|
+
+ `discard that process's rows. This process's data went to ${sidecar} instead — nothing is lost, `
|
|
1018
|
+
+ 'but the two databases must be reconciled by hand.');
|
|
1019
|
+
return { ok: false, sidecar };
|
|
1020
|
+
}, { label: 'vector-hub-db' });
|
|
1021
|
+
}
|
|
1022
|
+
|
|
1023
|
+
/**
|
|
1024
|
+
* Write and fsync the pid-scoped tmp file, resolving to its path. Does NOT rename it into place —
|
|
1025
|
+
* that is commitDb's job, because the rename has to share a critical section with the staleness check.
|
|
1026
|
+
* Staying async here is the point: fsync of a multi-megabyte export is the expensive part, and the tmp
|
|
1027
|
+
* path cannot be observed by another process.
|
|
1028
|
+
*/
|
|
1029
|
+
function writeTmpDurable(dbPath, buffer) {
|
|
573
1030
|
return new Promise((resolve, reject) => {
|
|
574
|
-
const tmpPath = `${dbPath}.tmp.${process.pid}`;
|
|
1031
|
+
const tmpPath = `${dbPath}.tmp.${process.pid}.async`;
|
|
575
1032
|
const fail = (err) => { fs.unlink(tmpPath, () => reject(err)); };
|
|
576
1033
|
fs.open(tmpPath, 'w', (openErr, fd) => {
|
|
577
1034
|
if (openErr) return reject(openErr);
|
|
@@ -581,10 +1038,7 @@ function writeDbDurable(dbPath, buffer) {
|
|
|
581
1038
|
fs.close(fd, (closeErr) => {
|
|
582
1039
|
if (syncErr) return fail(syncErr);
|
|
583
1040
|
if (closeErr) return fail(closeErr);
|
|
584
|
-
|
|
585
|
-
if (renameErr) return fail(renameErr);
|
|
586
|
-
resolve();
|
|
587
|
-
});
|
|
1041
|
+
resolve(tmpPath);
|
|
588
1042
|
});
|
|
589
1043
|
});
|
|
590
1044
|
});
|
|
@@ -611,6 +1065,7 @@ const lazyHub = new Proxy({}, {
|
|
|
611
1065
|
get(_, prop) {
|
|
612
1066
|
if (prop === 'VectorHub') return VectorHub;
|
|
613
1067
|
if (prop === 'createVectorHub') return createVectorHub;
|
|
1068
|
+
if (prop === 'buildFtsTerms') return buildFtsTerms;
|
|
614
1069
|
if (!_instance) _instance = new VectorHub();
|
|
615
1070
|
return typeof _instance[prop] === 'function'
|
|
616
1071
|
? _instance[prop].bind(_instance)
|
|
@@ -621,3 +1076,4 @@ const lazyHub = new Proxy({}, {
|
|
|
621
1076
|
module.exports = lazyHub;
|
|
622
1077
|
module.exports.VectorHub = VectorHub;
|
|
623
1078
|
module.exports.createVectorHub = createVectorHub;
|
|
1079
|
+
module.exports.buildFtsTerms = buildFtsTerms;
|