peon-mem 1.0.6 → 1.0.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/compression.js +4 -3
- package/dist/config.d.ts +13 -0
- package/dist/config.js +26 -0
- package/dist/daemon.js +9 -3
- package/dist/embedding-store.d.ts +12 -0
- package/dist/embedding-store.js +128 -3
- package/dist/embeddings.d.ts +30 -0
- package/dist/embeddings.js +68 -5
- package/dist/entity-extraction.js +4 -3
- package/dist/global-extraction.js +4 -3
- package/dist/hyde.js +4 -3
- package/dist/logger.d.ts +20 -0
- package/dist/logger.js +73 -6
- package/dist/memory-store.d.ts +5 -1
- package/dist/memory-store.js +158 -12
- package/dist/overview.d.ts +18 -6
- package/dist/overview.js +92 -14
- package/dist/processor.d.ts +36 -0
- package/dist/processor.js +85 -2
- package/dist/quality.d.ts +17 -1
- package/dist/quality.js +152 -25
- package/dist/recuration.js +4 -3
- package/dist/reranker.js +4 -3
- package/dist/tools.d.ts +4 -0
- package/dist/tools.js +27 -1
- package/package.json +3 -3
- package/scripts/peon-operations-watch.mjs +80 -0
package/dist/logger.js
CHANGED
|
@@ -1,12 +1,32 @@
|
|
|
1
|
-
import { appendFile, mkdir,
|
|
1
|
+
import { appendFile, mkdir, open, rename, stat } from "node:fs/promises";
|
|
2
2
|
import { homedir } from "node:os";
|
|
3
|
-
import { join } from "node:path";
|
|
3
|
+
import { dirname, join } from "node:path";
|
|
4
4
|
const DEFAULT_LOG_DIR = join(homedir(), "Library", "Logs", "Peon");
|
|
5
|
+
/**
|
|
6
|
+
* Reads are bounded to the tail of the log. recent() used to slurp the entire
|
|
7
|
+
* file and split it, which meant a months-old 100 MB daemon.jsonl blocked the
|
|
8
|
+
* event loop on every monitor poll (flattening a 100 MB rope + allocating ~400k
|
|
9
|
+
* strings, only to throw all but `limit` of them away). Same tail-read strategy
|
|
10
|
+
* the query-embedding cache already uses.
|
|
11
|
+
*
|
|
12
|
+
* Sized to comfortably cover the largest real caller (token-stat seeding asks
|
|
13
|
+
* for 50k entries; entries average ~260 bytes), so bounding reads does not
|
|
14
|
+
* silently truncate what the daemon rebuilds on boot.
|
|
15
|
+
*/
|
|
16
|
+
const DEFAULT_TAIL_BYTES = 16 * 1024 * 1024;
|
|
17
|
+
/** Rotate before the live file can reach a size that hurts anything. */
|
|
18
|
+
const DEFAULT_MAX_BYTES = 32 * 1024 * 1024;
|
|
5
19
|
export class PeonLogger {
|
|
6
20
|
logFile;
|
|
21
|
+
maxBytes;
|
|
22
|
+
tailBytes;
|
|
7
23
|
writeQueue = Promise.resolve();
|
|
24
|
+
bytesWritten = 0;
|
|
25
|
+
sizeKnown = false;
|
|
8
26
|
constructor(options = {}) {
|
|
9
27
|
this.logFile = join(options.logDir ?? DEFAULT_LOG_DIR, "daemon.jsonl");
|
|
28
|
+
this.maxBytes = options.maxBytes ?? DEFAULT_MAX_BYTES;
|
|
29
|
+
this.tailBytes = options.tailBytes ?? DEFAULT_TAIL_BYTES;
|
|
10
30
|
}
|
|
11
31
|
async log(type, fields = {}) {
|
|
12
32
|
const entry = {
|
|
@@ -18,10 +38,15 @@ export class PeonLogger {
|
|
|
18
38
|
await this.enqueueWrite(`${JSON.stringify(entry)}\n`);
|
|
19
39
|
return entry;
|
|
20
40
|
}
|
|
41
|
+
/**
|
|
42
|
+
* Newest-first entries from the tail of the log. Cost is bounded by
|
|
43
|
+
* `tailBytes`, not by the size of the file, so this stays flat as the log
|
|
44
|
+
* grows. Entries older than the tail window are not visible here — the log
|
|
45
|
+
* file itself (and its rotated siblings) remain the full record.
|
|
46
|
+
*/
|
|
21
47
|
async recent(limit = 100) {
|
|
22
|
-
const
|
|
23
|
-
return
|
|
24
|
-
.trim()
|
|
48
|
+
const text = await this.readTail();
|
|
49
|
+
return text
|
|
25
50
|
.split("\n")
|
|
26
51
|
.filter(Boolean)
|
|
27
52
|
.slice(-limit)
|
|
@@ -31,15 +56,57 @@ export class PeonLogger {
|
|
|
31
56
|
return [JSON.parse(line)];
|
|
32
57
|
}
|
|
33
58
|
catch {
|
|
59
|
+
// Skips both genuinely corrupt lines and the partial first line left
|
|
60
|
+
// by seeking into the middle of a record.
|
|
34
61
|
return [];
|
|
35
62
|
}
|
|
36
63
|
});
|
|
37
64
|
}
|
|
65
|
+
async readTail() {
|
|
66
|
+
let handle;
|
|
67
|
+
try {
|
|
68
|
+
const size = (await stat(this.logFile)).size;
|
|
69
|
+
const length = Math.min(size, this.tailBytes);
|
|
70
|
+
if (length === 0)
|
|
71
|
+
return "";
|
|
72
|
+
handle = await open(this.logFile, "r");
|
|
73
|
+
const buffer = Buffer.alloc(length);
|
|
74
|
+
await handle.read(buffer, 0, length, size - length);
|
|
75
|
+
const text = buffer.toString("utf8");
|
|
76
|
+
// A partial leading line is unavoidable when the window starts mid-record.
|
|
77
|
+
return length < size ? text.slice(text.indexOf("\n") + 1) : text;
|
|
78
|
+
}
|
|
79
|
+
catch {
|
|
80
|
+
return "";
|
|
81
|
+
}
|
|
82
|
+
finally {
|
|
83
|
+
await handle?.close().catch(() => undefined);
|
|
84
|
+
}
|
|
85
|
+
}
|
|
86
|
+
/**
|
|
87
|
+
* Move the live log aside once it exceeds maxBytes. History is preserved in a
|
|
88
|
+
* timestamped sibling rather than truncated, so nothing is lost.
|
|
89
|
+
*/
|
|
90
|
+
async rotateIfNeeded() {
|
|
91
|
+
if (this.bytesWritten <= this.maxBytes)
|
|
92
|
+
return;
|
|
93
|
+
const stamp = new Date().toISOString().replace(/[:.]/g, "-");
|
|
94
|
+
await rename(this.logFile, `${this.logFile}.${stamp}`).catch(() => undefined);
|
|
95
|
+
this.bytesWritten = 0;
|
|
96
|
+
}
|
|
38
97
|
async enqueueWrite(line) {
|
|
39
98
|
const write = async () => {
|
|
40
99
|
try {
|
|
41
|
-
await mkdir(
|
|
100
|
+
await mkdir(dirname(this.logFile), { recursive: true });
|
|
101
|
+
// Adopt the on-disk size once, so a daemon restart doesn't forget that
|
|
102
|
+
// an already-huge log is due for rotation.
|
|
103
|
+
if (!this.sizeKnown) {
|
|
104
|
+
this.bytesWritten = await stat(this.logFile).then((s) => s.size).catch(() => 0);
|
|
105
|
+
this.sizeKnown = true;
|
|
106
|
+
}
|
|
107
|
+
await this.rotateIfNeeded();
|
|
42
108
|
await appendFile(this.logFile, line, "utf8");
|
|
109
|
+
this.bytesWritten += Buffer.byteLength(line, "utf8");
|
|
43
110
|
}
|
|
44
111
|
catch {
|
|
45
112
|
// Best-effort logging: a log write failure must never crash the daemon
|
package/dist/memory-store.d.ts
CHANGED
|
@@ -164,9 +164,13 @@ export declare class PeonMemoryStore {
|
|
|
164
164
|
* No-op when embeddings are unavailable. supersededBy links to a merged-away id
|
|
165
165
|
* are re-pointed at the surviving record so history stays intact.
|
|
166
166
|
*/
|
|
167
|
-
mergeSimilarActiveRecords(records: MemoryRecord[], threshold?: number
|
|
167
|
+
mergeSimilarActiveRecords(records: MemoryRecord[], threshold?: number, options?: {
|
|
168
|
+
maxActive?: number;
|
|
169
|
+
exhaustive?: boolean;
|
|
170
|
+
}): Promise<{
|
|
168
171
|
records: MemoryRecord[];
|
|
169
172
|
merged: number;
|
|
173
|
+
comparisons: number;
|
|
170
174
|
}>;
|
|
171
175
|
readProcessingState(): Promise<ProcessingState>;
|
|
172
176
|
writeProcessingState(state: ProcessingState): Promise<void>;
|
package/dist/memory-store.js
CHANGED
|
@@ -33,6 +33,78 @@ async function atomicWrite(path, content) {
|
|
|
33
33
|
await writeFile(tmp, content, "utf8");
|
|
34
34
|
await rename(tmp, path);
|
|
35
35
|
}
|
|
36
|
+
/**
|
|
37
|
+
* Ceiling on how many active records the O(n^2) semantic dedup pass will consider.
|
|
38
|
+
* Override with PEON_DEDUP_MAX_ACTIVE. Set generously enough that ordinary project
|
|
39
|
+
* brains still dedup, low enough that a very large brain can't stall the daemon.
|
|
40
|
+
*/
|
|
41
|
+
const DEDUP_MAX_ACTIVE = Number(process.env.PEON_DEDUP_MAX_ACTIVE) > 0
|
|
42
|
+
? Number(process.env.PEON_DEDUP_MAX_ACTIVE)
|
|
43
|
+
: 25_000;
|
|
44
|
+
/**
|
|
45
|
+
* Deterministic random projections for LSH bucketing of dedup candidates.
|
|
46
|
+
* Seeded so bucketing is reproducible across runs and tests.
|
|
47
|
+
*/
|
|
48
|
+
const DEDUP_BANDS = 8;
|
|
49
|
+
const DEDUP_BITS_PER_BAND = 6;
|
|
50
|
+
/** Yield to the event loop every N records so a long pass can't starve the daemon. */
|
|
51
|
+
const DEDUP_YIELD_EVERY = 200;
|
|
52
|
+
/**
|
|
53
|
+
* Hard ceiling on candidates examined per record. Bucketing alone only buys a
|
|
54
|
+
* constant factor when vectors are correlated (buckets fill unevenly), leaving the
|
|
55
|
+
* pass quadratic. Capping candidates makes the work O(n * k) — genuinely linear in
|
|
56
|
+
* brain size — at the cost of occasionally missing a merge in a very crowded bucket.
|
|
57
|
+
* A missed merge leaves a near-duplicate belief; an uncapped pass hangs the daemon.
|
|
58
|
+
*/
|
|
59
|
+
const DEDUP_MAX_CANDIDATES = 128;
|
|
60
|
+
const projectionCache = new Map();
|
|
61
|
+
function mulberry32(seed) {
|
|
62
|
+
let a = seed >>> 0;
|
|
63
|
+
return () => {
|
|
64
|
+
a = (a + 0x6d2b79f5) >>> 0;
|
|
65
|
+
let t = Math.imul(a ^ (a >>> 15), 1 | a);
|
|
66
|
+
t = (t + Math.imul(t ^ (t >>> 7), 61 | t)) ^ t;
|
|
67
|
+
return ((t ^ (t >>> 14)) >>> 0) / 4294967296;
|
|
68
|
+
};
|
|
69
|
+
}
|
|
70
|
+
function projectionsFor(dim) {
|
|
71
|
+
const cached = projectionCache.get(dim);
|
|
72
|
+
if (cached)
|
|
73
|
+
return cached;
|
|
74
|
+
const rand = mulberry32(0x9e3779b9 ^ dim);
|
|
75
|
+
const planes = [];
|
|
76
|
+
for (let p = 0; p < DEDUP_BANDS * DEDUP_BITS_PER_BAND; p += 1) {
|
|
77
|
+
const plane = new Float32Array(dim);
|
|
78
|
+
for (let d = 0; d < dim; d += 1)
|
|
79
|
+
plane[d] = rand() * 2 - 1;
|
|
80
|
+
planes.push(plane);
|
|
81
|
+
}
|
|
82
|
+
projectionCache.set(dim, planes);
|
|
83
|
+
return planes;
|
|
84
|
+
}
|
|
85
|
+
/**
|
|
86
|
+
* Band keys for a vector: sign bits of random-hyperplane projections, grouped into
|
|
87
|
+
* bands. Two vectors with high cosine similarity agree on most bits, so they collide
|
|
88
|
+
* in at least one band with high probability — letting dedup compare a handful of
|
|
89
|
+
* plausible candidates instead of every record seen so far.
|
|
90
|
+
*/
|
|
91
|
+
function bandKeys(vector) {
|
|
92
|
+
const dim = vector.length;
|
|
93
|
+
const planes = projectionsFor(dim);
|
|
94
|
+
const keys = [];
|
|
95
|
+
for (let band = 0; band < DEDUP_BANDS; band += 1) {
|
|
96
|
+
let bits = "";
|
|
97
|
+
for (let b = 0; b < DEDUP_BITS_PER_BAND; b += 1) {
|
|
98
|
+
const plane = planes[band * DEDUP_BITS_PER_BAND + b];
|
|
99
|
+
let dot = 0;
|
|
100
|
+
for (let d = 0; d < dim; d += 1)
|
|
101
|
+
dot += vector[d] * plane[d];
|
|
102
|
+
bits += dot >= 0 ? "1" : "0";
|
|
103
|
+
}
|
|
104
|
+
keys.push(`${band}:${bits}`);
|
|
105
|
+
}
|
|
106
|
+
return keys;
|
|
107
|
+
}
|
|
36
108
|
export class PeonMemoryStore {
|
|
37
109
|
projectPath;
|
|
38
110
|
memoryDir;
|
|
@@ -612,18 +684,29 @@ export class PeonMemoryStore {
|
|
|
612
684
|
* No-op when embeddings are unavailable. supersededBy links to a merged-away id
|
|
613
685
|
* are re-pointed at the surviving record so history stays intact.
|
|
614
686
|
*/
|
|
615
|
-
async mergeSimilarActiveRecords(records, threshold = 0.9) {
|
|
687
|
+
async mergeSimilarActiveRecords(records, threshold = 0.9, options = {}) {
|
|
616
688
|
if (!this.embeddingClient || !this.embeddingStore)
|
|
617
|
-
return { records, merged: 0 };
|
|
689
|
+
return { records, merged: 0, comparisons: 0 };
|
|
690
|
+
// SCALE GUARD. The pass below is O(n^2) pairwise cosine over 1536-dim vectors.
|
|
691
|
+
// On a 30k-record brain (~6.4k active) that is ~20.5M comparisons / ~31.6B float
|
|
692
|
+
// ops on the main thread, plus every vector resident as float64 — measured at
|
|
693
|
+
// 99% CPU and >1.9 GB RSS, which wedged the daemon's event loop entirely.
|
|
694
|
+
// Above the threshold we skip dedup rather than take the daemon down: a brain
|
|
695
|
+
// that keeps a few near-duplicates is strictly better than a brain that hangs.
|
|
696
|
+
// Checked BEFORE sync() so the vector sidecar is never even loaded.
|
|
697
|
+
const activeCount = records.reduce((n, r) => (r.status === "active" ? n + 1 : n), 0);
|
|
698
|
+
if (activeCount > (options.maxActive ?? DEDUP_MAX_ACTIVE)) {
|
|
699
|
+
return { records, merged: 0, comparisons: 0 };
|
|
700
|
+
}
|
|
618
701
|
let vectorById;
|
|
619
702
|
try {
|
|
620
703
|
vectorById = (await this.embeddingStore.sync(records, this.embeddingClient)).vectorById;
|
|
621
704
|
}
|
|
622
705
|
catch {
|
|
623
|
-
return { records, merged: 0 };
|
|
706
|
+
return { records, merged: 0, comparisons: 0 };
|
|
624
707
|
}
|
|
625
708
|
if (vectorById.size === 0)
|
|
626
|
-
return { records, merged: 0 };
|
|
709
|
+
return { records, merged: 0, comparisons: 0 };
|
|
627
710
|
const active = records.filter((record) => record.status === "active");
|
|
628
711
|
const passthrough = records.filter((record) => record.status !== "active");
|
|
629
712
|
const kept = [];
|
|
@@ -631,22 +714,81 @@ export class PeonMemoryStore {
|
|
|
631
714
|
const remap = new Map();
|
|
632
715
|
const mergeNow = new Date().toISOString();
|
|
633
716
|
let merged = 0;
|
|
717
|
+
// Candidate index: band key -> indices into kept[]. Lets each record compare
|
|
718
|
+
// against a few plausible near-duplicates instead of every record so far,
|
|
719
|
+
// turning the old O(n^2) scan into roughly linear work.
|
|
720
|
+
const buckets = new Map();
|
|
721
|
+
const indexKept = (index, vector, type) => {
|
|
722
|
+
if (!vector)
|
|
723
|
+
return;
|
|
724
|
+
for (const key of bandKeys(vector)) {
|
|
725
|
+
const full = `${type}|${key}`;
|
|
726
|
+
const list = buckets.get(full);
|
|
727
|
+
if (list)
|
|
728
|
+
list.push(index);
|
|
729
|
+
else
|
|
730
|
+
buckets.set(full, [index]);
|
|
731
|
+
}
|
|
732
|
+
};
|
|
733
|
+
let comparisons = 0;
|
|
734
|
+
let processed = 0;
|
|
634
735
|
for (const record of active) {
|
|
736
|
+
// Long passes must never starve the daemon's event loop the way the old
|
|
737
|
+
// fully synchronous scan did.
|
|
738
|
+
processed += 1;
|
|
739
|
+
if (processed % DEDUP_YIELD_EVERY === 0)
|
|
740
|
+
await new Promise((resolve) => setImmediate(resolve));
|
|
635
741
|
const vec = vectorById.get(record.id);
|
|
636
742
|
let matchIndex = -1;
|
|
637
743
|
if (vec) {
|
|
638
|
-
|
|
639
|
-
|
|
640
|
-
|
|
641
|
-
|
|
642
|
-
|
|
643
|
-
|
|
644
|
-
|
|
744
|
+
if (options.exhaustive) {
|
|
745
|
+
for (let i = 0; i < kept.length; i += 1) {
|
|
746
|
+
if (kept[i].type !== record.type)
|
|
747
|
+
continue;
|
|
748
|
+
const other = vectorById.get(kept[i].id);
|
|
749
|
+
if (!other)
|
|
750
|
+
continue;
|
|
751
|
+
comparisons += 1;
|
|
752
|
+
if (cosineSimilarity(vec, other) >= threshold) {
|
|
753
|
+
matchIndex = i;
|
|
754
|
+
break;
|
|
755
|
+
}
|
|
756
|
+
}
|
|
757
|
+
}
|
|
758
|
+
else {
|
|
759
|
+
const seen = new Set();
|
|
760
|
+
let examined = 0;
|
|
761
|
+
for (const key of bandKeys(vec)) {
|
|
762
|
+
const candidates = buckets.get(`${record.type}|${key}`);
|
|
763
|
+
if (!candidates)
|
|
764
|
+
continue;
|
|
765
|
+
// Most recent entries first: a near-duplicate is likeliest among
|
|
766
|
+
// recently-seen beliefs, so a capped scan still finds the common case.
|
|
767
|
+
for (let c = candidates.length - 1; c >= 0; c -= 1) {
|
|
768
|
+
if (examined >= DEDUP_MAX_CANDIDATES)
|
|
769
|
+
break;
|
|
770
|
+
const i = candidates[c];
|
|
771
|
+
if (seen.has(i))
|
|
772
|
+
continue;
|
|
773
|
+
seen.add(i);
|
|
774
|
+
const other = vectorById.get(kept[i].id);
|
|
775
|
+
if (!other)
|
|
776
|
+
continue;
|
|
777
|
+
examined += 1;
|
|
778
|
+
comparisons += 1;
|
|
779
|
+
if (cosineSimilarity(vec, other) >= threshold) {
|
|
780
|
+
matchIndex = i;
|
|
781
|
+
break;
|
|
782
|
+
}
|
|
783
|
+
}
|
|
784
|
+
if (matchIndex !== -1 || examined >= DEDUP_MAX_CANDIDATES)
|
|
785
|
+
break;
|
|
645
786
|
}
|
|
646
787
|
}
|
|
647
788
|
}
|
|
648
789
|
if (matchIndex === -1) {
|
|
649
790
|
kept.push(record);
|
|
791
|
+
indexKept(kept.length - 1, vec, record.type);
|
|
650
792
|
continue;
|
|
651
793
|
}
|
|
652
794
|
const other = kept[matchIndex];
|
|
@@ -661,6 +803,10 @@ export class PeonMemoryStore {
|
|
|
661
803
|
entities: unique([...record.entities, ...other.entities]),
|
|
662
804
|
updatedAt: record.updatedAt > other.updatedAt ? record.updatedAt : other.updatedAt
|
|
663
805
|
};
|
|
806
|
+
// The survivor can be the incoming record, so make its vector findable at
|
|
807
|
+
// that slot too — otherwise later near-duplicates could miss the bucket.
|
|
808
|
+
if (canonical.id === record.id)
|
|
809
|
+
indexKept(matchIndex, vec, record.type);
|
|
664
810
|
remap.set(loser.id, canonical.id);
|
|
665
811
|
// Recoverable-loser rule: don't destroy the merged-away belief — retire it as superseded,
|
|
666
812
|
// linked to the survivor. It leaves active recall but its content stays recoverable and
|
|
@@ -682,7 +828,7 @@ export class PeonMemoryStore {
|
|
|
682
828
|
const fixed = [...passthrough, ...retired].map((record) => record.supersededBy && remap.has(record.supersededBy)
|
|
683
829
|
? { ...record, supersededBy: resolveRemap(record.supersededBy) }
|
|
684
830
|
: record);
|
|
685
|
-
return { records: [...kept, ...fixed], merged };
|
|
831
|
+
return { records: [...kept, ...fixed], merged, comparisons };
|
|
686
832
|
}
|
|
687
833
|
async readProcessingState() {
|
|
688
834
|
const raw = await readFile(join(this.memoryDir, "brain", "processing-state.json"), "utf8").catch(() => "");
|
package/dist/overview.d.ts
CHANGED
|
@@ -34,14 +34,26 @@ export interface DuplicatePair {
|
|
|
34
34
|
bContent: string;
|
|
35
35
|
similarity: number;
|
|
36
36
|
}
|
|
37
|
-
/**
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
export
|
|
37
|
+
/** How many record pairs the last duplicate scan actually compared. */
|
|
38
|
+
export declare function duplicateScanStats(): {
|
|
39
|
+
pairsEvaluated: number;
|
|
40
|
+
};
|
|
41
|
+
export interface DetectDuplicatesOptions {
|
|
42
42
|
threshold?: number;
|
|
43
43
|
limit?: number;
|
|
44
|
-
|
|
44
|
+
/** Compare every pair (the original scan). Kept for equivalence testing. */
|
|
45
|
+
exhaustive?: boolean;
|
|
46
|
+
maxTokenBucket?: number;
|
|
47
|
+
}
|
|
48
|
+
/**
|
|
49
|
+
* Near-duplicate belief pairs, strongest first.
|
|
50
|
+
*
|
|
51
|
+
* Previously this compared every active pair — ~20.4M jaccard computations on a
|
|
52
|
+
* 6.4k-active brain, to return the top 5 — on both the /overview endpoint and the
|
|
53
|
+
* consolidation auto-merge path. Candidates are now found through an inverted index
|
|
54
|
+
* on rare tokens, so the overwhelming majority of pairs are never scored.
|
|
55
|
+
*/
|
|
56
|
+
export declare function detectDuplicates(records: readonly MemoryRecord[], options?: DetectDuplicatesOptions): DuplicatePair[];
|
|
45
57
|
export interface TokenSavings {
|
|
46
58
|
onAvg: number;
|
|
47
59
|
offAvg: number;
|
package/dist/overview.js
CHANGED
|
@@ -50,29 +50,107 @@ function jaccard(a, b) {
|
|
|
50
50
|
* Flag near-duplicate ACTIVE beliefs of the same type — a nudge for the user to
|
|
51
51
|
* merge, not an automatic action. Conservative threshold to avoid false alarms.
|
|
52
52
|
*/
|
|
53
|
+
/**
|
|
54
|
+
* Only the rarest few tokens of a record are used as blocking keys: a pair at
|
|
55
|
+
* jaccard >= 0.6 shares most of its words, so it almost certainly shares one of
|
|
56
|
+
* them. Tokens common to a huge share of the brain are useless as keys and are
|
|
57
|
+
* skipped rather than producing a quadratic bucket.
|
|
58
|
+
*/
|
|
59
|
+
const DUPLICATE_BLOCKING_TOKENS = 12;
|
|
60
|
+
const MAX_TOKEN_BUCKET = Number(process.env.PEON_DUPLICATE_MAX_TOKEN_BUCKET) > 0
|
|
61
|
+
? Number(process.env.PEON_DUPLICATE_MAX_TOKEN_BUCKET)
|
|
62
|
+
: 400;
|
|
63
|
+
let lastDuplicatePairsEvaluated = 0;
|
|
64
|
+
/** How many record pairs the last duplicate scan actually compared. */
|
|
65
|
+
export function duplicateScanStats() {
|
|
66
|
+
return { pairsEvaluated: lastDuplicatePairsEvaluated };
|
|
67
|
+
}
|
|
68
|
+
/**
|
|
69
|
+
* Near-duplicate belief pairs, strongest first.
|
|
70
|
+
*
|
|
71
|
+
* Previously this compared every active pair — ~20.4M jaccard computations on a
|
|
72
|
+
* 6.4k-active brain, to return the top 5 — on both the /overview endpoint and the
|
|
73
|
+
* consolidation auto-merge path. Candidates are now found through an inverted index
|
|
74
|
+
* on rare tokens, so the overwhelming majority of pairs are never scored.
|
|
75
|
+
*/
|
|
53
76
|
export function detectDuplicates(records, options = {}) {
|
|
54
77
|
const threshold = options.threshold ?? 0.6;
|
|
55
78
|
const limit = options.limit ?? 5;
|
|
56
79
|
const active = records.filter((record) => record.status === "active");
|
|
57
80
|
const sets = active.map((record) => wordSet(record.content));
|
|
58
|
-
const
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
81
|
+
const found = [];
|
|
82
|
+
let pairsEvaluated = 0;
|
|
83
|
+
const score = (i, j) => {
|
|
84
|
+
if (active[i].type !== active[j].type)
|
|
85
|
+
return;
|
|
86
|
+
pairsEvaluated += 1;
|
|
87
|
+
const similarity = jaccard(sets[i], sets[j]);
|
|
88
|
+
if (similarity < threshold)
|
|
89
|
+
return;
|
|
90
|
+
found.push({
|
|
91
|
+
i,
|
|
92
|
+
j,
|
|
93
|
+
pair: {
|
|
94
|
+
aId: active[i].id,
|
|
95
|
+
aContent: active[i].content,
|
|
96
|
+
bId: active[j].id,
|
|
97
|
+
bContent: active[j].content,
|
|
98
|
+
similarity: Math.round(similarity * 100) / 100
|
|
99
|
+
}
|
|
100
|
+
});
|
|
101
|
+
};
|
|
102
|
+
if (options.exhaustive) {
|
|
103
|
+
for (let i = 0; i < active.length; i += 1) {
|
|
104
|
+
for (let j = i + 1; j < active.length; j += 1)
|
|
105
|
+
score(i, j);
|
|
106
|
+
}
|
|
107
|
+
}
|
|
108
|
+
else {
|
|
109
|
+
const documentFrequency = new Map();
|
|
110
|
+
for (const set of sets) {
|
|
111
|
+
for (const token of set)
|
|
112
|
+
documentFrequency.set(token, (documentFrequency.get(token) ?? 0) + 1);
|
|
113
|
+
}
|
|
114
|
+
const buckets = new Map();
|
|
115
|
+
for (let i = 0; i < active.length; i += 1) {
|
|
116
|
+
const rarest = [...sets[i]]
|
|
117
|
+
.sort((a, b) => (documentFrequency.get(a) ?? 0) - (documentFrequency.get(b) ?? 0))
|
|
118
|
+
.slice(0, DUPLICATE_BLOCKING_TOKENS);
|
|
119
|
+
for (const token of rarest) {
|
|
120
|
+
const bucket = buckets.get(token);
|
|
121
|
+
if (bucket)
|
|
122
|
+
bucket.push(i);
|
|
123
|
+
else
|
|
124
|
+
buckets.set(token, [i]);
|
|
125
|
+
}
|
|
126
|
+
}
|
|
127
|
+
const cap = options.maxTokenBucket ?? MAX_TOKEN_BUCKET;
|
|
128
|
+
const seen = new Set();
|
|
129
|
+
const width = active.length;
|
|
130
|
+
for (const bucket of buckets.values()) {
|
|
131
|
+
if (bucket.length > cap)
|
|
62
132
|
continue;
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
133
|
+
for (let a = 0; a < bucket.length; a += 1) {
|
|
134
|
+
for (let b = a + 1; b < bucket.length; b += 1) {
|
|
135
|
+
const i = Math.min(bucket[a], bucket[b]);
|
|
136
|
+
const j = Math.max(bucket[a], bucket[b]);
|
|
137
|
+
const key = i * width + j;
|
|
138
|
+
if (seen.has(key))
|
|
139
|
+
continue;
|
|
140
|
+
seen.add(key);
|
|
141
|
+
score(i, j);
|
|
142
|
+
}
|
|
72
143
|
}
|
|
73
144
|
}
|
|
145
|
+
// Match the original emission order before the (stable) similarity sort, so
|
|
146
|
+
// equal-similarity pairs come out in the same order as the exhaustive scan.
|
|
147
|
+
found.sort((x, y) => x.i - y.i || x.j - y.j);
|
|
74
148
|
}
|
|
75
|
-
|
|
149
|
+
lastDuplicatePairsEvaluated = pairsEvaluated;
|
|
150
|
+
return found
|
|
151
|
+
.map((entry) => entry.pair)
|
|
152
|
+
.sort((left, right) => right.similarity - left.similarity)
|
|
153
|
+
.slice(0, limit);
|
|
76
154
|
}
|
|
77
155
|
/**
|
|
78
156
|
* Compare Peon-on vs Peon-off session token totals for a project. Returns null
|
package/dist/processor.d.ts
CHANGED
|
@@ -88,3 +88,39 @@ export declare class OpenRouterMemoryModelClient implements MemoryModelClient {
|
|
|
88
88
|
}): Promise<MemoryModelResult>;
|
|
89
89
|
}
|
|
90
90
|
export declare function parseProcessedMemory(content: string): ProcessedMemory;
|
|
91
|
+
/**
|
|
92
|
+
* Did the model server silently truncate the prompt?
|
|
93
|
+
*
|
|
94
|
+
* OpenAI-compatible servers report usage.prompt_tokens: what the model actually read.
|
|
95
|
+
* Ollama, at its 4096-token default, reports exactly 4096 for a ~17K-token prompt.
|
|
96
|
+
*
|
|
97
|
+
* The estimate comes from estimatePromptTokensForTruncation, and the threshold has to
|
|
98
|
+
* respect how rough it is. For English that estimate is chars/4, which OVER-estimates
|
|
99
|
+
* (real text runs ~5.5 chars/token), so an untruncated prompt still reports ~0.73 of
|
|
100
|
+
* it. A truncated one reports ~0.24. Below 0.5 is unambiguous: reaching it without
|
|
101
|
+
* truncation would take 8+ chars per token. Small prompts are ignored, where
|
|
102
|
+
* estimation noise is a large share of the total.
|
|
103
|
+
*/
|
|
104
|
+
export declare function detectPromptTruncation(estimatedPromptTokens: number, reportedPromptTokens: number | undefined): boolean;
|
|
105
|
+
/**
|
|
106
|
+
* Prompt size estimate for detectPromptTruncation ONLY. estimateTokens (chars/4) stays
|
|
107
|
+
* the cost/reporting estimate; this one exists because chars/4 undercounts token-dense
|
|
108
|
+
* scripts, which let a truncated CJK prompt pass as untruncated.
|
|
109
|
+
*
|
|
110
|
+
* The detector flags reported/estimated < 0.5, so each weight has to sit between two
|
|
111
|
+
* limits: high enough that a truncated prompt falls below 0.5, and at most ~2x the
|
|
112
|
+
* MOST efficient tokenizer's rate, or an untruncated prompt falls below 0.5 too. That
|
|
113
|
+
* false positive is the worse failure: the session is refused on every retry.
|
|
114
|
+
*
|
|
115
|
+
* - ASCII, 1/4 per char: unchanged, so English behaves exactly as before (chars/4
|
|
116
|
+
* over-counts English ~1.4x; measured untruncated ratio 0.73, truncated 0.24).
|
|
117
|
+
* - Han, kana, hangul, bopomofo, CJK and fullwidth punctuation, 0.75 per char: efficient
|
|
118
|
+
* tokenizers run ~0.45-0.6 tokens per CJK char, giving an untruncated ratio of 0.6-0.8.
|
|
119
|
+
* Qwen2.5 on Ollama runs ~0.65, so a truncated CJK prompt now reads well under 0.5.
|
|
120
|
+
* - Any other non-ASCII, 0.35 per char: Cyrillic, Greek, Arabic, accented Latin and the
|
|
121
|
+
* like pack into ~0.22-0.35 tokens per char on large-vocabulary tokenizers. A flat 0.75
|
|
122
|
+
* here would read an untruncated Russian log at ~0.31 and block it forever.
|
|
123
|
+
*
|
|
124
|
+
* Iterates code points, so an astral character (CJK Extension B, emoji) counts once.
|
|
125
|
+
*/
|
|
126
|
+
export declare function estimatePromptTokensForTruncation(text: string): number;
|
package/dist/processor.js
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { PeonMemoryStore } from "./memory-store.js";
|
|
2
|
-
import { loadPeonConfig } from "./config.js";
|
|
2
|
+
import { llmEnabled, loadPeonConfig } from "./config.js";
|
|
3
3
|
import { createQualityReport } from "./quality.js";
|
|
4
4
|
import { extractDomainEntitiesViaModel } from "./entity-extraction.js";
|
|
5
5
|
export class PeonMemoryProcessor {
|
|
@@ -99,7 +99,9 @@ export class PeonMemoryProcessor {
|
|
|
99
99
|
trigger: input.trigger,
|
|
100
100
|
force: input.force ?? false,
|
|
101
101
|
aiMode: this.config.aiMode,
|
|
102
|
-
|
|
102
|
+
// A local provider (Ollama) needs no key. Gating on openRouterApiKey here meant a
|
|
103
|
+
// fully-local setup skipped every automatic consolidation as "missing_api_key".
|
|
104
|
+
hasApiKey: llmEnabled(this.config),
|
|
103
105
|
hasManualAiResult: Boolean(input.aiResult)
|
|
104
106
|
});
|
|
105
107
|
if (decision.action === "skip") {
|
|
@@ -202,6 +204,11 @@ export class OpenRouterMemoryModelClient {
|
|
|
202
204
|
// Reserve explicit output room. Without this, OpenRouter applies the provider's default
|
|
203
205
|
// completion cap, which — paired with the delta cap on the input side — keeps the JSON
|
|
204
206
|
// reply from truncating mid-object. Env-tunable for very large brains.
|
|
207
|
+
// Force a JSON-object reply. A hosted model usually obeys "reply with JSON" from
|
|
208
|
+
// the prompt alone; a local 7B often answers conversationally instead ("It sounds
|
|
209
|
+
// like..."), which fails the parse and loses the whole consolidation. Ollama and
|
|
210
|
+
// the OpenAI API both honour this flag, so ask for it rather than trusting prose.
|
|
211
|
+
response_format: { type: "json_object" },
|
|
205
212
|
max_tokens: Number(process.env.PEON_CONSOLIDATION_MAX_TOKENS) || 8192
|
|
206
213
|
})
|
|
207
214
|
});
|
|
@@ -210,6 +217,19 @@ export class OpenRouterMemoryModelClient {
|
|
|
210
217
|
throw new Error(`OpenRouter memory processing failed with ${response.status}${body ? `: ${body}` : ""}`);
|
|
211
218
|
}
|
|
212
219
|
const json = (await response.json());
|
|
220
|
+
// A server whose context window is smaller than this prompt does not error: it
|
|
221
|
+
// keeps only the TAIL and silently drops the rest — which is the system prompt and
|
|
222
|
+
// the JSON schema. The model then answers without its instructions, and the result
|
|
223
|
+
// is empty or wrong. Refuse it rather than mark the session as consolidated.
|
|
224
|
+
const estimatedPromptTokens = estimatePromptTokensForTruncation(systemPrompt) + estimatePromptTokensForTruncation(userPrompt);
|
|
225
|
+
const reportedPromptTokens = json.usage?.prompt_tokens;
|
|
226
|
+
if (detectPromptTruncation(estimatedPromptTokens, reportedPromptTokens)) {
|
|
227
|
+
throw new Error(`The model server truncated the consolidation prompt: it processed ${reportedPromptTokens} tokens ` +
|
|
228
|
+
`of roughly ${estimatedPromptTokens} sent. Its context window is too small, so the instructions and ` +
|
|
229
|
+
`schema were cut off and the result would be empty. The session log was NOT consumed and will be ` +
|
|
230
|
+
`retried. Fix: give the model a larger context window — for Ollama, create a model with ` +
|
|
231
|
+
`"PARAMETER num_ctx 32768" (or set OLLAMA_CONTEXT_LENGTH) and point PEON_PROCESSING_MODEL at it.`);
|
|
232
|
+
}
|
|
213
233
|
const content = json.choices?.[0]?.message?.content;
|
|
214
234
|
if (!content)
|
|
215
235
|
throw new Error("OpenRouter memory processing response did not include content.");
|
|
@@ -351,6 +371,69 @@ function isMemoryStatus(value) {
|
|
|
351
371
|
function estimateTokens(text) {
|
|
352
372
|
return Math.max(1, Math.ceil(text.length / 4));
|
|
353
373
|
}
|
|
374
|
+
/**
|
|
375
|
+
* Did the model server silently truncate the prompt?
|
|
376
|
+
*
|
|
377
|
+
* OpenAI-compatible servers report usage.prompt_tokens: what the model actually read.
|
|
378
|
+
* Ollama, at its 4096-token default, reports exactly 4096 for a ~17K-token prompt.
|
|
379
|
+
*
|
|
380
|
+
* The estimate comes from estimatePromptTokensForTruncation, and the threshold has to
|
|
381
|
+
* respect how rough it is. For English that estimate is chars/4, which OVER-estimates
|
|
382
|
+
* (real text runs ~5.5 chars/token), so an untruncated prompt still reports ~0.73 of
|
|
383
|
+
* it. A truncated one reports ~0.24. Below 0.5 is unambiguous: reaching it without
|
|
384
|
+
* truncation would take 8+ chars per token. Small prompts are ignored, where
|
|
385
|
+
* estimation noise is a large share of the total.
|
|
386
|
+
*/
|
|
387
|
+
export function detectPromptTruncation(estimatedPromptTokens, reportedPromptTokens) {
|
|
388
|
+
if (!reportedPromptTokens || reportedPromptTokens <= 0)
|
|
389
|
+
return false; // no usage reported: cannot tell
|
|
390
|
+
if (estimatedPromptTokens - reportedPromptTokens < 1000)
|
|
391
|
+
return false;
|
|
392
|
+
return reportedPromptTokens / estimatedPromptTokens < 0.5;
|
|
393
|
+
}
|
|
394
|
+
/** Scripts that tokenize at roughly one token per character or more, not one per word. */
|
|
395
|
+
const TOKEN_DENSE_SCRIPT = /[\p{Script=Han}\p{Script=Hiragana}\p{Script=Katakana}\p{Script=Hangul}\p{Script=Bopomofo}\u3000-\u303F\uFF00-\uFFEF]/u;
|
|
396
|
+
const TOKENS_PER_ASCII_CHAR = 1 / 4;
|
|
397
|
+
const TOKENS_PER_DENSE_CHAR = 0.75;
|
|
398
|
+
const TOKENS_PER_OTHER_CHAR = 0.35;
|
|
399
|
+
/**
|
|
400
|
+
* Prompt size estimate for detectPromptTruncation ONLY. estimateTokens (chars/4) stays
|
|
401
|
+
* the cost/reporting estimate; this one exists because chars/4 undercounts token-dense
|
|
402
|
+
* scripts, which let a truncated CJK prompt pass as untruncated.
|
|
403
|
+
*
|
|
404
|
+
* The detector flags reported/estimated < 0.5, so each weight has to sit between two
|
|
405
|
+
* limits: high enough that a truncated prompt falls below 0.5, and at most ~2x the
|
|
406
|
+
* MOST efficient tokenizer's rate, or an untruncated prompt falls below 0.5 too. That
|
|
407
|
+
* false positive is the worse failure: the session is refused on every retry.
|
|
408
|
+
*
|
|
409
|
+
* - ASCII, 1/4 per char: unchanged, so English behaves exactly as before (chars/4
|
|
410
|
+
* over-counts English ~1.4x; measured untruncated ratio 0.73, truncated 0.24).
|
|
411
|
+
* - Han, kana, hangul, bopomofo, CJK and fullwidth punctuation, 0.75 per char: efficient
|
|
412
|
+
* tokenizers run ~0.45-0.6 tokens per CJK char, giving an untruncated ratio of 0.6-0.8.
|
|
413
|
+
* Qwen2.5 on Ollama runs ~0.65, so a truncated CJK prompt now reads well under 0.5.
|
|
414
|
+
* - Any other non-ASCII, 0.35 per char: Cyrillic, Greek, Arabic, accented Latin and the
|
|
415
|
+
* like pack into ~0.22-0.35 tokens per char on large-vocabulary tokenizers. A flat 0.75
|
|
416
|
+
* here would read an untruncated Russian log at ~0.31 and block it forever.
|
|
417
|
+
*
|
|
418
|
+
* Iterates code points, so an astral character (CJK Extension B, emoji) counts once.
|
|
419
|
+
*/
|
|
420
|
+
export function estimatePromptTokensForTruncation(text) {
|
|
421
|
+
// Count per class and multiply once: summing 0.35 thousands of times drifts past the
|
|
422
|
+
// integer and Math.ceil would round it up.
|
|
423
|
+
let ascii = 0;
|
|
424
|
+
let dense = 0;
|
|
425
|
+
let other = 0;
|
|
426
|
+
for (const char of text) {
|
|
427
|
+
if ((char.codePointAt(0) ?? 0) < 0x80)
|
|
428
|
+
ascii += 1;
|
|
429
|
+
else if (TOKEN_DENSE_SCRIPT.test(char))
|
|
430
|
+
dense += 1;
|
|
431
|
+
else
|
|
432
|
+
other += 1;
|
|
433
|
+
}
|
|
434
|
+
const tokens = ascii * TOKENS_PER_ASCII_CHAR + dense * TOKENS_PER_DENSE_CHAR + other * TOKENS_PER_OTHER_CHAR;
|
|
435
|
+
return Math.max(1, Math.ceil(tokens));
|
|
436
|
+
}
|
|
354
437
|
function estimateTokensByChars(chars) {
|
|
355
438
|
return Math.max(0, Math.ceil(chars / 4));
|
|
356
439
|
}
|