peon-mem 1.0.6 → 1.0.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/logger.js CHANGED
@@ -1,12 +1,32 @@
1
- import { appendFile, mkdir, readFile } from "node:fs/promises";
1
+ import { appendFile, mkdir, open, rename, stat } from "node:fs/promises";
2
2
  import { homedir } from "node:os";
3
- import { join } from "node:path";
3
+ import { dirname, join } from "node:path";
4
4
  const DEFAULT_LOG_DIR = join(homedir(), "Library", "Logs", "Peon");
5
+ /**
6
+ * Reads are bounded to the tail of the log. recent() used to slurp the entire
7
+ * file and split it, which meant a months-old 100 MB daemon.jsonl blocked the
8
+ * event loop on every monitor poll (flattening a 100 MB rope + allocating ~400k
9
+ * strings, only to throw all but `limit` of them away). Same tail-read strategy
10
+ * the query-embedding cache already uses.
11
+ *
12
+ * Sized to comfortably cover the largest real caller (token-stat seeding asks
13
+ * for 50k entries; entries average ~260 bytes), so bounding reads does not
14
+ * silently truncate what the daemon rebuilds on boot.
15
+ */
16
+ const DEFAULT_TAIL_BYTES = 16 * 1024 * 1024;
17
+ /** Rotate before the live file can reach a size that hurts anything. */
18
+ const DEFAULT_MAX_BYTES = 32 * 1024 * 1024;
5
19
  export class PeonLogger {
6
20
  logFile;
21
+ maxBytes;
22
+ tailBytes;
7
23
  writeQueue = Promise.resolve();
24
+ bytesWritten = 0;
25
+ sizeKnown = false;
8
26
  constructor(options = {}) {
9
27
  this.logFile = join(options.logDir ?? DEFAULT_LOG_DIR, "daemon.jsonl");
28
+ this.maxBytes = options.maxBytes ?? DEFAULT_MAX_BYTES;
29
+ this.tailBytes = options.tailBytes ?? DEFAULT_TAIL_BYTES;
10
30
  }
11
31
  async log(type, fields = {}) {
12
32
  const entry = {
@@ -18,10 +38,15 @@ export class PeonLogger {
18
38
  await this.enqueueWrite(`${JSON.stringify(entry)}\n`);
19
39
  return entry;
20
40
  }
41
+ /**
42
+ * Newest-first entries from the tail of the log. Cost is bounded by
43
+ * `tailBytes`, not by the size of the file, so this stays flat as the log
44
+ * grows. Entries older than the tail window are not visible here — the log
45
+ * file itself (and its rotated siblings) remain the full record.
46
+ */
21
47
  async recent(limit = 100) {
22
- const raw = await readFile(this.logFile, "utf8").catch(() => "");
23
- return raw
24
- .trim()
48
+ const text = await this.readTail();
49
+ return text
25
50
  .split("\n")
26
51
  .filter(Boolean)
27
52
  .slice(-limit)
@@ -31,15 +56,57 @@ export class PeonLogger {
31
56
  return [JSON.parse(line)];
32
57
  }
33
58
  catch {
59
+ // Skips both genuinely corrupt lines and the partial first line left
60
+ // by seeking into the middle of a record.
34
61
  return [];
35
62
  }
36
63
  });
37
64
  }
65
+ async readTail() {
66
+ let handle;
67
+ try {
68
+ const size = (await stat(this.logFile)).size;
69
+ const length = Math.min(size, this.tailBytes);
70
+ if (length === 0)
71
+ return "";
72
+ handle = await open(this.logFile, "r");
73
+ const buffer = Buffer.alloc(length);
74
+ await handle.read(buffer, 0, length, size - length);
75
+ const text = buffer.toString("utf8");
76
+ // A partial leading line is unavoidable when the window starts mid-record.
77
+ return length < size ? text.slice(text.indexOf("\n") + 1) : text;
78
+ }
79
+ catch {
80
+ return "";
81
+ }
82
+ finally {
83
+ await handle?.close().catch(() => undefined);
84
+ }
85
+ }
86
+ /**
87
+ * Move the live log aside once it exceeds maxBytes. History is preserved in a
88
+ * timestamped sibling rather than truncated, so nothing is lost.
89
+ */
90
+ async rotateIfNeeded() {
91
+ if (this.bytesWritten <= this.maxBytes)
92
+ return;
93
+ const stamp = new Date().toISOString().replace(/[:.]/g, "-");
94
+ await rename(this.logFile, `${this.logFile}.${stamp}`).catch(() => undefined);
95
+ this.bytesWritten = 0;
96
+ }
38
97
  async enqueueWrite(line) {
39
98
  const write = async () => {
40
99
  try {
41
- await mkdir(this.logFile.slice(0, this.logFile.lastIndexOf("/")), { recursive: true });
100
+ await mkdir(dirname(this.logFile), { recursive: true });
101
+ // Adopt the on-disk size once, so a daemon restart doesn't forget that
102
+ // an already-huge log is due for rotation.
103
+ if (!this.sizeKnown) {
104
+ this.bytesWritten = await stat(this.logFile).then((s) => s.size).catch(() => 0);
105
+ this.sizeKnown = true;
106
+ }
107
+ await this.rotateIfNeeded();
42
108
  await appendFile(this.logFile, line, "utf8");
109
+ this.bytesWritten += Buffer.byteLength(line, "utf8");
43
110
  }
44
111
  catch {
45
112
  // Best-effort logging: a log write failure must never crash the daemon
@@ -164,9 +164,13 @@ export declare class PeonMemoryStore {
164
164
  * No-op when embeddings are unavailable. supersededBy links to a merged-away id
165
165
  * are re-pointed at the surviving record so history stays intact.
166
166
  */
167
- mergeSimilarActiveRecords(records: MemoryRecord[], threshold?: number): Promise<{
167
+ mergeSimilarActiveRecords(records: MemoryRecord[], threshold?: number, options?: {
168
+ maxActive?: number;
169
+ exhaustive?: boolean;
170
+ }): Promise<{
168
171
  records: MemoryRecord[];
169
172
  merged: number;
173
+ comparisons: number;
170
174
  }>;
171
175
  readProcessingState(): Promise<ProcessingState>;
172
176
  writeProcessingState(state: ProcessingState): Promise<void>;
@@ -33,6 +33,78 @@ async function atomicWrite(path, content) {
33
33
  await writeFile(tmp, content, "utf8");
34
34
  await rename(tmp, path);
35
35
  }
36
+ /**
37
+ * Ceiling on how many active records the O(n^2) semantic dedup pass will consider.
38
+ * Override with PEON_DEDUP_MAX_ACTIVE. Set generously enough that ordinary project
39
+ * brains still dedup, low enough that a very large brain can't stall the daemon.
40
+ */
41
+ const DEDUP_MAX_ACTIVE = Number(process.env.PEON_DEDUP_MAX_ACTIVE) > 0
42
+ ? Number(process.env.PEON_DEDUP_MAX_ACTIVE)
43
+ : 25_000;
44
+ /**
45
+ * Deterministic random projections for LSH bucketing of dedup candidates.
46
+ * Seeded so bucketing is reproducible across runs and tests.
47
+ */
48
+ const DEDUP_BANDS = 8;
49
+ const DEDUP_BITS_PER_BAND = 6;
50
+ /** Yield to the event loop every N records so a long pass can't starve the daemon. */
51
+ const DEDUP_YIELD_EVERY = 200;
52
+ /**
53
+ * Hard ceiling on candidates examined per record. Bucketing alone only buys a
54
+ * constant factor when vectors are correlated (buckets fill unevenly), leaving the
55
+ * pass quadratic. Capping candidates makes the work O(n * k) — genuinely linear in
56
+ * brain size — at the cost of occasionally missing a merge in a very crowded bucket.
57
+ * A missed merge leaves a near-duplicate belief; an uncapped pass hangs the daemon.
58
+ */
59
+ const DEDUP_MAX_CANDIDATES = 128;
60
+ const projectionCache = new Map();
61
+ function mulberry32(seed) {
62
+ let a = seed >>> 0;
63
+ return () => {
64
+ a = (a + 0x6d2b79f5) >>> 0;
65
+ let t = Math.imul(a ^ (a >>> 15), 1 | a);
66
+ t = (t + Math.imul(t ^ (t >>> 7), 61 | t)) ^ t;
67
+ return ((t ^ (t >>> 14)) >>> 0) / 4294967296;
68
+ };
69
+ }
70
+ function projectionsFor(dim) {
71
+ const cached = projectionCache.get(dim);
72
+ if (cached)
73
+ return cached;
74
+ const rand = mulberry32(0x9e3779b9 ^ dim);
75
+ const planes = [];
76
+ for (let p = 0; p < DEDUP_BANDS * DEDUP_BITS_PER_BAND; p += 1) {
77
+ const plane = new Float32Array(dim);
78
+ for (let d = 0; d < dim; d += 1)
79
+ plane[d] = rand() * 2 - 1;
80
+ planes.push(plane);
81
+ }
82
+ projectionCache.set(dim, planes);
83
+ return planes;
84
+ }
85
+ /**
86
+ * Band keys for a vector: sign bits of random-hyperplane projections, grouped into
87
+ * bands. Two vectors with high cosine similarity agree on most bits, so they collide
88
+ * in at least one band with high probability — letting dedup compare a handful of
89
+ * plausible candidates instead of every record seen so far.
90
+ */
91
+ function bandKeys(vector) {
92
+ const dim = vector.length;
93
+ const planes = projectionsFor(dim);
94
+ const keys = [];
95
+ for (let band = 0; band < DEDUP_BANDS; band += 1) {
96
+ let bits = "";
97
+ for (let b = 0; b < DEDUP_BITS_PER_BAND; b += 1) {
98
+ const plane = planes[band * DEDUP_BITS_PER_BAND + b];
99
+ let dot = 0;
100
+ for (let d = 0; d < dim; d += 1)
101
+ dot += vector[d] * plane[d];
102
+ bits += dot >= 0 ? "1" : "0";
103
+ }
104
+ keys.push(`${band}:${bits}`);
105
+ }
106
+ return keys;
107
+ }
36
108
  export class PeonMemoryStore {
37
109
  projectPath;
38
110
  memoryDir;
@@ -612,18 +684,29 @@ export class PeonMemoryStore {
612
684
  * No-op when embeddings are unavailable. supersededBy links to a merged-away id
613
685
  * are re-pointed at the surviving record so history stays intact.
614
686
  */
615
- async mergeSimilarActiveRecords(records, threshold = 0.9) {
687
+ async mergeSimilarActiveRecords(records, threshold = 0.9, options = {}) {
616
688
  if (!this.embeddingClient || !this.embeddingStore)
617
- return { records, merged: 0 };
689
+ return { records, merged: 0, comparisons: 0 };
690
+ // SCALE GUARD. The pass below is O(n^2) pairwise cosine over 1536-dim vectors.
691
+ // On a 30k-record brain (~6.4k active) that is ~20.5M comparisons / ~31.6B float
692
+ // ops on the main thread, plus every vector resident as float64 — measured at
693
+ // 99% CPU and >1.9 GB RSS, which wedged the daemon's event loop entirely.
694
+ // Above the threshold we skip dedup rather than take the daemon down: a brain
695
+ // that keeps a few near-duplicates is strictly better than a brain that hangs.
696
+ // Checked BEFORE sync() so the vector sidecar is never even loaded.
697
+ const activeCount = records.reduce((n, r) => (r.status === "active" ? n + 1 : n), 0);
698
+ if (activeCount > (options.maxActive ?? DEDUP_MAX_ACTIVE)) {
699
+ return { records, merged: 0, comparisons: 0 };
700
+ }
618
701
  let vectorById;
619
702
  try {
620
703
  vectorById = (await this.embeddingStore.sync(records, this.embeddingClient)).vectorById;
621
704
  }
622
705
  catch {
623
- return { records, merged: 0 };
706
+ return { records, merged: 0, comparisons: 0 };
624
707
  }
625
708
  if (vectorById.size === 0)
626
- return { records, merged: 0 };
709
+ return { records, merged: 0, comparisons: 0 };
627
710
  const active = records.filter((record) => record.status === "active");
628
711
  const passthrough = records.filter((record) => record.status !== "active");
629
712
  const kept = [];
@@ -631,22 +714,81 @@ export class PeonMemoryStore {
631
714
  const remap = new Map();
632
715
  const mergeNow = new Date().toISOString();
633
716
  let merged = 0;
717
+ // Candidate index: band key -> indices into kept[]. Lets each record compare
718
+ // against a few plausible near-duplicates instead of every record so far,
719
+ // turning the old O(n^2) scan into roughly linear work.
720
+ const buckets = new Map();
721
+ const indexKept = (index, vector, type) => {
722
+ if (!vector)
723
+ return;
724
+ for (const key of bandKeys(vector)) {
725
+ const full = `${type}|${key}`;
726
+ const list = buckets.get(full);
727
+ if (list)
728
+ list.push(index);
729
+ else
730
+ buckets.set(full, [index]);
731
+ }
732
+ };
733
+ let comparisons = 0;
734
+ let processed = 0;
634
735
  for (const record of active) {
736
+ // Long passes must never starve the daemon's event loop the way the old
737
+ // fully synchronous scan did.
738
+ processed += 1;
739
+ if (processed % DEDUP_YIELD_EVERY === 0)
740
+ await new Promise((resolve) => setImmediate(resolve));
635
741
  const vec = vectorById.get(record.id);
636
742
  let matchIndex = -1;
637
743
  if (vec) {
638
- for (let i = 0; i < kept.length; i += 1) {
639
- if (kept[i].type !== record.type)
640
- continue;
641
- const other = vectorById.get(kept[i].id);
642
- if (other && cosineSimilarity(vec, other) >= threshold) {
643
- matchIndex = i;
644
- break;
744
+ if (options.exhaustive) {
745
+ for (let i = 0; i < kept.length; i += 1) {
746
+ if (kept[i].type !== record.type)
747
+ continue;
748
+ const other = vectorById.get(kept[i].id);
749
+ if (!other)
750
+ continue;
751
+ comparisons += 1;
752
+ if (cosineSimilarity(vec, other) >= threshold) {
753
+ matchIndex = i;
754
+ break;
755
+ }
756
+ }
757
+ }
758
+ else {
759
+ const seen = new Set();
760
+ let examined = 0;
761
+ for (const key of bandKeys(vec)) {
762
+ const candidates = buckets.get(`${record.type}|${key}`);
763
+ if (!candidates)
764
+ continue;
765
+ // Most recent entries first: a near-duplicate is likeliest among
766
+ // recently-seen beliefs, so a capped scan still finds the common case.
767
+ for (let c = candidates.length - 1; c >= 0; c -= 1) {
768
+ if (examined >= DEDUP_MAX_CANDIDATES)
769
+ break;
770
+ const i = candidates[c];
771
+ if (seen.has(i))
772
+ continue;
773
+ seen.add(i);
774
+ const other = vectorById.get(kept[i].id);
775
+ if (!other)
776
+ continue;
777
+ examined += 1;
778
+ comparisons += 1;
779
+ if (cosineSimilarity(vec, other) >= threshold) {
780
+ matchIndex = i;
781
+ break;
782
+ }
783
+ }
784
+ if (matchIndex !== -1 || examined >= DEDUP_MAX_CANDIDATES)
785
+ break;
645
786
  }
646
787
  }
647
788
  }
648
789
  if (matchIndex === -1) {
649
790
  kept.push(record);
791
+ indexKept(kept.length - 1, vec, record.type);
650
792
  continue;
651
793
  }
652
794
  const other = kept[matchIndex];
@@ -661,6 +803,10 @@ export class PeonMemoryStore {
661
803
  entities: unique([...record.entities, ...other.entities]),
662
804
  updatedAt: record.updatedAt > other.updatedAt ? record.updatedAt : other.updatedAt
663
805
  };
806
+ // The survivor can be the incoming record, so make its vector findable at
807
+ // that slot too — otherwise later near-duplicates could miss the bucket.
808
+ if (canonical.id === record.id)
809
+ indexKept(matchIndex, vec, record.type);
664
810
  remap.set(loser.id, canonical.id);
665
811
  // Recoverable-loser rule: don't destroy the merged-away belief — retire it as superseded,
666
812
  // linked to the survivor. It leaves active recall but its content stays recoverable and
@@ -682,7 +828,7 @@ export class PeonMemoryStore {
682
828
  const fixed = [...passthrough, ...retired].map((record) => record.supersededBy && remap.has(record.supersededBy)
683
829
  ? { ...record, supersededBy: resolveRemap(record.supersededBy) }
684
830
  : record);
685
- return { records: [...kept, ...fixed], merged };
831
+ return { records: [...kept, ...fixed], merged, comparisons };
686
832
  }
687
833
  async readProcessingState() {
688
834
  const raw = await readFile(join(this.memoryDir, "brain", "processing-state.json"), "utf8").catch(() => "");
@@ -34,14 +34,26 @@ export interface DuplicatePair {
34
34
  bContent: string;
35
35
  similarity: number;
36
36
  }
37
- /**
38
- * Flag near-duplicate ACTIVE beliefs of the same type — a nudge for the user to
39
- * merge, not an automatic action. Conservative threshold to avoid false alarms.
40
- */
41
- export declare function detectDuplicates(records: readonly MemoryRecord[], options?: {
37
+ /** How many record pairs the last duplicate scan actually compared. */
38
+ export declare function duplicateScanStats(): {
39
+ pairsEvaluated: number;
40
+ };
41
+ export interface DetectDuplicatesOptions {
42
42
  threshold?: number;
43
43
  limit?: number;
44
- }): DuplicatePair[];
44
+ /** Compare every pair (the original scan). Kept for equivalence testing. */
45
+ exhaustive?: boolean;
46
+ maxTokenBucket?: number;
47
+ }
48
+ /**
49
+ * Near-duplicate belief pairs, strongest first.
50
+ *
51
+ * Previously this compared every active pair — ~20.4M jaccard computations on a
52
+ * 6.4k-active brain, to return the top 5 — on both the /overview endpoint and the
53
+ * consolidation auto-merge path. Candidates are now found through an inverted index
54
+ * on rare tokens, so the overwhelming majority of pairs are never scored.
55
+ */
56
+ export declare function detectDuplicates(records: readonly MemoryRecord[], options?: DetectDuplicatesOptions): DuplicatePair[];
45
57
  export interface TokenSavings {
46
58
  onAvg: number;
47
59
  offAvg: number;
package/dist/overview.js CHANGED
@@ -50,29 +50,107 @@ function jaccard(a, b) {
50
50
  * Flag near-duplicate ACTIVE beliefs of the same type — a nudge for the user to
51
51
  * merge, not an automatic action. Conservative threshold to avoid false alarms.
52
52
  */
53
+ /**
54
+ * Only the rarest few tokens of a record are used as blocking keys: a pair at
55
+ * jaccard >= 0.6 shares most of its words, so it almost certainly shares one of
56
+ * them. Tokens common to a huge share of the brain are useless as keys and are
57
+ * skipped rather than producing a quadratic bucket.
58
+ */
59
+ const DUPLICATE_BLOCKING_TOKENS = 12;
60
+ const MAX_TOKEN_BUCKET = Number(process.env.PEON_DUPLICATE_MAX_TOKEN_BUCKET) > 0
61
+ ? Number(process.env.PEON_DUPLICATE_MAX_TOKEN_BUCKET)
62
+ : 400;
63
+ let lastDuplicatePairsEvaluated = 0;
64
+ /** How many record pairs the last duplicate scan actually compared. */
65
+ export function duplicateScanStats() {
66
+ return { pairsEvaluated: lastDuplicatePairsEvaluated };
67
+ }
68
+ /**
69
+ * Near-duplicate belief pairs, strongest first.
70
+ *
71
+ * Previously this compared every active pair — ~20.4M jaccard computations on a
72
+ * 6.4k-active brain, to return the top 5 — on both the /overview endpoint and the
73
+ * consolidation auto-merge path. Candidates are now found through an inverted index
74
+ * on rare tokens, so the overwhelming majority of pairs are never scored.
75
+ */
53
76
  export function detectDuplicates(records, options = {}) {
54
77
  const threshold = options.threshold ?? 0.6;
55
78
  const limit = options.limit ?? 5;
56
79
  const active = records.filter((record) => record.status === "active");
57
80
  const sets = active.map((record) => wordSet(record.content));
58
- const pairs = [];
59
- for (let i = 0; i < active.length; i += 1) {
60
- for (let j = i + 1; j < active.length; j += 1) {
61
- if (active[i].type !== active[j].type)
81
+ const found = [];
82
+ let pairsEvaluated = 0;
83
+ const score = (i, j) => {
84
+ if (active[i].type !== active[j].type)
85
+ return;
86
+ pairsEvaluated += 1;
87
+ const similarity = jaccard(sets[i], sets[j]);
88
+ if (similarity < threshold)
89
+ return;
90
+ found.push({
91
+ i,
92
+ j,
93
+ pair: {
94
+ aId: active[i].id,
95
+ aContent: active[i].content,
96
+ bId: active[j].id,
97
+ bContent: active[j].content,
98
+ similarity: Math.round(similarity * 100) / 100
99
+ }
100
+ });
101
+ };
102
+ if (options.exhaustive) {
103
+ for (let i = 0; i < active.length; i += 1) {
104
+ for (let j = i + 1; j < active.length; j += 1)
105
+ score(i, j);
106
+ }
107
+ }
108
+ else {
109
+ const documentFrequency = new Map();
110
+ for (const set of sets) {
111
+ for (const token of set)
112
+ documentFrequency.set(token, (documentFrequency.get(token) ?? 0) + 1);
113
+ }
114
+ const buckets = new Map();
115
+ for (let i = 0; i < active.length; i += 1) {
116
+ const rarest = [...sets[i]]
117
+ .sort((a, b) => (documentFrequency.get(a) ?? 0) - (documentFrequency.get(b) ?? 0))
118
+ .slice(0, DUPLICATE_BLOCKING_TOKENS);
119
+ for (const token of rarest) {
120
+ const bucket = buckets.get(token);
121
+ if (bucket)
122
+ bucket.push(i);
123
+ else
124
+ buckets.set(token, [i]);
125
+ }
126
+ }
127
+ const cap = options.maxTokenBucket ?? MAX_TOKEN_BUCKET;
128
+ const seen = new Set();
129
+ const width = active.length;
130
+ for (const bucket of buckets.values()) {
131
+ if (bucket.length > cap)
62
132
  continue;
63
- const similarity = jaccard(sets[i], sets[j]);
64
- if (similarity >= threshold) {
65
- pairs.push({
66
- aId: active[i].id,
67
- aContent: active[i].content,
68
- bId: active[j].id,
69
- bContent: active[j].content,
70
- similarity: Math.round(similarity * 100) / 100
71
- });
133
+ for (let a = 0; a < bucket.length; a += 1) {
134
+ for (let b = a + 1; b < bucket.length; b += 1) {
135
+ const i = Math.min(bucket[a], bucket[b]);
136
+ const j = Math.max(bucket[a], bucket[b]);
137
+ const key = i * width + j;
138
+ if (seen.has(key))
139
+ continue;
140
+ seen.add(key);
141
+ score(i, j);
142
+ }
72
143
  }
73
144
  }
145
+ // Match the original emission order before the (stable) similarity sort, so
146
+ // equal-similarity pairs come out in the same order as the exhaustive scan.
147
+ found.sort((x, y) => x.i - y.i || x.j - y.j);
74
148
  }
75
- return pairs.sort((left, right) => right.similarity - left.similarity).slice(0, limit);
149
+ lastDuplicatePairsEvaluated = pairsEvaluated;
150
+ return found
151
+ .map((entry) => entry.pair)
152
+ .sort((left, right) => right.similarity - left.similarity)
153
+ .slice(0, limit);
76
154
  }
77
155
  /**
78
156
  * Compare Peon-on vs Peon-off session token totals for a project. Returns null
@@ -88,3 +88,39 @@ export declare class OpenRouterMemoryModelClient implements MemoryModelClient {
88
88
  }): Promise<MemoryModelResult>;
89
89
  }
90
90
  export declare function parseProcessedMemory(content: string): ProcessedMemory;
91
+ /**
92
+ * Did the model server silently truncate the prompt?
93
+ *
94
+ * OpenAI-compatible servers report usage.prompt_tokens: what the model actually read.
95
+ * Ollama, at its 4096-token default, reports exactly 4096 for a ~17K-token prompt.
96
+ *
97
+ * The estimate comes from estimatePromptTokensForTruncation, and the threshold has to
98
+ * respect how rough it is. For English that estimate is chars/4, which OVER-estimates
99
+ * (real text runs ~5.5 chars/token), so an untruncated prompt still reports ~0.73 of
100
+ * it. A truncated one reports ~0.24. Below 0.5 is unambiguous: reaching it without
101
+ * truncation would take 8+ chars per token. Small prompts are ignored, where
102
+ * estimation noise is a large share of the total.
103
+ */
104
+ export declare function detectPromptTruncation(estimatedPromptTokens: number, reportedPromptTokens: number | undefined): boolean;
105
+ /**
106
+ * Prompt size estimate for detectPromptTruncation ONLY. estimateTokens (chars/4) stays
107
+ * the cost/reporting estimate; this one exists because chars/4 undercounts token-dense
108
+ * scripts, which let a truncated CJK prompt pass as untruncated.
109
+ *
110
+ * The detector flags reported/estimated < 0.5, so each weight has to sit between two
111
+ * limits: high enough that a truncated prompt falls below 0.5, and at most ~2x the
112
+ * MOST efficient tokenizer's rate, or an untruncated prompt falls below 0.5 too. That
113
+ * false positive is the worse failure: the session is refused on every retry.
114
+ *
115
+ * - ASCII, 1/4 per char: unchanged, so English behaves exactly as before (chars/4
116
+ * over-counts English ~1.4x; measured untruncated ratio 0.73, truncated 0.24).
117
+ * - Han, kana, hangul, bopomofo, CJK and fullwidth punctuation, 0.75 per char: efficient
118
+ * tokenizers run ~0.45-0.6 tokens per CJK char, giving an untruncated ratio of 0.6-0.8.
119
+ * Qwen2.5 on Ollama runs ~0.65, so a truncated CJK prompt now reads well under 0.5.
120
+ * - Any other non-ASCII, 0.35 per char: Cyrillic, Greek, Arabic, accented Latin and the
121
+ * like pack into ~0.22-0.35 tokens per char on large-vocabulary tokenizers. A flat 0.75
122
+ * here would read an untruncated Russian log at ~0.31 and block it forever.
123
+ *
124
+ * Iterates code points, so an astral character (CJK Extension B, emoji) counts once.
125
+ */
126
+ export declare function estimatePromptTokensForTruncation(text: string): number;
package/dist/processor.js CHANGED
@@ -1,5 +1,5 @@
1
1
  import { PeonMemoryStore } from "./memory-store.js";
2
- import { loadPeonConfig } from "./config.js";
2
+ import { llmEnabled, loadPeonConfig } from "./config.js";
3
3
  import { createQualityReport } from "./quality.js";
4
4
  import { extractDomainEntitiesViaModel } from "./entity-extraction.js";
5
5
  export class PeonMemoryProcessor {
@@ -99,7 +99,9 @@ export class PeonMemoryProcessor {
99
99
  trigger: input.trigger,
100
100
  force: input.force ?? false,
101
101
  aiMode: this.config.aiMode,
102
- hasApiKey: Boolean(this.config.openRouterApiKey),
102
+ // A local provider (Ollama) needs no key. Gating on openRouterApiKey here meant a
103
+ // fully-local setup skipped every automatic consolidation as "missing_api_key".
104
+ hasApiKey: llmEnabled(this.config),
103
105
  hasManualAiResult: Boolean(input.aiResult)
104
106
  });
105
107
  if (decision.action === "skip") {
@@ -202,6 +204,11 @@ export class OpenRouterMemoryModelClient {
202
204
  // Reserve explicit output room. Without this, OpenRouter applies the provider's default
203
205
  // completion cap, which — paired with the delta cap on the input side — keeps the JSON
204
206
  // reply from truncating mid-object. Env-tunable for very large brains.
207
+ // Force a JSON-object reply. A hosted model usually obeys "reply with JSON" from
208
+ // the prompt alone; a local 7B often answers conversationally instead ("It sounds
209
+ // like..."), which fails the parse and loses the whole consolidation. Ollama and
210
+ // the OpenAI API both honour this flag, so ask for it rather than trusting prose.
211
+ response_format: { type: "json_object" },
205
212
  max_tokens: Number(process.env.PEON_CONSOLIDATION_MAX_TOKENS) || 8192
206
213
  })
207
214
  });
@@ -210,6 +217,19 @@ export class OpenRouterMemoryModelClient {
210
217
  throw new Error(`OpenRouter memory processing failed with ${response.status}${body ? `: ${body}` : ""}`);
211
218
  }
212
219
  const json = (await response.json());
220
+ // A server whose context window is smaller than this prompt does not error: it
221
+ // keeps only the TAIL and silently drops the rest — which is the system prompt and
222
+ // the JSON schema. The model then answers without its instructions, and the result
223
+ // is empty or wrong. Refuse it rather than mark the session as consolidated.
224
+ const estimatedPromptTokens = estimatePromptTokensForTruncation(systemPrompt) + estimatePromptTokensForTruncation(userPrompt);
225
+ const reportedPromptTokens = json.usage?.prompt_tokens;
226
+ if (detectPromptTruncation(estimatedPromptTokens, reportedPromptTokens)) {
227
+ throw new Error(`The model server truncated the consolidation prompt: it processed ${reportedPromptTokens} tokens ` +
228
+ `of roughly ${estimatedPromptTokens} sent. Its context window is too small, so the instructions and ` +
229
+ `schema were cut off and the result would be empty. The session log was NOT consumed and will be ` +
230
+ `retried. Fix: give the model a larger context window — for Ollama, create a model with ` +
231
+ `"PARAMETER num_ctx 32768" (or set OLLAMA_CONTEXT_LENGTH) and point PEON_PROCESSING_MODEL at it.`);
232
+ }
213
233
  const content = json.choices?.[0]?.message?.content;
214
234
  if (!content)
215
235
  throw new Error("OpenRouter memory processing response did not include content.");
@@ -351,6 +371,69 @@ function isMemoryStatus(value) {
351
371
  function estimateTokens(text) {
352
372
  return Math.max(1, Math.ceil(text.length / 4));
353
373
  }
374
+ /**
375
+ * Did the model server silently truncate the prompt?
376
+ *
377
+ * OpenAI-compatible servers report usage.prompt_tokens: what the model actually read.
378
+ * Ollama, at its 4096-token default, reports exactly 4096 for a ~17K-token prompt.
379
+ *
380
+ * The estimate comes from estimatePromptTokensForTruncation, and the threshold has to
381
+ * respect how rough it is. For English that estimate is chars/4, which OVER-estimates
382
+ * (real text runs ~5.5 chars/token), so an untruncated prompt still reports ~0.73 of
383
+ * it. A truncated one reports ~0.24. Below 0.5 is unambiguous: reaching it without
384
+ * truncation would take 8+ chars per token. Small prompts are ignored, where
385
+ * estimation noise is a large share of the total.
386
+ */
387
+ export function detectPromptTruncation(estimatedPromptTokens, reportedPromptTokens) {
388
+ if (!reportedPromptTokens || reportedPromptTokens <= 0)
389
+ return false; // no usage reported: cannot tell
390
+ if (estimatedPromptTokens - reportedPromptTokens < 1000)
391
+ return false;
392
+ return reportedPromptTokens / estimatedPromptTokens < 0.5;
393
+ }
394
+ /** Scripts that tokenize at roughly one token per character or more, not one per word. */
395
+ const TOKEN_DENSE_SCRIPT = /[\p{Script=Han}\p{Script=Hiragana}\p{Script=Katakana}\p{Script=Hangul}\p{Script=Bopomofo}\u3000-\u303F\uFF00-\uFFEF]/u;
396
+ const TOKENS_PER_ASCII_CHAR = 1 / 4;
397
+ const TOKENS_PER_DENSE_CHAR = 0.75;
398
+ const TOKENS_PER_OTHER_CHAR = 0.35;
399
+ /**
400
+ * Prompt size estimate for detectPromptTruncation ONLY. estimateTokens (chars/4) stays
401
+ * the cost/reporting estimate; this one exists because chars/4 undercounts token-dense
402
+ * scripts, which let a truncated CJK prompt pass as untruncated.
403
+ *
404
+ * The detector flags reported/estimated < 0.5, so each weight has to sit between two
405
+ * limits: high enough that a truncated prompt falls below 0.5, and at most ~2x the
406
+ * MOST efficient tokenizer's rate, or an untruncated prompt falls below 0.5 too. That
407
+ * false positive is the worse failure: the session is refused on every retry.
408
+ *
409
+ * - ASCII, 1/4 per char: unchanged, so English behaves exactly as before (chars/4
410
+ * over-counts English ~1.4x; measured untruncated ratio 0.73, truncated 0.24).
411
+ * - Han, kana, hangul, bopomofo, CJK and fullwidth punctuation, 0.75 per char: efficient
412
+ * tokenizers run ~0.45-0.6 tokens per CJK char, giving an untruncated ratio of 0.6-0.8.
413
+ * Qwen2.5 on Ollama runs ~0.65, so a truncated CJK prompt now reads well under 0.5.
414
+ * - Any other non-ASCII, 0.35 per char: Cyrillic, Greek, Arabic, accented Latin and the
415
+ * like pack into ~0.22-0.35 tokens per char on large-vocabulary tokenizers. A flat 0.75
416
+ * here would read an untruncated Russian log at ~0.31 and block it forever.
417
+ *
418
+ * Iterates code points, so an astral character (CJK Extension B, emoji) counts once.
419
+ */
420
+ export function estimatePromptTokensForTruncation(text) {
421
+ // Count per class and multiply once: summing 0.35 thousands of times drifts past the
422
+ // integer and Math.ceil would round it up.
423
+ let ascii = 0;
424
+ let dense = 0;
425
+ let other = 0;
426
+ for (const char of text) {
427
+ if ((char.codePointAt(0) ?? 0) < 0x80)
428
+ ascii += 1;
429
+ else if (TOKEN_DENSE_SCRIPT.test(char))
430
+ dense += 1;
431
+ else
432
+ other += 1;
433
+ }
434
+ const tokens = ascii * TOKENS_PER_ASCII_CHAR + dense * TOKENS_PER_DENSE_CHAR + other * TOKENS_PER_OTHER_CHAR;
435
+ return Math.max(1, Math.ceil(tokens));
436
+ }
354
437
  function estimateTokensByChars(chars) {
355
438
  return Math.max(0, Math.ceil(chars / 4));
356
439
  }