peon-mem 1.0.6 → 1.0.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,3 +1,4 @@
1
+ import { llmEnabled, llmEndpoint, llmHeaders } from "./config.js";
1
2
  /**
2
3
  * Builds the LLM summarizer the brain uses to compress a topic cluster into one
3
4
  * gist belief. Kept separate from brain.ts so the curation logic stays pure and
@@ -5,7 +6,7 @@
5
6
  * or no key is configured (the brain then runs cost-free, skipping compression).
6
7
  */
7
8
  export function createClusterSummarizer(config) {
8
- if (config.aiMode === "off" || !config.openRouterApiKey)
9
+ if (!llmEnabled(config))
9
10
  return null;
10
11
  return async (cluster) => {
11
12
  const beliefs = cluster.members.map((m, i) => `${i + 1}. ${m.content}`).join("\n");
@@ -14,9 +15,9 @@ export function createClusterSummarizer(config) {
14
15
  "Losing a specific fact is a failure; merge wording, never drop information. Drop only redundancy and filler. " +
15
16
  "Output ONLY the summary sentence(s) — no preamble, no markdown, no quotes around it. Max 240 characters.";
16
17
  const user = `Topic: ${cluster.entity}\n\nBeliefs to compress:\n${beliefs}\n\nOne compact summary:`;
17
- const response = await fetch("https://openrouter.ai/api/v1/chat/completions", {
18
+ const response = await fetch(llmEndpoint(config), {
18
19
  method: "POST",
19
- headers: { Authorization: `Bearer ${config.openRouterApiKey}`, "Content-Type": "application/json" },
20
+ headers: llmHeaders(config),
20
21
  body: JSON.stringify({
21
22
  model: config.processingModel,
22
23
  messages: [
package/dist/config.d.ts CHANGED
@@ -19,4 +19,17 @@ export interface PeonConfig {
19
19
  type Env = Record<string, string | undefined>;
20
20
  export declare function loadPeonConfig(env?: Env): PeonConfig;
21
21
  export declare function readEnvFile(startDir?: string): Env;
22
+ /**
23
+ * Is an LLM available for optional AI passes (compression, entity extraction, HyDE,
24
+ * global extraction)?
25
+ *
26
+ * These used to gate on `openRouterApiKey`, which meant a fully-local setup
27
+ * (PEON_PROVIDER=ollama, no OpenRouter key) silently skipped every one of them —
28
+ * "local mode" was not actually local. A local provider needs no key; a hosted one does.
29
+ */
30
+ export declare function llmEnabled(config: PeonConfig): boolean;
31
+ /** The chat-completions endpoint for the configured provider. */
32
+ export declare function llmEndpoint(config: PeonConfig): string;
33
+ /** Auth + content headers for the configured provider (local providers need no key). */
34
+ export declare function llmHeaders(config: PeonConfig): Record<string, string>;
22
35
  export {};
package/dist/config.js CHANGED
@@ -97,3 +97,29 @@ function findEnvFile(startDir) {
97
97
  current = dirname(current);
98
98
  }
99
99
  }
100
+ /**
101
+ * Is an LLM available for optional AI passes (compression, entity extraction, HyDE,
102
+ * global extraction)?
103
+ *
104
+ * These used to gate on `openRouterApiKey`, which meant a fully-local setup
105
+ * (PEON_PROVIDER=ollama, no OpenRouter key) silently skipped every one of them —
106
+ * "local mode" was not actually local. A local provider needs no key; a hosted one does.
107
+ */
108
+ export function llmEnabled(config) {
109
+ if (config.aiMode === "off")
110
+ return false;
111
+ if (config.provider === "ollama")
112
+ return true;
113
+ return Boolean(config.llmApiKey ?? config.openRouterApiKey);
114
+ }
115
+ /** The chat-completions endpoint for the configured provider. */
116
+ export function llmEndpoint(config) {
117
+ return `${config.llmBaseUrl.replace(/\/$/, "")}/chat/completions`;
118
+ }
119
+ /** Auth + content headers for the configured provider (local providers need no key). */
120
+ export function llmHeaders(config) {
121
+ return {
122
+ Authorization: `Bearer ${config.llmApiKey ?? config.openRouterApiKey ?? ""}`,
123
+ "Content-Type": "application/json"
124
+ };
125
+ }
package/dist/daemon.js CHANGED
@@ -102,9 +102,9 @@ export function canonicalProjectPath(projectPath, home = homedir()) {
102
102
  /**
103
103
  * Reject requests that aren't from a loopback caller — defeats DNS-rebinding and drive-by-localhost
104
104
  * attacks where a malicious web page POSTs to the daemon (which would otherwise write/read a brain
105
- * at an attacker-controlled path). A bad Host header (rebinding) or a cross-origin Origin/Referer
106
- * (browser drive-by) is refused. The node hook (no Origin) and the local monitor UI (loopback
107
- * Origin) both pass.
105
+ * at an attacker-controlled path). A bad Host header (rebinding), a cross-origin Origin/Referer
106
+ * (browser drive-by) or a browser's Sec-Fetch-Site: cross-site is refused. The node hook (no
107
+ * Origin) and the local monitor UI (loopback Origin, same-origin fetches) all pass.
108
108
  */
109
109
  function isLoopbackHostname(hostname) {
110
110
  const h = hostname.toLowerCase().replace(/^\[|\]$/g, "");
@@ -128,6 +128,12 @@ function isLocalRequest(request) {
128
128
  // A present Host must be loopback (a missing Host — rare, HTTP/1.0 — is allowed; bind is 127.0.0.1).
129
129
  if (rawHost !== undefined && !isLoopbackHostname(hostnameFromHostHeader(String(rawHost))))
130
130
  return false;
131
+ // Browsers label every request with where it came from. An <img> or no-referrer fetch from
132
+ // another site carries no Origin or Referer, so the loop below cannot see it, yet a GET like
133
+ // /context?projectPath=... still creates a brain at that path. Non-browser callers (the hook,
134
+ // the MCP server, curl) send no Sec-Fetch-Site and are unaffected.
135
+ if (String(request.headers["sec-fetch-site"] ?? "").toLowerCase() === "cross-site")
136
+ return false;
131
137
  for (const header of [request.headers.origin, request.headers.referer]) {
132
138
  if (!header)
133
139
  continue;
@@ -20,6 +20,16 @@ export interface SyncResult {
20
20
  reused: number;
21
21
  pruned: number;
22
22
  }
23
+ export declare function embeddingCacheStats(): {
24
+ cachedStores: number;
25
+ diskReads: number;
26
+ reads: number;
27
+ expireOlderThan: (now: number) => void;
28
+ };
29
+ /** Test helper: forget every cached sidecar. */
30
+ export declare function resetEmbeddingCaches(): void;
31
+ /** Test helper: forget learned widths, simulating a fresh daemon process. */
32
+ export declare function resetEmbeddingDimensionCache(): void;
23
33
  export declare class EmbeddingStore {
24
34
  private readonly filePath;
25
35
  private cache?;
@@ -33,6 +43,8 @@ export declare class EmbeddingStore {
33
43
  * empty map rather than throwing (retrieval falls back to lexical-only).
34
44
  */
35
45
  sync(records: MemoryRecord[], client: EmbeddingClient | null): Promise<SyncResult>;
46
+ /** Release this store's parsed sidecar. Costs one re-read, never correctness. */
47
+ dropCache(): void;
36
48
  /** Read vectors without recomputing — used by read-only retrieval paths. */
37
49
  vectorById(): Promise<Map<string, EmbeddingVector>>;
38
50
  private persist;
@@ -1,6 +1,73 @@
1
1
  import { mkdir, readFile, rename, stat, writeFile } from "node:fs/promises";
2
2
  import { dirname, join } from "node:path";
3
3
  import { contentHash } from "./embeddings.js";
4
+ /**
5
+ * Bounded registry of live sidecar caches.
6
+ *
7
+ * The cache below keeps a whole embeddings.jsonl parsed in memory so read-only
8
+ * retrieval doesn't re-parse a multi-MB file on every prompt. That is a real win,
9
+ * but the daemon holds one store per project and nothing evicted: a heap snapshot
10
+ * showed 8 stores pinning 60,288 Float32Array vectors / 338 MB of native backing,
11
+ * with RSS past 1.9 GB — enough GC pressure to peg the CPU and stop the daemon
12
+ * answering. So caches are now LRU-bounded and released once idle; a dropped cache
13
+ * costs one re-read, never correctness.
14
+ */
15
+ const MAX_CACHED_STORES = Number(process.env.PEON_EMBED_CACHE_STORES) > 0
16
+ ? Number(process.env.PEON_EMBED_CACHE_STORES)
17
+ : 2;
18
+ const CACHE_TTL_MS = Number(process.env.PEON_EMBED_CACHE_TTL_MS) > 0
19
+ ? Number(process.env.PEON_EMBED_CACHE_TTL_MS)
20
+ : 5 * 60 * 1000;
21
+ /** Insertion order is LRU order: least-recently-used first. */
22
+ const liveCaches = new Map();
23
+ let diskReads = 0;
24
+ function touchCache(store, at) {
25
+ liveCaches.delete(store);
26
+ liveCaches.set(store, at);
27
+ pruneCaches(at);
28
+ }
29
+ function pruneCaches(now) {
30
+ for (const [store, touchedAt] of [...liveCaches]) {
31
+ if (now - touchedAt > CACHE_TTL_MS) {
32
+ store.dropCache();
33
+ liveCaches.delete(store);
34
+ }
35
+ }
36
+ while (liveCaches.size > MAX_CACHED_STORES) {
37
+ const oldest = liveCaches.keys().next().value;
38
+ if (!oldest)
39
+ break;
40
+ oldest.dropCache();
41
+ liveCaches.delete(oldest);
42
+ }
43
+ }
44
+ // TTL alone only fires on activity, so an idle daemon would hold its last caches
45
+ // forever. A low-frequency sweeper lets a quiet daemon settle back down; unref'd
46
+ // so it never keeps the process alive on its own.
47
+ const sweeper = setInterval(() => pruneCaches(Date.now()), 60_000);
48
+ if (typeof sweeper.unref === "function")
49
+ sweeper.unref();
50
+ export function embeddingCacheStats() {
51
+ return {
52
+ cachedStores: liveCaches.size,
53
+ diskReads,
54
+ reads: diskReads,
55
+ expireOlderThan: (now) => pruneCaches(now)
56
+ };
57
+ }
58
+ /** Test helper: forget every cached sidecar. */
59
+ export function resetEmbeddingCaches() {
60
+ for (const store of [...liveCaches.keys()])
61
+ store.dropCache();
62
+ liveCaches.clear();
63
+ diskReads = 0;
64
+ }
65
+ /** Real output width per embedding model, learned once per process. */
66
+ const modelDimensions = new Map();
67
+ /** Test helper: forget learned widths, simulating a fresh daemon process. */
68
+ export function resetEmbeddingDimensionCache() {
69
+ modelDimensions.clear();
70
+ }
4
71
  export class EmbeddingStore {
5
72
  filePath;
6
73
  // mtime-keyed cache so the (multi-MB) sidecar isn't re-read+parsed on every prompt's
@@ -23,8 +90,11 @@ export class EmbeddingStore {
23
90
  catch {
24
91
  mtimeMs = 0; // missing file → treat as empty, mtime 0
25
92
  }
26
- if (this.cache && this.cache.mtimeMs === mtimeMs)
93
+ if (this.cache && this.cache.mtimeMs === mtimeMs) {
94
+ touchCache(this, Date.now());
27
95
  return this.cache.map;
96
+ }
97
+ diskReads += 1;
28
98
  const raw = await readFile(this.filePath, "utf8").catch(() => "");
29
99
  const map = new Map();
30
100
  for (const line of raw.split(/\r?\n/)) {
@@ -41,6 +111,7 @@ export class EmbeddingStore {
41
111
  }
42
112
  }
43
113
  this.cache = { mtimeMs, map };
114
+ touchCache(this, Date.now());
44
115
  return map;
45
116
  }
46
117
  /**
@@ -56,11 +127,40 @@ export class EmbeddingStore {
56
127
  const existing = await this.load();
57
128
  const liveIds = new Set(records.map((record) => record.id));
58
129
  const pruned = [...existing.keys()].filter((id) => !liveIds.has(id)).length;
130
+ // A stored vector can carry the right model name and hash yet the wrong width —
131
+ // that is exactly what a degraded fallback wrote — and cosineSimilarity scores any
132
+ // width mismatch as 0, so those records vanish from semantic recall without ever
133
+ // erroring. Width is part of validity, so we need to know the client's real width.
134
+ //
135
+ // Learning it must not cost a round trip on every sync: the width is cached per
136
+ // model for the process, and only probed when there is nothing to compute (the
137
+ // one case where a fully-poisoned sidecar would otherwise look entirely reusable).
138
+ let expectedDim = modelDimensions.get(client.model) ?? 0;
139
+ const nothingToRecompute = records.every((record) => {
140
+ const prior = existing.get(record.id);
141
+ return prior && prior.model === client.model && prior.hash === contentHash(embeddingText(record));
142
+ });
143
+ if (expectedDim === 0 && records.length > 0 && nothingToRecompute) {
144
+ try {
145
+ const probe = await client.embed([embeddingText(records[0])]);
146
+ if (!client.degraded && probe[0]?.length) {
147
+ expectedDim = probe[0].length;
148
+ modelDimensions.set(client.model, expectedDim);
149
+ }
150
+ }
151
+ catch {
152
+ expectedDim = 0; // cannot probe — fall back to model+hash validity only
153
+ }
154
+ }
155
+ const validDim = (vector) => expectedDim === 0 || vector.length === expectedDim;
59
156
  const toCompute = [];
60
157
  let reused = 0;
61
158
  for (const record of records) {
62
159
  const prior = existing.get(record.id);
63
- if (prior && prior.model === client.model && prior.hash === contentHash(embeddingText(record))) {
160
+ if (prior &&
161
+ prior.model === client.model &&
162
+ prior.hash === contentHash(embeddingText(record)) &&
163
+ validDim(prior.vector)) {
64
164
  reused += 1;
65
165
  }
66
166
  else {
@@ -70,7 +170,10 @@ export class EmbeddingStore {
70
170
  const result = new Map();
71
171
  for (const record of records) {
72
172
  const prior = existing.get(record.id);
73
- if (prior && prior.model === client.model && prior.hash === contentHash(embeddingText(record))) {
173
+ if (prior &&
174
+ prior.model === client.model &&
175
+ prior.hash === contentHash(embeddingText(record)) &&
176
+ validDim(prior.vector)) {
74
177
  result.set(record.id, prior);
75
178
  }
76
179
  }
@@ -78,6 +181,20 @@ export class EmbeddingStore {
78
181
  if (toCompute.length > 0) {
79
182
  try {
80
183
  const vectors = await client.embed(toCompute.map((record) => embeddingText(record)));
184
+ // A degraded run returns local trigram vectors. Serving them for THIS call is
185
+ // fine (graceful degradation); writing them under the primary's model name is
186
+ // not — they would be reused forever as if they were real embeddings.
187
+ if (client.degraded) {
188
+ const degradedById = new Map();
189
+ for (const [id, stored] of result)
190
+ degradedById.set(id, stored.vector);
191
+ toCompute.forEach((record, i) => {
192
+ const vector = vectors[i];
193
+ if (vector)
194
+ degradedById.set(record.id, vector);
195
+ });
196
+ return { vectorById: degradedById, computed: 0, reused, pruned };
197
+ }
81
198
  toCompute.forEach((record, i) => {
82
199
  result.set(record.id, {
83
200
  id: record.id,
@@ -87,6 +204,9 @@ export class EmbeddingStore {
87
204
  });
88
205
  });
89
206
  computed = toCompute.length;
207
+ const width = vectors[0]?.length ?? 0;
208
+ if (width > 0)
209
+ modelDimensions.set(client.model, width);
90
210
  }
91
211
  catch {
92
212
  // On a hard failure, keep whatever we already had and continue lexical-only.
@@ -101,6 +221,10 @@ export class EmbeddingStore {
101
221
  vectorById.set(id, stored.vector);
102
222
  return { vectorById, computed, reused, pruned };
103
223
  }
224
+ /** Release this store's parsed sidecar. Costs one re-read, never correctness. */
225
+ dropCache() {
226
+ this.cache = undefined;
227
+ }
104
228
  /** Read vectors without recomputing — used by read-only retrieval paths. */
105
229
  async vectorById() {
106
230
  const stored = await this.load();
@@ -123,6 +247,7 @@ export class EmbeddingStore {
123
247
  await writeFile(tmp, lines.length > 0 ? `${lines.join("\n")}\n` : "", "utf8");
124
248
  await rename(tmp, this.filePath);
125
249
  this.cache = undefined; // invalidate; next load() re-reads the fresh file
250
+ liveCaches.delete(this);
126
251
  }
127
252
  }
128
253
  /** Embed the record type alongside content so type acts as a soft semantic anchor. */
@@ -81,6 +81,15 @@ export declare class FallbackEmbeddingClient implements EmbeddingClient {
81
81
  private readonly fallback;
82
82
  private readonly onFallback?;
83
83
  readonly model: string;
84
+ /**
85
+ * True when the most recent embed() degraded to the fallback. The vectors it
86
+ * returns are a different model AND a different width, so persisting them under
87
+ * the primary's name makes them indistinguishable from real ones — every later
88
+ * sync then "reuses" trigram vectors as if they were embeddings, and retrieval
89
+ * silently scores them 0 (cosineSimilarity returns 0 on a length mismatch).
90
+ * Callers that persist vectors must check this and skip writing.
91
+ */
92
+ degraded: boolean;
84
93
  constructor(primary: EmbeddingClient, fallback?: EmbeddingClient, onFallback?: ((error: unknown) => void) | undefined);
85
94
  embed(texts: string[]): Promise<EmbeddingVector[]>;
86
95
  }
@@ -90,4 +99,25 @@ export interface CreateEmbeddingClientOptions {
90
99
  onFallback?: (error: unknown) => void;
91
100
  }
92
101
  /** Build the embedding client implied by config, or null when embeddings are off. */
102
+ /**
103
+ * What was asked for versus what will actually run.
104
+ *
105
+ * Peon degrades to deterministic local trigram embeddings whenever the configured
106
+ * embedder is unavailable. That is deliberate — retrieval keeps working — but it was
107
+ * silent, and a silent downgrade is indistinguishable from working correctly while
108
+ * semantic recall quietly collapses. Two real incidents: an Ollama blip embedding 30k+
109
+ * records with trigram vectors, and a script whose .env was not found resolving to
110
+ * "local" with no warning at all.
111
+ */
112
+ export interface EmbeddingPlan {
113
+ intended: PeonConfig["embeddingMode"];
114
+ effective: "off" | "local" | "api" | "ollama";
115
+ downgraded: boolean;
116
+ reason?: string;
117
+ }
118
+ /** Only the fields the decision actually depends on, matching the client factory. */
119
+ export type EmbeddingPlanInput = Pick<PeonConfig, "embeddingMode" | "embeddingModel" | "openRouterApiKey" | "provider" | "llmApiKey">;
120
+ export declare function resolveEmbeddingPlan(config: EmbeddingPlanInput): EmbeddingPlan;
121
+ /** Test helper: forget which downgrade warnings have already been emitted. */
122
+ export declare function resetEmbeddingWarnings(): void;
93
123
  export declare function createEmbeddingClient(options: CreateEmbeddingClientOptions): EmbeddingClient | null;
@@ -290,6 +290,15 @@ export class FallbackEmbeddingClient {
290
290
  fallback;
291
291
  onFallback;
292
292
  model;
293
+ /**
294
+ * True when the most recent embed() degraded to the fallback. The vectors it
295
+ * returns are a different model AND a different width, so persisting them under
296
+ * the primary's name makes them indistinguishable from real ones — every later
297
+ * sync then "reuses" trigram vectors as if they were embeddings, and retrieval
298
+ * silently scores them 0 (cosineSimilarity returns 0 on a length mismatch).
299
+ * Callers that persist vectors must check this and skip writing.
300
+ */
301
+ degraded = false;
293
302
  constructor(primary, fallback = new LocalEmbeddingClient(), onFallback) {
294
303
  this.primary = primary;
295
304
  this.fallback = fallback;
@@ -298,19 +307,73 @@ export class FallbackEmbeddingClient {
298
307
  }
299
308
  async embed(texts) {
300
309
  try {
301
- return await this.primary.embed(texts);
310
+ const vectors = await this.primary.embed(texts);
311
+ this.degraded = false;
312
+ return vectors;
302
313
  }
303
314
  catch (error) {
304
315
  this.onFallback?.(error);
316
+ this.degraded = true;
305
317
  return this.fallback.embed(texts);
306
318
  }
307
319
  }
308
320
  }
309
- /** Build the embedding client implied by config, or null when embeddings are off. */
321
+ export function resolveEmbeddingPlan(config) {
322
+ const intended = config.embeddingMode;
323
+ if (intended === "off")
324
+ return { intended, effective: "off", downgraded: false };
325
+ if (intended === "local")
326
+ return { intended, effective: "local", downgraded: false };
327
+ if (intended === "ollama") {
328
+ // Reachability cannot be known at construction time; a dead server surfaces at
329
+ // the first embed() as a degraded run, which the store refuses to persist.
330
+ return { intended, effective: "ollama", downgraded: false };
331
+ }
332
+ const apiKey = config.llmApiKey ?? config.openRouterApiKey;
333
+ if (!apiKey) {
334
+ return { intended, effective: "local", downgraded: true, reason: "no API key/credentials configured" };
335
+ }
336
+ if (!config.embeddingModel) {
337
+ return { intended, effective: "local", downgraded: true, reason: "PEON_EMBEDDING_MODEL is not set" };
338
+ }
339
+ if (config.provider === "anthropic") {
340
+ return { intended, effective: "local", downgraded: true, reason: "Anthropic has no embeddings API" };
341
+ }
342
+ return { intended, effective: "api", downgraded: false };
343
+ }
344
+ /** Warn once per distinct reason, so a long-lived daemon does not spam its log. */
345
+ const warnedDowngrades = new Set();
346
+ function warnOnce(plan) {
347
+ if (!plan.downgraded || !plan.reason)
348
+ return;
349
+ if (warnedDowngrades.has(plan.reason))
350
+ return;
351
+ warnedDowngrades.add(plan.reason);
352
+ console.warn(`[peon] embedding mode "${plan.intended}" is not available (${plan.reason}); ` +
353
+ `falling back to local trigram embeddings. Semantic recall will be much weaker — ` +
354
+ `set PEON_EMBEDDING_MODE=local to silence this, or fix the configuration.`);
355
+ }
356
+ /** Test helper: forget which downgrade warnings have already been emitted. */
357
+ export function resetEmbeddingWarnings() {
358
+ warnedDowngrades.clear();
359
+ }
310
360
  export function createEmbeddingClient(options) {
311
361
  const { config } = options;
362
+ warnOnce(resolveEmbeddingPlan(config));
312
363
  if (config.embeddingMode === "off")
313
364
  return null;
365
+ // No caller ever supplied onFallback, so a runtime degrade (embedding server down)
366
+ // was completely silent. Default to warning once per process: the vectors from that
367
+ // run are trigram, not semantic, and the operator needs to know retrieval got worse.
368
+ const onFallback = options.onFallback ??
369
+ ((error) => {
370
+ if (warnedDowngrades.has("runtime-fallback"))
371
+ return;
372
+ warnedDowngrades.add("runtime-fallback");
373
+ console.warn(`[peon] embedding request failed (${error instanceof Error ? error.message : String(error)}); ` +
374
+ `falling back to local trigram embeddings for this run. Semantic recall is degraded ` +
375
+ `until the embedding server is reachable again.`);
376
+ });
314
377
  if (config.embeddingMode === "ollama") {
315
378
  // Local semantic embeddings. Fall back to the API client (if configured) then trigram-local,
316
379
  // so a stopped Ollama service degrades instead of breaking retrieval.
@@ -319,9 +382,9 @@ export function createEmbeddingClient(options) {
319
382
  baseUrl: config.ollamaBaseUrl
320
383
  });
321
384
  const fallback = config.openRouterApiKey && config.embeddingModel && config.embeddingModel.includes("/")
322
- ? new FallbackEmbeddingClient(new OpenRouterEmbeddingClient({ apiKey: config.openRouterApiKey, model: config.embeddingModel }), new LocalEmbeddingClient(), options.onFallback)
385
+ ? new FallbackEmbeddingClient(new OpenRouterEmbeddingClient({ apiKey: config.openRouterApiKey, model: config.embeddingModel }), new LocalEmbeddingClient(), onFallback)
323
386
  : new LocalEmbeddingClient();
324
- return new FallbackEmbeddingClient(ollama, fallback, options.onFallback);
387
+ return new FallbackEmbeddingClient(ollama, fallback, onFallback);
325
388
  }
326
389
  const apiKey = config.llmApiKey ?? config.openRouterApiKey;
327
390
  const embeddable = config.provider !== "anthropic"; // Anthropic has no embeddings API — local fallback
@@ -331,7 +394,7 @@ export function createEmbeddingClient(options) {
331
394
  model: config.embeddingModel,
332
395
  baseUrl: config.llmBaseUrl
333
396
  });
334
- return new FallbackEmbeddingClient(primary, new LocalEmbeddingClient(), options.onFallback);
397
+ return new FallbackEmbeddingClient(primary, new LocalEmbeddingClient(), onFallback);
335
398
  }
336
399
  // Default and "api"-without-credentials both resolve to deterministic local embeddings.
337
400
  return new LocalEmbeddingClient();
@@ -1,9 +1,10 @@
1
+ import { llmEnabled, llmEndpoint, llmHeaders } from "./config.js";
1
2
  const DEFAULT_MAX_ITEMS = 40;
2
3
  const DEFAULT_SNIPPET_CHARS = 240;
3
4
  export async function extractDomainEntitiesViaModel(items, options) {
4
5
  const out = new Map();
5
6
  const { config } = options;
6
- if (items.length === 0 || config.aiMode === "off" || !config.openRouterApiKey)
7
+ if (items.length === 0 || !llmEnabled(config))
7
8
  return out;
8
9
  const doFetch = options.fetchImpl ?? globalThis.fetch;
9
10
  if (!doFetch)
@@ -19,9 +20,9 @@ export async function extractDomainEntitiesViaModel(items, options) {
19
20
  '{"n": <snippet number>, "entities": ["..."]}, empty array when a snippet names none. No prose, no fences.';
20
21
  const user = `Snippets:\n${numbered}\n\nJSON array:`;
21
22
  try {
22
- const response = await doFetch("https://openrouter.ai/api/v1/chat/completions", {
23
+ const response = await doFetch(llmEndpoint(config), {
23
24
  method: "POST",
24
- headers: { Authorization: `Bearer ${config.openRouterApiKey}`, "Content-Type": "application/json" },
25
+ headers: llmHeaders(config),
25
26
  body: JSON.stringify({
26
27
  model: options.model ?? config.processingModel,
27
28
  messages: [
@@ -1,5 +1,6 @@
1
+ import { llmEnabled, llmEndpoint, llmHeaders } from "./config.js";
1
2
  export function createGlobalExtractor(config) {
2
- if (config.aiMode === "off" || !config.openRouterApiKey)
3
+ if (!llmEnabled(config))
3
4
  return null;
4
5
  return async (records) => {
5
6
  // Send the highest-signal beliefs only — bounds tokens, focuses the model.
@@ -22,9 +23,9 @@ export function createGlobalExtractor(config) {
22
23
  "Example DROP (project-internal): 'The daemon exposes a /global/extract endpoint.' " +
23
24
  "Rewrite each as one self-contained sentence with zero project context. " +
24
25
  "Output ONLY a JSON array of strings — no markdown fences, no prose. If nothing qualifies, return []. Max 8 items.";
25
- const response = await fetch("https://openrouter.ai/api/v1/chat/completions", {
26
+ const response = await fetch(llmEndpoint(config), {
26
27
  method: "POST",
27
- headers: { Authorization: `Bearer ${config.openRouterApiKey}`, "Content-Type": "application/json" },
28
+ headers: llmHeaders(config),
28
29
  body: JSON.stringify({
29
30
  model: config.processingModel,
30
31
  messages: [
package/dist/hyde.js CHANGED
@@ -1,10 +1,11 @@
1
+ import { llmEnabled, llmEndpoint } from "./config.js";
1
2
  const DEFAULT_MAX_CHARS = 320;
2
3
  export async function expandQuery(query, options) {
3
4
  const q = (query ?? "").trim();
4
5
  if (!q)
5
6
  return { expanded: "", hypothetical: "" };
6
7
  const { config } = options;
7
- if (config.aiMode === "off" || !config.openRouterApiKey)
8
+ if (!llmEnabled(config))
8
9
  return { expanded: q, hypothetical: "" };
9
10
  const doFetch = options.fetchImpl ?? globalThis.fetch;
10
11
  if (!doFetch)
@@ -16,10 +17,10 @@ export async function expandQuery(query, options) {
16
17
  "Do not hedge, do not say you lack context, do not ask questions. Output the sentences only.";
17
18
  const user = `Question: ${q}\n\nHypothetical answer:`;
18
19
  try {
19
- const response = await doFetch("https://openrouter.ai/api/v1/chat/completions", {
20
+ const response = await doFetch(llmEndpoint(config), {
20
21
  method: "POST",
21
22
  headers: {
22
- Authorization: `Bearer ${config.openRouterApiKey}`,
23
+ Authorization: `Bearer ${config.llmApiKey ?? config.openRouterApiKey ?? ""}`,
23
24
  "Content-Type": "application/json"
24
25
  },
25
26
  body: JSON.stringify({
package/dist/logger.d.ts CHANGED
@@ -1,5 +1,9 @@
1
1
  export interface PeonLoggerOptions {
2
2
  logDir?: string;
3
+ /** Rotate the live log once it exceeds this many bytes. */
4
+ maxBytes?: number;
5
+ /** Upper bound on how much of the log tail recent() will read. */
6
+ tailBytes?: number;
3
7
  }
4
8
  export interface PeonLogEntry {
5
9
  id: string;
@@ -9,9 +13,25 @@ export interface PeonLogEntry {
9
13
  }
10
14
  export declare class PeonLogger {
11
15
  private readonly logFile;
16
+ private readonly maxBytes;
17
+ private readonly tailBytes;
12
18
  private writeQueue;
19
+ private bytesWritten;
20
+ private sizeKnown;
13
21
  constructor(options?: PeonLoggerOptions);
14
22
  log(type: string, fields?: Record<string, unknown>): Promise<PeonLogEntry>;
23
+ /**
24
+ * Newest-first entries from the tail of the log. Cost is bounded by
25
+ * `tailBytes`, not by the size of the file, so this stays flat as the log
26
+ * grows. Entries older than the tail window are not visible here — the log
27
+ * file itself (and its rotated siblings) remain the full record.
28
+ */
15
29
  recent(limit?: number): Promise<PeonLogEntry[]>;
30
+ private readTail;
31
+ /**
32
+ * Move the live log aside once it exceeds maxBytes. History is preserved in a
33
+ * timestamped sibling rather than truncated, so nothing is lost.
34
+ */
35
+ private rotateIfNeeded;
16
36
  private enqueueWrite;
17
37
  }