@gmickel/gno 1.34.1 → 1.34.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1 @@
1
+ ee0e3d4032a41a097c85fba552a9ab449906f258e14844ad5e3b5f99001716e0 gno-browser-clipper-v1.34.2.zip
@@ -21,5 +21,5 @@
21
21
  "content_security_policy": {
22
22
  "extension_pages": "script-src 'self'; object-src 'none'; connect-src http://127.0.0.1:*"
23
23
  },
24
- "version": "1.34.1"
24
+ "version": "1.34.2"
25
25
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@gmickel/gno",
3
- "version": "1.34.1",
3
+ "version": "1.34.2",
4
4
  "description": "Local semantic search for your documents. Index Markdown, PDF, and Office files with hybrid BM25 + vector search.",
5
5
  "keywords": [
6
6
  "embeddings",
@@ -60,6 +60,29 @@ const DEFAULT_TEMPERATURE = 0;
60
60
  const DEFAULT_SEED = 42;
61
61
  const DEFAULT_MAX_TOKENS = 256;
62
62
 
63
+ // Context sizing: without an explicit contextSize, node-llama-cpp defaults to
64
+ // "auto", which grows the KV cache to fill available VRAM up to the model's
65
+ // trained context length (OOM risk on small GPUs — see issue #189). Instead,
66
+ // size the context to what the call actually needs: prompt tokens + output
67
+ // budget + margin for chat-template wrapping and special tokens.
68
+ const GEN_CONTEXT_MARGIN_TOKENS = 512;
69
+ const GEN_CONTEXT_MIN_TOKENS = 1024;
70
+
71
+ export const resolveGenContextSize = (input: {
72
+ promptTokenCount: number;
73
+ maxTokens: number;
74
+ trainContextSize?: number;
75
+ }): number => {
76
+ const needed = Math.max(
77
+ GEN_CONTEXT_MIN_TOKENS,
78
+ input.promptTokenCount + input.maxTokens + GEN_CONTEXT_MARGIN_TOKENS
79
+ );
80
+ if (input.trainContextSize && input.trainContextSize > 0) {
81
+ return Math.min(needed, input.trainContextSize);
82
+ }
83
+ return needed;
84
+ };
85
+
63
86
  // ─────────────────────────────────────────────────────────────────────────────
64
87
  // Implementation
65
88
  // ─────────────────────────────────────────────────────────────────────────────
@@ -97,9 +120,14 @@ export class NodeLlamaCppGeneration implements GenerationPort {
97
120
  await this.manager.getLlama()
98
121
  ).createGrammarForJsonSchema(params.jsonSchema as JsonGrammarSchema)
99
122
  : undefined;
100
- context = await llamaModel.createContext(
101
- params?.contextSize ? { contextSize: params.contextSize } : undefined
102
- );
123
+ const contextSize =
124
+ params?.contextSize ??
125
+ resolveGenContextSize({
126
+ promptTokenCount: llamaModel.tokenize(prompt).length,
127
+ maxTokens: params?.maxTokens ?? DEFAULT_MAX_TOKENS,
128
+ trainContextSize: llamaModel.trainContextSize,
129
+ });
130
+ context = await llamaModel.createContext({ contextSize });
103
131
  // Import LlamaChatSession dynamically
104
132
  const { LlamaChatSession } = await import("node-llama-cpp");
105
133
  const session = new LlamaChatSession({
@@ -1 +0,0 @@
1
- 72e70b96bc421db0d18489837817d4ed147bb57888f087bf13afb4cfb814cdb9 gno-browser-clipper-v1.34.1.zip