gitnexus 1.6.10-rc.65 → 1.6.10-rc.67
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md
CHANGED
|
@@ -284,6 +284,7 @@ Set these env vars to use a remote OpenAI-compatible `/v1/embeddings` endpoint i
|
|
|
284
284
|
export GITNEXUS_EMBEDDING_URL=http://your-server:8080/v1
|
|
285
285
|
export GITNEXUS_EMBEDDING_MODEL=BAAI/bge-large-en-v1.5
|
|
286
286
|
export GITNEXUS_EMBEDDING_DIMS=1024 # optional, default 384
|
|
287
|
+
export GITNEXUS_EMBEDDING_REQUEST_DIMS=omit # optional: omit "dimensions", or an integer to override it
|
|
287
288
|
export GITNEXUS_EMBEDDING_API_KEY=your-key # optional, default: "unused"
|
|
288
289
|
export GITNEXUS_EMBEDDING_MAX_ATTEMPTS=3 # optional, total attempts (1-20)
|
|
289
290
|
export GITNEXUS_EMBEDDING_RETRY_CAP_MS=5000 # optional, maximum retry delay
|
|
@@ -291,6 +292,15 @@ export GITNEXUS_EMBEDDING_MIN_INTERVAL_MS=0 # optional, minimum request spacing
|
|
|
291
292
|
gitnexus analyze . --embeddings
|
|
292
293
|
```
|
|
293
294
|
|
|
295
|
+
`GITNEXUS_EMBEDDING_REQUEST_DIMS` controls only the `dimensions` field sent in
|
|
296
|
+
the request body, independently of `GITNEXUS_EMBEDDING_DIMS` (which still
|
|
297
|
+
validates the returned vector's length):
|
|
298
|
+
|
|
299
|
+
- `omit` (or `none`, `off`, `false`, `0`) — do not send `dimensions` at all, for
|
|
300
|
+
strict backends that return the right vector size but reject the field.
|
|
301
|
+
- a positive integer — send that value instead of `GITNEXUS_EMBEDDING_DIMS`.
|
|
302
|
+
- unset — send `GITNEXUS_EMBEDDING_DIMS` (the previous behavior).
|
|
303
|
+
|
|
294
304
|
Works with Infinity, vLLM, TEI, llama.cpp, Ollama, LM Studio, or OpenAI. Retry and pacing settings are provider-neutral; provider-specific limits should be supplied through configuration. When unset, local embeddings are used unchanged.
|
|
295
305
|
|
|
296
306
|
## Multi-Repo Support
|
|
@@ -85,18 +85,24 @@ const paceHttpRequest = async (minIntervalMs, signal) => {
|
|
|
85
85
|
await waitTurn;
|
|
86
86
|
};
|
|
87
87
|
/**
|
|
88
|
-
* Stable lead of
|
|
89
|
-
*
|
|
90
|
-
*
|
|
91
|
-
*
|
|
92
|
-
*
|
|
88
|
+
* Stable lead of a {@link readConfig} malformed dims-env error. `readConfig`
|
|
89
|
+
* throws a plain `Error` (not an {@link HttpEmbeddingError}) for a malformed
|
|
90
|
+
* `GITNEXUS_EMBEDDING_DIMS` or `GITNEXUS_EMBEDDING_REQUEST_DIMS` because it's a
|
|
91
|
+
* *config* mistake, not an endpoint failure — so the CLI recognizes it by this
|
|
92
|
+
* lead ({@link isHttpEmbeddingDimsError}) and prints a clean config message
|
|
93
|
+
* instead of a raw stack dump. Each var names itself so the message points the
|
|
94
|
+
* operator at the variable they actually set, not a sibling. See #2385.
|
|
93
95
|
*/
|
|
94
|
-
const
|
|
96
|
+
const dimsEnvErrorLead = (name) => `${name} must be a positive integer`;
|
|
97
|
+
const EMBEDDING_DIMS_ENV_ERROR_LEAD = dimsEnvErrorLead('GITNEXUS_EMBEDDING_DIMS');
|
|
98
|
+
const EMBEDDING_REQUEST_DIMS_ENV_ERROR_LEAD = dimsEnvErrorLead('GITNEXUS_EMBEDDING_REQUEST_DIMS');
|
|
95
99
|
/**
|
|
96
|
-
* @internal Exported for the CLI analyze error handler. True when `message` is
|
|
97
|
-
*
|
|
100
|
+
* @internal Exported for the CLI analyze error handler. True when `message` is a
|
|
101
|
+
* {@link readConfig} malformed dims-env config error (a plain `Error`) — for
|
|
102
|
+
* either `GITNEXUS_EMBEDDING_DIMS` or `GITNEXUS_EMBEDDING_REQUEST_DIMS`.
|
|
98
103
|
*/
|
|
99
|
-
export const isHttpEmbeddingDimsError = (message) => message.includes(EMBEDDING_DIMS_ENV_ERROR_LEAD)
|
|
104
|
+
export const isHttpEmbeddingDimsError = (message) => message.includes(EMBEDDING_DIMS_ENV_ERROR_LEAD) ||
|
|
105
|
+
message.includes(EMBEDDING_REQUEST_DIMS_ENV_ERROR_LEAD);
|
|
100
106
|
/**
|
|
101
107
|
* Build config from the current process.env snapshot.
|
|
102
108
|
* Returns null when GITNEXUS_EMBEDDING_URL + GITNEXUS_EMBEDDING_MODEL are unset.
|
|
@@ -122,6 +128,23 @@ const readConfig = () => {
|
|
|
122
128
|
}
|
|
123
129
|
dimensions = parsed;
|
|
124
130
|
}
|
|
131
|
+
const rawRequestDims = process.env.GITNEXUS_EMBEDDING_REQUEST_DIMS?.trim();
|
|
132
|
+
let requestDimensions = dimensions;
|
|
133
|
+
if (rawRequestDims) {
|
|
134
|
+
if (/^(omit|none|off|false|0)$/i.test(rawRequestDims)) {
|
|
135
|
+
requestDimensions = undefined;
|
|
136
|
+
}
|
|
137
|
+
else {
|
|
138
|
+
if (!/^\d+$/.test(rawRequestDims)) {
|
|
139
|
+
throw new Error(`${EMBEDDING_REQUEST_DIMS_ENV_ERROR_LEAD}, got "${rawRequestDims}"`);
|
|
140
|
+
}
|
|
141
|
+
const parsed = parseInt(rawRequestDims, 10);
|
|
142
|
+
if (parsed <= 0) {
|
|
143
|
+
throw new Error(`${EMBEDDING_REQUEST_DIMS_ENV_ERROR_LEAD}, got "${rawRequestDims}"`);
|
|
144
|
+
}
|
|
145
|
+
requestDimensions = parsed;
|
|
146
|
+
}
|
|
147
|
+
}
|
|
125
148
|
return {
|
|
126
149
|
baseUrl: baseUrl.replace(/\/+$/, ''),
|
|
127
150
|
model,
|
|
@@ -130,6 +153,7 @@ const readConfig = () => {
|
|
|
130
153
|
maxAttempts: parsePositiveIntegerEnv('GITNEXUS_EMBEDDING_MAX_ATTEMPTS', HTTP_MAX_RETRIES + 1, 20),
|
|
131
154
|
retryCapMs: parsePositiveIntegerEnv('GITNEXUS_EMBEDDING_RETRY_CAP_MS', HTTP_RETRY_CAP_MS, 300_000),
|
|
132
155
|
minIntervalMs: parseNonNegativeIntegerEnv('GITNEXUS_EMBEDDING_MIN_INTERVAL_MS', 0, 300_000),
|
|
156
|
+
requestDimensions,
|
|
133
157
|
};
|
|
134
158
|
};
|
|
135
159
|
/**
|
|
@@ -236,9 +260,9 @@ const isEmbeddingItem = (item) => typeof item === 'object' &&
|
|
|
236
260
|
* the `dimensions` field in the request body. Endpoints that implement
|
|
237
261
|
* Matryoshka truncation (OpenAI text-embedding-3-*, Cohere embed-v3,
|
|
238
262
|
* Voyage) return a truncated vector at that size; endpoints that do not
|
|
239
|
-
* recognise the field may ignore it or return 400.
|
|
240
|
-
* `
|
|
241
|
-
*
|
|
263
|
+
* recognise the field may ignore it or return 400. Set
|
|
264
|
+
* `GITNEXUS_EMBEDDING_REQUEST_DIMS=omit` for strict backends while keeping
|
|
265
|
+
* `GITNEXUS_EMBEDDING_DIMS` set to the returned vector size.
|
|
242
266
|
*/
|
|
243
267
|
const httpEmbedBatch = async (url, batch, model, apiKey, batchIndex = 0, dimensions, requestOptions = {}, maxAttempts = HTTP_MAX_RETRIES + 1, retryCapMs = HTTP_RETRY_CAP_MS, minIntervalMs = 0) => {
|
|
244
268
|
const requestBody = {
|
|
@@ -336,7 +360,7 @@ export const httpEmbed = async (texts, requestOptions = {}) => {
|
|
|
336
360
|
for (let i = 0; i < texts.length; i += HTTP_BATCH_SIZE) {
|
|
337
361
|
const batch = texts.slice(i, i + HTTP_BATCH_SIZE);
|
|
338
362
|
const batchIndex = Math.floor(i / HTTP_BATCH_SIZE);
|
|
339
|
-
const items = await httpEmbedBatch(url, batch, config.model, config.apiKey, batchIndex, config.
|
|
363
|
+
const items = await httpEmbedBatch(url, batch, config.model, config.apiKey, batchIndex, config.requestDimensions, requestOptions, config.maxAttempts, config.retryCapMs, config.minIntervalMs);
|
|
340
364
|
if (items.length !== batch.length) {
|
|
341
365
|
throw new HttpEmbeddingError(`Embedding endpoint returned ${items.length} vectors for ${batch.length} texts ` +
|
|
342
366
|
`(${safeUrl(url)}, batch ${batchIndex})`);
|
|
@@ -370,7 +394,7 @@ export const httpEmbedQuery = async (text, requestOptions = {}) => {
|
|
|
370
394
|
if (!config)
|
|
371
395
|
throw new Error('HTTP embedding not configured');
|
|
372
396
|
const url = `${config.baseUrl}/embeddings`;
|
|
373
|
-
const items = await httpEmbedBatch(url, [text], config.model, config.apiKey, 0, config.
|
|
397
|
+
const items = await httpEmbedBatch(url, [text], config.model, config.apiKey, 0, config.requestDimensions, requestOptions, config.maxAttempts, config.retryCapMs, config.minIntervalMs);
|
|
374
398
|
if (!items.length) {
|
|
375
399
|
throw new HttpEmbeddingError(`Embedding endpoint returned empty response (${safeUrl(url)})`);
|
|
376
400
|
}
|
|
@@ -13,4 +13,12 @@ import { type ProcessDetectionResult } from '../process-processor.js';
|
|
|
13
13
|
export interface ProcessesOutput {
|
|
14
14
|
processResult: ProcessDetectionResult;
|
|
15
15
|
}
|
|
16
|
+
/**
|
|
17
|
+
* Compute the dynamic max-processes budget from the symbol count.
|
|
18
|
+
*
|
|
19
|
+
* Scales proportionally (symbolCount / 10) with a floor of 20.
|
|
20
|
+
* Prior to #2198 this was capped at 300 via `Math.min(300, …)`,
|
|
21
|
+
* silently truncating process detection on large repositories.
|
|
22
|
+
*/
|
|
23
|
+
export declare function computeDynamicMaxProcesses(symbolCount: number): number;
|
|
16
24
|
export declare const processesPhase: PipelinePhase<ProcessesOutput>;
|
|
@@ -14,6 +14,16 @@ import { generateId } from '../../../lib/utils.js';
|
|
|
14
14
|
import { routeNodeKey } from '../route-extractors/route-path.js';
|
|
15
15
|
import { isDev } from '../utils/env.js';
|
|
16
16
|
import { logger } from '../../logger.js';
|
|
17
|
+
/**
|
|
18
|
+
* Compute the dynamic max-processes budget from the symbol count.
|
|
19
|
+
*
|
|
20
|
+
* Scales proportionally (symbolCount / 10) with a floor of 20.
|
|
21
|
+
* Prior to #2198 this was capped at 300 via `Math.min(300, …)`,
|
|
22
|
+
* silently truncating process detection on large repositories.
|
|
23
|
+
*/
|
|
24
|
+
export function computeDynamicMaxProcesses(symbolCount) {
|
|
25
|
+
return Math.max(20, Math.round(symbolCount / 10));
|
|
26
|
+
}
|
|
17
27
|
export const processesPhase = {
|
|
18
28
|
name: 'processes',
|
|
19
29
|
// `structure` supplies `totalFiles` (progress counter) without the spurious
|
|
@@ -37,7 +47,7 @@ export const processesPhase = {
|
|
|
37
47
|
if (n.label !== 'File')
|
|
38
48
|
symbolCount++;
|
|
39
49
|
});
|
|
40
|
-
const dynamicMaxProcesses =
|
|
50
|
+
const dynamicMaxProcesses = computeDynamicMaxProcesses(symbolCount);
|
|
41
51
|
const processResult = await processProcesses(ctx.graph, communityResult.memberships, (message, progress) => {
|
|
42
52
|
const processProgress = 99 + progress * 0.01;
|
|
43
53
|
ctx.onProgress({
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "gitnexus",
|
|
3
|
-
"version": "1.6.10-rc.
|
|
3
|
+
"version": "1.6.10-rc.67",
|
|
4
4
|
"description": "Graph-powered code intelligence for AI agents. Index any codebase, query via MCP or CLI.",
|
|
5
5
|
"author": "Abhigyan Patwari",
|
|
6
6
|
"license": "PolyForm-Noncommercial-1.0.0",
|
|
@@ -70,7 +70,7 @@
|
|
|
70
70
|
"graphology-indices": "^0.17.0",
|
|
71
71
|
"graphology-utils": "^2.3.0",
|
|
72
72
|
"ignore": "^7.0.5",
|
|
73
|
-
"js-yaml": "^
|
|
73
|
+
"js-yaml": "^4.1.1",
|
|
74
74
|
"jsonc-parser": "^3.3.1",
|
|
75
75
|
"mnemonist": "^0.40.3",
|
|
76
76
|
"node-addon-api": "^8.0.0",
|
|
@@ -106,7 +106,7 @@
|
|
|
106
106
|
"@types/cors": "^2.8.17",
|
|
107
107
|
"@types/express": "^5.0.6",
|
|
108
108
|
"@types/js-yaml": "^4.0.9",
|
|
109
|
-
"@types/node": "^
|
|
109
|
+
"@types/node": "^26.0.0",
|
|
110
110
|
"@types/uuid": "^11.0.0",
|
|
111
111
|
"@vitest/coverage-v8": "^4.0.18",
|
|
112
112
|
"gitnexus-shared": "file:../gitnexus-shared",
|