archgraph-argo 0.24.0 → 0.24.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +6 -3
- package/argo/scripts/graph-rag/defaultSemanticRetrieval.js +17 -5
- package/argo/scripts/graph-rag/embeddingProviderProfile.js +89 -89
- package/argo/scripts/graph-rag/liveEmbeddingProviderConfig.js +23 -0
- package/argo/scripts/graph-rag/semanticInitConfiguration.js +24 -0
- package/argo/scripts/systemarchitecture-mcp-server.js +12 -54
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -57,12 +57,15 @@ Everything works out of the box except **semantic (Graph RAG) queries**, which n
|
|
|
57
57
|
- **Embedding / vector engine** — powers semantic Graph RAG retrieval. Configure
|
|
58
58
|
`ARGO_EMBEDDING_BASE_URL`, `ARGO_EMBEDDING_MODEL`, `ARGO_EMBEDDING_PROVIDER`,
|
|
59
59
|
`ARGO_EMBEDDING_MODEL_VERSION`, `ARGO_EMBEDDING_DIMENSIONS`, plus the API key `QWEN_KEY`.
|
|
60
|
+
It points at **any OpenAI-compatible embedding endpoint** — a cloud provider, or a self-hosted
|
|
61
|
+
server for offline / intranet / private deployments via `ARGO_EMBEDDING_PROFILE=openai-compatible`
|
|
62
|
+
(see the [self-hosted embedding guide](docs/self-hosted-embedding-deployment.md)).
|
|
60
63
|
|
|
61
64
|
Where do the values come from? The Neo4j credentials come from the Neo4j instance you own or
|
|
62
65
|
provision (URI, username, password). The embedding configuration and `QWEN_KEY` come from your
|
|
63
|
-
embedding provider's dashboard — for example Alibaba DashScope
|
|
64
|
-
|
|
65
|
-
and re-run.
|
|
66
|
+
embedding provider's dashboard — for example Alibaba DashScope — or from a self-hosted
|
|
67
|
+
OpenAI-compatible server. `argo-deploy` walks you through the prompt (existing non-empty values in
|
|
68
|
+
`~/.argo/.env` are kept); you can also edit the file afterwards and re-run.
|
|
66
69
|
|
|
67
70
|
## How to use
|
|
68
71
|
|
|
@@ -85,32 +85,35 @@ const CHANNELS = Object.freeze([
|
|
|
85
85
|
indexName: 'argo_production_semantic_view_vector',
|
|
86
86
|
}),
|
|
87
87
|
]);
|
|
88
|
+
// Project an explicit metadata allowlist instead of properties(node): Neo4j map
|
|
89
|
+
// projection `.*` lets an existing property win, so properties(node) / node{.*, vector:null}
|
|
90
|
+
// would still ship the 1536-dim vector; the read path only consumes canonicalIdentity, channel, searchText and score.
|
|
88
91
|
const VECTOR_QUERY_CYPHER = [
|
|
89
92
|
'CALL db.index.vector.queryNodes($indexName, $topK, $vector)',
|
|
90
93
|
'YIELD node, score',
|
|
91
94
|
'WHERE node.channel = $channel',
|
|
92
|
-
'RETURN
|
|
95
|
+
'RETURN {canonicalIdentity: node.canonicalIdentity, channel: node.channel, searchText: node.searchText, dimensions: node.dimensions, provider: node.provider, model: node.model, modelVersion: node.modelVersion, canonicalVersion: node.canonicalVersion, contentVersion: node.contentVersion, indexVersion: node.indexVersion} AS record, score',
|
|
93
96
|
'ORDER BY score DESC',
|
|
94
97
|
].join('\n');
|
|
95
98
|
const VECTOR_QUERY_CYPHER_SCOPED = [
|
|
96
99
|
'CALL db.index.vector.queryNodes($indexName, $topK, $vector)',
|
|
97
100
|
'YIELD node, score',
|
|
98
101
|
'WHERE node.channel = $channel AND node.canonicalIdentity IN $canonicalIdentities',
|
|
99
|
-
'RETURN
|
|
102
|
+
'RETURN {canonicalIdentity: node.canonicalIdentity, channel: node.channel, searchText: node.searchText, dimensions: node.dimensions, provider: node.provider, model: node.model, modelVersion: node.modelVersion, canonicalVersion: node.canonicalVersion, contentVersion: node.contentVersion, indexVersion: node.indexVersion} AS record, score',
|
|
100
103
|
'ORDER BY score DESC',
|
|
101
104
|
].join('\n');
|
|
102
105
|
const LEXICAL_QUERY_CYPHER = [
|
|
103
106
|
'CALL db.index.fulltext.queryNodes($indexName, $queryText, { limit: $topK })',
|
|
104
107
|
'YIELD node, score',
|
|
105
108
|
'WHERE node.channel = $channel',
|
|
106
|
-
'RETURN
|
|
109
|
+
'RETURN {canonicalIdentity: node.canonicalIdentity, channel: node.channel, searchText: node.searchText, dimensions: node.dimensions, provider: node.provider, model: node.model, modelVersion: node.modelVersion, canonicalVersion: node.canonicalVersion, contentVersion: node.contentVersion, indexVersion: node.indexVersion} AS record, score',
|
|
107
110
|
'ORDER BY score DESC',
|
|
108
111
|
].join('\n');
|
|
109
112
|
const LEXICAL_QUERY_CYPHER_SCOPED = [
|
|
110
113
|
'CALL db.index.fulltext.queryNodes($indexName, $queryText, { limit: $topK })',
|
|
111
114
|
'YIELD node, score',
|
|
112
115
|
'WHERE node.channel = $channel AND node.canonicalIdentity IN $canonicalIdentities',
|
|
113
|
-
'RETURN
|
|
116
|
+
'RETURN {canonicalIdentity: node.canonicalIdentity, channel: node.channel, searchText: node.searchText, dimensions: node.dimensions, provider: node.provider, model: node.model, modelVersion: node.modelVersion, canonicalVersion: node.canonicalVersion, contentVersion: node.contentVersion, indexVersion: node.indexVersion} AS record, score',
|
|
114
117
|
'ORDER BY score DESC',
|
|
115
118
|
].join('\n');
|
|
116
119
|
const READINESS_QUERY_CYPHER = [
|
|
@@ -149,6 +152,12 @@ function resolveTopK() {
|
|
|
149
152
|
const k = envNumber('ARGO_SEMANTIC_TOP_K', DEFAULT_TOP_K);
|
|
150
153
|
return Number.isInteger(k) && k > 0 ? k : DEFAULT_TOP_K;
|
|
151
154
|
}
|
|
155
|
+
// The single production seed gate: a retrieved record becomes a seed only when
|
|
156
|
+
// its TRUE vector similarity reaches the purpose-aware threshold. Exposed so
|
|
157
|
+
// acceptance tests exercise the exact predicate the retrieval runtime uses.
|
|
158
|
+
function isSeedAboveThreshold(score, threshold) {
|
|
159
|
+
return Number.isFinite(score) && Number.isFinite(threshold) && score >= threshold;
|
|
160
|
+
}
|
|
152
161
|
const testCompositionStorage = new AsyncLocalStorage();
|
|
153
162
|
|
|
154
163
|
function createDefaultSemanticRetrieval(dependencies = {}) {
|
|
@@ -828,7 +837,7 @@ async function exhaustChannel({
|
|
|
828
837
|
for (const raw of newlyVisible) {
|
|
829
838
|
if (accepted.length >= effectiveMax) break;
|
|
830
839
|
const record = normalizeVectorRecord(raw, channel);
|
|
831
|
-
if (record && record.score
|
|
840
|
+
if (record && isSeedAboveThreshold(record.score, effectiveThreshold) && !seen.has(record.id)) {
|
|
832
841
|
seen.add(record.id);
|
|
833
842
|
accepted.push(Object.freeze(record));
|
|
834
843
|
}
|
|
@@ -1238,8 +1247,11 @@ module.exports = {
|
|
|
1238
1247
|
memoryThresholdFor,
|
|
1239
1248
|
auditThresholdFor,
|
|
1240
1249
|
resolveTopK,
|
|
1250
|
+
isSeedAboveThreshold,
|
|
1241
1251
|
scopeCanonicalIdentitiesForChannel,
|
|
1242
1252
|
bareCanonicalId,
|
|
1243
1253
|
normalizeVectorRecord,
|
|
1254
|
+
exhaustChannel,
|
|
1255
|
+
exhaustLexicalChannel,
|
|
1244
1256
|
AUDIT_PURPOSES,
|
|
1245
1257
|
};
|
|
@@ -1,89 +1,89 @@
|
|
|
1
|
-
'use strict';
|
|
2
|
-
|
|
3
|
-
// Embedding provider profile (ArchGraph ARGO).
|
|
4
|
-
//
|
|
5
|
-
// The semantic-retrieval embedding backend is either the human-approved cloud
|
|
6
|
-
// profile (`approved`, the default, byte-for-byte unchanged) or a self-hosted
|
|
7
|
-
// OpenAI-compatible endpoint (`openai-compatible`) for intranet / offline
|
|
8
|
-
// deployments. This module owns the profile vocabulary, the dimension contract,
|
|
9
|
-
// and the query-side instruction composition so the live configuration resolver
|
|
10
|
-
// and the retrieval runtime share one implementation.
|
|
11
|
-
|
|
12
|
-
const PROFILE_KEY = 'ARGO_EMBEDDING_PROFILE';
|
|
13
|
-
const QUERY_INSTRUCTION_KEY = 'ARGO_EMBEDDING_QUERY_INSTRUCTION';
|
|
14
|
-
const API_KEY = 'ARGO_EMBEDDING_API_KEY';
|
|
15
|
-
const PROFILE_APPROVED = 'approved';
|
|
16
|
-
const PROFILE_OPENAI_COMPATIBLE = 'openai-compatible';
|
|
17
|
-
const SUPPORTED_EMBEDDING_PROFILES = Object.freeze([PROFILE_APPROVED, PROFILE_OPENAI_COMPATIBLE]);
|
|
18
|
-
const DEFAULT_EMBEDDING_DIMENSIONS = 1536;
|
|
19
|
-
|
|
20
|
-
function resolveEmbeddingProfile(value) {
|
|
21
|
-
const normalized = typeof value === 'string' ? value.trim().toLowerCase() : '';
|
|
22
|
-
return normalized === '' ? PROFILE_APPROVED : normalized;
|
|
23
|
-
}
|
|
24
|
-
|
|
25
|
-
function isApprovedEmbeddingProfile(value) {
|
|
26
|
-
return resolveEmbeddingProfile(value) === PROFILE_APPROVED;
|
|
27
|
-
}
|
|
28
|
-
|
|
29
|
-
function requireSupportedEmbeddingProfile(value) {
|
|
30
|
-
const profile = resolveEmbeddingProfile(value);
|
|
31
|
-
if (!SUPPORTED_EMBEDDING_PROFILES.includes(profile)) {
|
|
32
|
-
throw profileError('EMBEDDING_PROFILE_UNSUPPORTED');
|
|
33
|
-
}
|
|
34
|
-
return profile;
|
|
35
|
-
}
|
|
36
|
-
|
|
37
|
-
function resolveEmbeddingDimensions(value, fallback = DEFAULT_EMBEDDING_DIMENSIONS) {
|
|
38
|
-
const parsed = typeof value === 'string' && value.trim() !== '' ? Number(value) : NaN;
|
|
39
|
-
return Number.isInteger(parsed) && parsed > 0 ? parsed : fallback;
|
|
40
|
-
}
|
|
41
|
-
|
|
42
|
-
function requireEmbeddingDimensions(value) {
|
|
43
|
-
const parsed = typeof value === 'string' && value.trim() !== '' ? Number(value) : NaN;
|
|
44
|
-
if (!Number.isInteger(parsed) || parsed <= 0) {
|
|
45
|
-
throw profileError('EMBEDDING_DIMENSIONS_INVALID');
|
|
46
|
-
}
|
|
47
|
-
return parsed;
|
|
48
|
-
}
|
|
49
|
-
|
|
50
|
-
// The query-side input may carry an instruction prefix (e.g. gte-Qwen2's
|
|
51
|
-
// "Instruct: <task>\nQuery: <query>"); document-side input stays raw so stored
|
|
52
|
-
// vectors are never polluted by a query-only prefix. The prefix commonly comes
|
|
53
|
-
// from a single-line `.env` value, so `\n`/`\r`/`\t` escapes are decoded here.
|
|
54
|
-
function normalizeInstruction(instruction) {
|
|
55
|
-
if (typeof instruction !== 'string') return '';
|
|
56
|
-
return instruction
|
|
57
|
-
.replace(/\\r\\n/g, '\n')
|
|
58
|
-
.replace(/\\n/g, '\n')
|
|
59
|
-
.replace(/\\r/g, '\r')
|
|
60
|
-
.replace(/\\t/g, '\t');
|
|
61
|
-
}
|
|
62
|
-
|
|
63
|
-
function composeQueryEmbeddingInput(text, instruction) {
|
|
64
|
-
const query = text == null ? '' : String(text);
|
|
65
|
-
const prefix = normalizeInstruction(instruction);
|
|
66
|
-
return prefix === '' ? query : `${prefix}${query}`;
|
|
67
|
-
}
|
|
68
|
-
|
|
69
|
-
function profileError(category) {
|
|
70
|
-
const error = new Error(category);
|
|
71
|
-
error.category = category;
|
|
72
|
-
return error;
|
|
73
|
-
}
|
|
74
|
-
|
|
75
|
-
module.exports = {
|
|
76
|
-
PROFILE_KEY,
|
|
77
|
-
QUERY_INSTRUCTION_KEY,
|
|
78
|
-
API_KEY,
|
|
79
|
-
PROFILE_APPROVED,
|
|
80
|
-
PROFILE_OPENAI_COMPATIBLE,
|
|
81
|
-
SUPPORTED_EMBEDDING_PROFILES,
|
|
82
|
-
DEFAULT_EMBEDDING_DIMENSIONS,
|
|
83
|
-
resolveEmbeddingProfile,
|
|
84
|
-
isApprovedEmbeddingProfile,
|
|
85
|
-
requireSupportedEmbeddingProfile,
|
|
86
|
-
resolveEmbeddingDimensions,
|
|
87
|
-
requireEmbeddingDimensions,
|
|
88
|
-
composeQueryEmbeddingInput,
|
|
89
|
-
};
|
|
1
|
+
'use strict';
|
|
2
|
+
|
|
3
|
+
// Embedding provider profile (ArchGraph ARGO).
|
|
4
|
+
//
|
|
5
|
+
// The semantic-retrieval embedding backend is either the human-approved cloud
|
|
6
|
+
// profile (`approved`, the default, byte-for-byte unchanged) or a self-hosted
|
|
7
|
+
// OpenAI-compatible endpoint (`openai-compatible`) for intranet / offline
|
|
8
|
+
// deployments. This module owns the profile vocabulary, the dimension contract,
|
|
9
|
+
// and the query-side instruction composition so the live configuration resolver
|
|
10
|
+
// and the retrieval runtime share one implementation.
|
|
11
|
+
|
|
12
|
+
const PROFILE_KEY = 'ARGO_EMBEDDING_PROFILE';
|
|
13
|
+
const QUERY_INSTRUCTION_KEY = 'ARGO_EMBEDDING_QUERY_INSTRUCTION';
|
|
14
|
+
const API_KEY = 'ARGO_EMBEDDING_API_KEY';
|
|
15
|
+
const PROFILE_APPROVED = 'approved';
|
|
16
|
+
const PROFILE_OPENAI_COMPATIBLE = 'openai-compatible';
|
|
17
|
+
const SUPPORTED_EMBEDDING_PROFILES = Object.freeze([PROFILE_APPROVED, PROFILE_OPENAI_COMPATIBLE]);
|
|
18
|
+
const DEFAULT_EMBEDDING_DIMENSIONS = 1536;
|
|
19
|
+
|
|
20
|
+
function resolveEmbeddingProfile(value) {
|
|
21
|
+
const normalized = typeof value === 'string' ? value.trim().toLowerCase() : '';
|
|
22
|
+
return normalized === '' ? PROFILE_APPROVED : normalized;
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
function isApprovedEmbeddingProfile(value) {
|
|
26
|
+
return resolveEmbeddingProfile(value) === PROFILE_APPROVED;
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
function requireSupportedEmbeddingProfile(value) {
|
|
30
|
+
const profile = resolveEmbeddingProfile(value);
|
|
31
|
+
if (!SUPPORTED_EMBEDDING_PROFILES.includes(profile)) {
|
|
32
|
+
throw profileError('EMBEDDING_PROFILE_UNSUPPORTED');
|
|
33
|
+
}
|
|
34
|
+
return profile;
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
function resolveEmbeddingDimensions(value, fallback = DEFAULT_EMBEDDING_DIMENSIONS) {
|
|
38
|
+
const parsed = typeof value === 'string' && value.trim() !== '' ? Number(value) : NaN;
|
|
39
|
+
return Number.isInteger(parsed) && parsed > 0 ? parsed : fallback;
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
function requireEmbeddingDimensions(value) {
|
|
43
|
+
const parsed = typeof value === 'string' && value.trim() !== '' ? Number(value) : NaN;
|
|
44
|
+
if (!Number.isInteger(parsed) || parsed <= 0) {
|
|
45
|
+
throw profileError('EMBEDDING_DIMENSIONS_INVALID');
|
|
46
|
+
}
|
|
47
|
+
return parsed;
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
// The query-side input may carry an instruction prefix (e.g. gte-Qwen2's
|
|
51
|
+
// "Instruct: <task>\nQuery: <query>"); document-side input stays raw so stored
|
|
52
|
+
// vectors are never polluted by a query-only prefix. The prefix commonly comes
|
|
53
|
+
// from a single-line `.env` value, so `\n`/`\r`/`\t` escapes are decoded here.
|
|
54
|
+
function normalizeInstruction(instruction) {
|
|
55
|
+
if (typeof instruction !== 'string') return '';
|
|
56
|
+
return instruction
|
|
57
|
+
.replace(/\\r\\n/g, '\n')
|
|
58
|
+
.replace(/\\n/g, '\n')
|
|
59
|
+
.replace(/\\r/g, '\r')
|
|
60
|
+
.replace(/\\t/g, '\t');
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
function composeQueryEmbeddingInput(text, instruction) {
|
|
64
|
+
const query = text == null ? '' : String(text);
|
|
65
|
+
const prefix = normalizeInstruction(instruction);
|
|
66
|
+
return prefix === '' ? query : `${prefix}${query}`;
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
function profileError(category) {
|
|
70
|
+
const error = new Error(category);
|
|
71
|
+
error.category = category;
|
|
72
|
+
return error;
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
module.exports = {
|
|
76
|
+
PROFILE_KEY,
|
|
77
|
+
QUERY_INSTRUCTION_KEY,
|
|
78
|
+
API_KEY,
|
|
79
|
+
PROFILE_APPROVED,
|
|
80
|
+
PROFILE_OPENAI_COMPATIBLE,
|
|
81
|
+
SUPPORTED_EMBEDDING_PROFILES,
|
|
82
|
+
DEFAULT_EMBEDDING_DIMENSIONS,
|
|
83
|
+
resolveEmbeddingProfile,
|
|
84
|
+
isApprovedEmbeddingProfile,
|
|
85
|
+
requireSupportedEmbeddingProfile,
|
|
86
|
+
resolveEmbeddingDimensions,
|
|
87
|
+
requireEmbeddingDimensions,
|
|
88
|
+
composeQueryEmbeddingInput,
|
|
89
|
+
};
|
|
@@ -77,6 +77,27 @@ const READABLE_KEYS = Object.freeze([
|
|
|
77
77
|
...Object.keys(OPT_IN_KEYS),
|
|
78
78
|
]);
|
|
79
79
|
const LEGACY_KEYS = Object.freeze(['ARGO_NEO4J_URI', 'ARGO_NEO4J_USERNAME', 'ARGO_NEO4J_PASSWORD']);
|
|
80
|
+
// Host / process-level keys: read from the environment at launch, but NOT
|
|
81
|
+
// accepted inside the .env file (see argo/.env.example Part 2). Kept explicit so
|
|
82
|
+
// the coverage test (tests/env-key-coverage.test.js) can prove every env key
|
|
83
|
+
// referenced by argo/scripts is classified — either a .env key, a host-only key,
|
|
84
|
+
// or a legacy alias. This is what prevents "a key the code already uses is still
|
|
85
|
+
// rejected as unknown".
|
|
86
|
+
const HOST_ONLY_ENV_KEYS = Object.freeze([
|
|
87
|
+
'ARGO_ENV_FILE',
|
|
88
|
+
'ARGO_REPO_ROOT',
|
|
89
|
+
'ARGO_EA_QEA',
|
|
90
|
+
'ARGO_WORKSPACE_ROOTS',
|
|
91
|
+
'ARGO_SERVER_PATH',
|
|
92
|
+
'GRAPH_MCP_URL',
|
|
93
|
+
'EA_QEA_DEBUG',
|
|
94
|
+
'WORKSPACE_FOLDER',
|
|
95
|
+
'ARGO_TEST_TIMEOUT_MS',
|
|
96
|
+
'ARGO_MCP_MUTATION_RESPONSE_DEBUG',
|
|
97
|
+
'ARGO_MCP_SEMANTIC_DEDUP',
|
|
98
|
+
'ARGO_MCP_SEMANTIC_DEDUP_THRESHOLD',
|
|
99
|
+
'ARGO_SEMANTIC_DEDUP_THRESHOLD',
|
|
100
|
+
]);
|
|
80
101
|
const PROHIBITED_RUNTIME_FIELD_KEYS = Object.freeze(['neo4jUri', 'embeddingCredential']);
|
|
81
102
|
const SECRET_KEYS = new Set(['ARGO_NEO4J_DATABASE_PASSWORD', 'QWEN_KEY', 'ARGO_RERANK_API_KEY', EMBEDDING_API_KEY_KEY]);
|
|
82
103
|
const APPROVED = Object.freeze({
|
|
@@ -600,4 +621,6 @@ module.exports = {
|
|
|
600
621
|
// preflighted as secrets.
|
|
601
622
|
ENV_FILE_KEYS: READABLE_KEYS,
|
|
602
623
|
ENV_FILE_SECRET_KEYS: Object.freeze(Array.from(SECRET_KEYS)),
|
|
624
|
+
HOST_ONLY_ENV_KEYS,
|
|
625
|
+
ENV_FILE_LEGACY_KEYS: LEGACY_KEYS,
|
|
603
626
|
};
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
'use strict';
|
|
2
|
+
|
|
3
|
+
// Maps profile-aware configuration evidence (from resolveApprovedLiveConfiguration)
|
|
4
|
+
// into the shape the canonical semantic-init / backfill runtime consumes.
|
|
5
|
+
//
|
|
6
|
+
// This is the single place that turns "which embedding profile" into the init
|
|
7
|
+
// backfill's configuration, so argo init embeds through the SAME provider the
|
|
8
|
+
// retrieval path uses (approved cloud by default, or a self-hosted endpoint via
|
|
9
|
+
// ARGO_EMBEDDING_PROFILE=openai-compatible) — instead of a hardcoded profile.
|
|
10
|
+
|
|
11
|
+
function buildDefaultSemanticConfiguration(evidence) {
|
|
12
|
+
const configuration = evidence && typeof evidence === 'object' && evidence.configuration
|
|
13
|
+
? evidence.configuration
|
|
14
|
+
: evidence;
|
|
15
|
+
if (!configuration || typeof configuration !== 'object') {
|
|
16
|
+
throw new TypeError('resolved configuration evidence is required');
|
|
17
|
+
}
|
|
18
|
+
return Object.freeze({
|
|
19
|
+
...configuration,
|
|
20
|
+
embeddingCredential: configuration.embeddingApiKey || configuration.qwenKey,
|
|
21
|
+
});
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
module.exports = { buildDefaultSemanticConfiguration };
|
|
@@ -144,12 +144,12 @@ const {
|
|
|
144
144
|
const {
|
|
145
145
|
semanticOperatorErrorResult,
|
|
146
146
|
} = require('./graph-rag/semanticOperatorError.js');
|
|
147
|
-
const {
|
|
148
|
-
resolveExternalProductionConfig,
|
|
149
|
-
} = require('./graph-rag/externalProductionConfig.js');
|
|
150
147
|
const {
|
|
151
148
|
resolveApprovedLiveConfiguration,
|
|
152
149
|
} = require('./graph-rag/liveEmbeddingProviderConfig.js');
|
|
150
|
+
const {
|
|
151
|
+
buildDefaultSemanticConfiguration,
|
|
152
|
+
} = require('./graph-rag/semanticInitConfiguration.js');
|
|
153
153
|
const {
|
|
154
154
|
createLiveEmbeddingProviderClient,
|
|
155
155
|
} = require('./graph-rag/liveEmbeddingProviderClient.js');
|
|
@@ -4056,59 +4056,17 @@ async function createDefaultProductionSemanticRuntime(options = {}) {
|
|
|
4056
4056
|
});
|
|
4057
4057
|
}
|
|
4058
4058
|
|
|
4059
|
+
// Resolve the canonical semantic-init/backfill configuration through the SAME
|
|
4060
|
+
// profile-aware resolver the retrieval path uses, so argo init embeds via the
|
|
4061
|
+
// configured provider (ARGO_EMBEDDING_PROFILE) instead of a hardcoded cloud
|
|
4062
|
+
// profile. `resolveApprovedLiveConfiguration` still fails closed when a required
|
|
4063
|
+
// key is missing or an unknown key is present in the .env file.
|
|
4059
4064
|
async function resolveDefaultSemanticConfiguration(repositoryRoot) {
|
|
4060
|
-
|
|
4061
|
-
|
|
4062
|
-
|
|
4063
|
-
neo4jUri: process.env.ARGO_NEO4J_DATABASE_URL,
|
|
4064
|
-
neo4jUsername: process.env.ARGO_NEO4J_DATABASE_USERNAME,
|
|
4065
|
-
neo4jPassword: process.env.ARGO_NEO4J_DATABASE_PASSWORD,
|
|
4066
|
-
embeddingCredential: process.env.QWEN_KEY,
|
|
4067
|
-
neo4jDatabase: process.env.ARGO_NEO4J_DATABASE || getDefaultSemanticNeo4jDatabaseName(repositoryRoot),
|
|
4068
|
-
}, {
|
|
4069
|
-
operation: 'semantic-backfill',
|
|
4070
|
-
sourceKeys: new Map([
|
|
4071
|
-
['neo4jUri', 'ARGO_NEO4J_DATABASE_URL'],
|
|
4072
|
-
['neo4jUsername', 'ARGO_NEO4J_DATABASE_USERNAME'],
|
|
4073
|
-
['neo4jPassword', 'ARGO_NEO4J_DATABASE_PASSWORD'],
|
|
4074
|
-
['embeddingCredential', 'QWEN_KEY'],
|
|
4075
|
-
]),
|
|
4076
|
-
});
|
|
4077
|
-
} catch (error) {
|
|
4078
|
-
if (error && error.category === 'EXTERNAL_CREDENTIALS_REQUIRED') {
|
|
4079
|
-
const missing = new Error('EXTERNAL_CREDENTIALS_REQUIRED');
|
|
4080
|
-
missing.category = 'EXTERNAL_CREDENTIALS_REQUIRED';
|
|
4081
|
-
missing.field = error.field;
|
|
4082
|
-
throw missing;
|
|
4083
|
-
}
|
|
4084
|
-
throw error;
|
|
4085
|
-
}
|
|
4086
|
-
return Object.freeze({
|
|
4087
|
-
embeddingBaseUrl: W31_APPROVED_PROFILE.baseUrl,
|
|
4088
|
-
embeddingModel: W31_APPROVED_PROFILE.model,
|
|
4089
|
-
embeddingProvider: W31_APPROVED_PROFILE.provider,
|
|
4090
|
-
embeddingModelVersion: W31_APPROVED_PROFILE.version,
|
|
4091
|
-
embeddingDimensions: W31_APPROVED_PROFILE.dimensions,
|
|
4092
|
-
neo4jDatabaseUrl: external.neo4jUri,
|
|
4093
|
-
neo4jDatabaseUsername: external.neo4jUsername,
|
|
4094
|
-
neo4jDatabasePassword: external.neo4jPassword,
|
|
4095
|
-
qwenKey: external.embeddingCredential,
|
|
4096
|
-
embeddingCredential: external.embeddingCredential,
|
|
4097
|
-
...(external.neo4jDatabase === undefined ? {} : { neo4jDatabase: external.neo4jDatabase }),
|
|
4065
|
+
const evidence = await resolveApprovedLiveConfiguration({
|
|
4066
|
+
repositoryRoot,
|
|
4067
|
+
requiredOptIns: [LIVE_PROVIDER_OPT_IN, W31_LIVE_OPT_IN],
|
|
4098
4068
|
});
|
|
4099
|
-
|
|
4100
|
-
|
|
4101
|
-
function getDefaultSemanticNeo4jDatabaseName(repositoryRoot) {
|
|
4102
|
-
const repoName = path.basename(repositoryRoot || resolveWorkspaceRoot());
|
|
4103
|
-
const normalized = String(repoName)
|
|
4104
|
-
.toLowerCase()
|
|
4105
|
-
.replace(/[^a-z0-9.-]+/g, '-')
|
|
4106
|
-
.replace(/^-+|-+$/g, '')
|
|
4107
|
-
.replace(/\.{2,}/g, '.')
|
|
4108
|
-
.replace(/-{2,}/g, '-');
|
|
4109
|
-
const safe = normalized || 'workspace';
|
|
4110
|
-
const prefixed = /^[a-z]/.test(safe) ? safe : `db-${safe}`;
|
|
4111
|
-
return prefixed.slice(0, 63);
|
|
4069
|
+
return buildDefaultSemanticConfiguration(evidence);
|
|
4112
4070
|
}
|
|
4113
4071
|
|
|
4114
4072
|
function deriveSemanticCanonicalVersion(document) {
|
package/package.json
CHANGED