gitnexus 1.6.10-rc.147 → 1.6.10-rc.148
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli/analyze.js +37 -8
- package/dist/cli/cli-message.d.ts +1 -1
- package/dist/core/ingestion/scope-resolution/graph-bridge/ids.d.ts +20 -0
- package/dist/core/ingestion/scope-resolution/graph-bridge/ids.js +11 -8
- package/dist/core/ingestion/scope-resolution/graph-bridge/node-lookup.d.ts +14 -0
- package/dist/core/ingestion/scope-resolution/graph-bridge/node-lookup.js +47 -31
- package/dist/core/lbug/csv-generator.js +4 -4
- package/dist/core/lbug/graph-emit-sink.d.ts +0 -1
- package/dist/core/lbug/graph-emit-sink.js +14 -14
- package/dist/core/lbug/pdg-emit-sink.d.ts +2 -3
- package/dist/core/lbug/pdg-emit-sink.js +12 -13
- package/dist/core/lbug/rel-pair-routing.d.ts +143 -10
- package/dist/core/lbug/rel-pair-routing.js +202 -20
- package/dist/core/lbug/schema.d.ts +38 -1
- package/dist/core/lbug/schema.js +236 -161
- package/dist/core/run-analyze.js +15 -3
- package/dist/lib/utils.d.ts +36 -0
- package/dist/lib/utils.js +48 -0
- package/dist/mcp/local/local-backend.js +7 -0
- package/dist/storage/repo-manager.d.ts +21 -1
- package/dist/storage/repo-manager.js +21 -1
- package/package.json +1 -1
package/dist/cli/analyze.js
CHANGED
|
@@ -14,6 +14,8 @@ import v8 from 'v8';
|
|
|
14
14
|
import cliProgress from 'cli-progress';
|
|
15
15
|
import { isLbugReady, LbugWipeError } from '../core/lbug/lbug-adapter.js';
|
|
16
16
|
import { boundedCheckpointBeforeExit } from '../core/lbug/shutdown-helpers.js';
|
|
17
|
+
import { findUndeclaredRelationPairError } from '../core/lbug/rel-pair-routing.js';
|
|
18
|
+
import { causeChain } from '../lib/utils.js';
|
|
17
19
|
import { getOsPageSize, isLbugCheckpointIoError, isLbugCheckpointBusyError, isLbugPageSizeFrameError, isPageSizeAwareLadybug, isWalCorruptionError, parseWalCheckpointThreshold, WAL_RECOVERY_SUGGESTION, } from '../core/lbug/lbug-config.js';
|
|
18
20
|
import { getStoragePaths, getGlobalRegistryPath, RegistryNameCollisionError, AnalysisNotFinalizedError, assertAnalysisFinalized, } from '../storage/repo-manager.js';
|
|
19
21
|
import { getGitRoot, hasGitDir, getDefaultBranch, selfCommitContextFiles, snapshotSelfCommitSafety, } from '../storage/git.js';
|
|
@@ -54,15 +56,13 @@ const writeFatalToStderr = (label, err) => {
|
|
|
54
56
|
// #2068) is only reachable via `.cause`. Without this the user sees the
|
|
55
57
|
// wrapper's main-thread stack and never the real frame. `cause.stack` already
|
|
56
58
|
// begins with the cause's message, so we print the stack alone (not message +
|
|
57
|
-
// stack) to avoid repeating it.
|
|
58
|
-
//
|
|
59
|
-
//
|
|
60
|
-
//
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
for (let depth = 0; depth < MAX_CAUSE_DEPTH && cause instanceof Error; depth++) {
|
|
59
|
+
// stack) to avoid repeating it. `causeChain` owns the traversal and the depth
|
|
60
|
+
// bound that stops a cyclic `cause` looping — this used to be one of four
|
|
61
|
+
// hand-rolled copies that had already drifted apart on both. Uses
|
|
62
|
+
// realStderrWrite so the redirected console.error's ANSI clear-line wrapping
|
|
63
|
+
// can't erase it (#1169). The head is skipped: it was just printed above.
|
|
64
|
+
for (const cause of causeChain(isErr ? err.cause : undefined)) {
|
|
64
65
|
realStderrWrite(`\n Caused by: ${cause.stack ?? cause.message}\n`);
|
|
65
|
-
cause = cause.cause;
|
|
66
66
|
}
|
|
67
67
|
};
|
|
68
68
|
let fatalHandlersInstalled = false;
|
|
@@ -1314,6 +1314,35 @@ const analyzeCommandImpl = async (inputPath, cliOptions, runnerIdentityAtBootstr
|
|
|
1314
1314
|
process.exitCode = 1;
|
|
1315
1315
|
return;
|
|
1316
1316
|
}
|
|
1317
|
+
// An extracted edge whose FROM→TO label pair is missing from GitNexus's own
|
|
1318
|
+
// relation DDL (#2789). `assertDeclaredPair` aborts the run rather than let
|
|
1319
|
+
// the bulk COPY fail late and silently drop the edge, so the user sees a
|
|
1320
|
+
// mid-run crash inside GitNexus internals with nothing to act on. Name the
|
|
1321
|
+
// pair, the relationship and the file that produced it, and say plainly that
|
|
1322
|
+
// a re-run cannot help — this is deterministic for the same input.
|
|
1323
|
+
// Checked by TYPE (repo norm, #2385) BEFORE the message-text heuristics
|
|
1324
|
+
// below, and through the `cause` chain because the ingestion phase runner
|
|
1325
|
+
// rewraps every phase failure as `Phase 'X' failed: …`.
|
|
1326
|
+
const undeclaredPair = findUndeclaredRelationPairError(err);
|
|
1327
|
+
if (undeclaredPair !== undefined) {
|
|
1328
|
+
// Render the error's OWN message indented — same idiom as the
|
|
1329
|
+
// `LbugWipeError` and page-size branches below. `UndeclaredRelationPairError`
|
|
1330
|
+
// builds a fully self-contained message (pair, relationship type, both node
|
|
1331
|
+
// ids, source file, issue URL, `.gitnexusignore` workaround) precisely
|
|
1332
|
+
// because `gitnexus serve` forwards only `err.message` over worker IPC.
|
|
1333
|
+
// Re-rendering those fields here would be a second copy of one string, free
|
|
1334
|
+
// to drift from the first — and the actionable half would reach CLI users
|
|
1335
|
+
// only. `undeclaredPair.message`, not the outer `msg`: the real error may be
|
|
1336
|
+
// several `cause` levels below the phase wrapper `msg` came from.
|
|
1337
|
+
cliError(` ${undeclaredPair.message.replace(/\n/g, '\n ')}\n`, {
|
|
1338
|
+
recoveryHint: 'undeclared-relation-pair',
|
|
1339
|
+
labelPair: undeclaredPair.pairKey,
|
|
1340
|
+
relationType: undeclaredPair.relationType,
|
|
1341
|
+
sourceFile: undeclaredPair.sourceFile,
|
|
1342
|
+
});
|
|
1343
|
+
process.exitCode = 1;
|
|
1344
|
+
return;
|
|
1345
|
+
}
|
|
1317
1346
|
// WAL corruption — the index file is unreadable. Give a clear recovery
|
|
1318
1347
|
// path without a confusing stack trace (the native error message alone
|
|
1319
1348
|
// is enough signal).
|
|
@@ -12,7 +12,7 @@ import { type CliMessageKey, type CliMessageVars } from './i18n/index.js';
|
|
|
12
12
|
* Consumers can import this type to narrow log-record `recoveryHint`
|
|
13
13
|
* fields without restating the literal list.
|
|
14
14
|
*/
|
|
15
|
-
export type RecoveryHint = 'wal-corruption' | 'wal-checkpoint-threshold' | 'lbug-wipe-failed' | 'lbug-page-size' | 'heap-oom-respawn' | 'native-worker-abort' | 'hf-endpoint-unreachable' | 'http-embedding-endpoint-error' | 'embedding-dims-invalid' | 'local-embedding-unsupported' | 'local-embedding-stack-missing' | 'large-repo' | 'npm-resolution' | 'module-not-found' | 'gitnexusrc-invalid' | 'default-branch-invalid' | 'index-lock-timeout';
|
|
15
|
+
export type RecoveryHint = 'wal-corruption' | 'wal-checkpoint-threshold' | 'lbug-wipe-failed' | 'lbug-page-size' | 'heap-oom-respawn' | 'native-worker-abort' | 'hf-endpoint-unreachable' | 'http-embedding-endpoint-error' | 'embedding-dims-invalid' | 'local-embedding-unsupported' | 'local-embedding-stack-missing' | 'large-repo' | 'npm-resolution' | 'module-not-found' | 'gitnexusrc-invalid' | 'default-branch-invalid' | 'index-lock-timeout' | 'undeclared-relation-pair';
|
|
16
16
|
/**
|
|
17
17
|
* Common shape for the optional structured-field bag passed to
|
|
18
18
|
* `cliError`/`cliWarn`/`cliInfo`. Typed so the `recoveryHint` slot is
|
|
@@ -19,6 +19,26 @@
|
|
|
19
19
|
import type { NodeLabel, ParameterTypeClass, ScopeId, SymbolDefinition } from '../../../../_shared/index.js';
|
|
20
20
|
import type { ScopeResolutionIndexes } from '../../model/scope-resolution-indexes.js';
|
|
21
21
|
import { type GraphNodeLookup } from '../graph-bridge/node-lookup.js';
|
|
22
|
+
/**
|
|
23
|
+
* Labels that may legitimately ANCHOR a CALLS/ACCESSES edge as the
|
|
24
|
+
* source ("caller"). A Variable / Property can be the TARGET of an
|
|
25
|
+
* edge (e.g., a write-access to `user.name`), but it cannot be a
|
|
26
|
+
* caller — variables don't execute code, so attributing a call to a
|
|
27
|
+
* sibling Variable in the same scope produces nonsense edges like
|
|
28
|
+
* `Variable:create → Function:create` (which the simpleKey fallback
|
|
29
|
+
* in `resolveDefGraphId` then silently rewrites to
|
|
30
|
+
* `Function:create → Function:create`, a self-loop that doesn't exist
|
|
31
|
+
* in the source).
|
|
32
|
+
*
|
|
33
|
+
* Module-level call expressions inside a `const X = expr(args)`
|
|
34
|
+
* declaration are the canonical case where this used to fail: the
|
|
35
|
+
* walk-up over module scope's ownedDefs (only Variables) would land
|
|
36
|
+
* on the FIRST Variable, get name-aliased to its sibling Function
|
|
37
|
+
* with the same simple name, and emit a self-CALLS. With this label
|
|
38
|
+
* restricted to function/class-likes, those calls correctly fall
|
|
39
|
+
* through to the File-node fallback at the bottom of the walk.
|
|
40
|
+
*/
|
|
41
|
+
export declare const CALLER_ANCHOR_LABELS: ReadonlySet<NodeLabel>;
|
|
22
42
|
export declare function resolveDefGraphId(filePath: string, def: {
|
|
23
43
|
/** Scope-resolution def id — carries the declaration position (#2699). */
|
|
24
44
|
nodeId?: string;
|
|
@@ -41,15 +41,18 @@ import { definitionIdPosition } from '../utils/definition-id.js';
|
|
|
41
41
|
* restricted to function/class-likes, those calls correctly fall
|
|
42
42
|
* through to the File-node fallback at the bottom of the walk.
|
|
43
43
|
*/
|
|
44
|
+
export const CALLER_ANCHOR_LABELS = new Set([
|
|
45
|
+
'Function',
|
|
46
|
+
'Method',
|
|
47
|
+
'Constructor',
|
|
48
|
+
'Module',
|
|
49
|
+
'Class',
|
|
50
|
+
'Interface',
|
|
51
|
+
'Struct',
|
|
52
|
+
'Enum',
|
|
53
|
+
]);
|
|
44
54
|
function isCallerAnchorLabel(label) {
|
|
45
|
-
return (label
|
|
46
|
-
label === 'Method' ||
|
|
47
|
-
label === 'Constructor' ||
|
|
48
|
-
label === 'Module' ||
|
|
49
|
-
label === 'Class' ||
|
|
50
|
-
label === 'Interface' ||
|
|
51
|
-
label === 'Struct' ||
|
|
52
|
-
label === 'Enum');
|
|
55
|
+
return CALLER_ANCHOR_LABELS.has(label);
|
|
53
56
|
}
|
|
54
57
|
function rangeContainsPoint(range, at) {
|
|
55
58
|
if (at.startLine < range.startLine || at.startLine > range.endLine)
|
|
@@ -75,4 +75,18 @@ export declare function localNameKey(filePath: string, label: NodeLabel, name: s
|
|
|
75
75
|
/** Tombstone for a position claimed by two nodes — see `positionKey`. */
|
|
76
76
|
export declare const AMBIGUOUS_POSITION = "";
|
|
77
77
|
export declare function buildGraphNodeLookup(graph: KnowledgeGraph): GraphNodeLookup;
|
|
78
|
+
/**
|
|
79
|
+
* Every label {@link buildGraphNodeLookup} registers — and therefore the ONLY
|
|
80
|
+
* labels `resolveDefGraphId` can ever return an id for. Both endpoints of every
|
|
81
|
+
* scope-resolution edge come from that lookup (the one exception is the File
|
|
82
|
+
* fallback in `resolveCallerGraphId`), so this set defines the whole FROM/TO
|
|
83
|
+
* surface those edges can produce.
|
|
84
|
+
*
|
|
85
|
+
* That makes it load-bearing for the LadybugDB relation DDL: a label added here
|
|
86
|
+
* without the matching `FROM x TO y` pairs in `RELATION_SCHEMA` crashes
|
|
87
|
+
* `analyze` at `assertDeclaredPair` on whichever codebase first emits the pair
|
|
88
|
+
* (#2792). `test/unit/schema-pair-coverage.test.ts` derives the required pairs
|
|
89
|
+
* from this set and fails in CI instead.
|
|
90
|
+
*/
|
|
91
|
+
export declare const LINKABLE_LABELS: ReadonlySet<NodeLabel>;
|
|
78
92
|
export declare function isLinkableLabel(label: NodeLabel): boolean;
|
|
@@ -209,36 +209,52 @@ export function buildGraphNodeLookup(graph) {
|
|
|
209
209
|
}
|
|
210
210
|
return lookup;
|
|
211
211
|
}
|
|
212
|
+
/**
|
|
213
|
+
* Every label {@link buildGraphNodeLookup} registers — and therefore the ONLY
|
|
214
|
+
* labels `resolveDefGraphId` can ever return an id for. Both endpoints of every
|
|
215
|
+
* scope-resolution edge come from that lookup (the one exception is the File
|
|
216
|
+
* fallback in `resolveCallerGraphId`), so this set defines the whole FROM/TO
|
|
217
|
+
* surface those edges can produce.
|
|
218
|
+
*
|
|
219
|
+
* That makes it load-bearing for the LadybugDB relation DDL: a label added here
|
|
220
|
+
* without the matching `FROM x TO y` pairs in `RELATION_SCHEMA` crashes
|
|
221
|
+
* `analyze` at `assertDeclaredPair` on whichever codebase first emits the pair
|
|
222
|
+
* (#2792). `test/unit/schema-pair-coverage.test.ts` derives the required pairs
|
|
223
|
+
* from this set and fails in CI instead.
|
|
224
|
+
*/
|
|
225
|
+
export const LINKABLE_LABELS = new Set([
|
|
226
|
+
'Function',
|
|
227
|
+
'Method',
|
|
228
|
+
'Constructor',
|
|
229
|
+
// Program-like module declarations are provider-gated callable-value
|
|
230
|
+
// targets and need the same def→graph bridge.
|
|
231
|
+
'Module',
|
|
232
|
+
'Class',
|
|
233
|
+
'Interface',
|
|
234
|
+
'Struct',
|
|
235
|
+
'Enum',
|
|
236
|
+
// Trait nodes are linkable so MRO builders can bridge PHP/Rust trait
|
|
237
|
+
// defs between scope-resolution DefIds and the graph's node ids.
|
|
238
|
+
// IMPLEMENTS edges from classes to traits are otherwise invisible to
|
|
239
|
+
// the scope-resolution MRO pass.
|
|
240
|
+
'Trait',
|
|
241
|
+
// Variable / Property are linkable too — receiver-bound write/read
|
|
242
|
+
// ACCESSES edges target field nodes (e.g. `user.name = "x"` →
|
|
243
|
+
// ACCESSES edge to User's `name` Variable/Property node).
|
|
244
|
+
'Variable',
|
|
245
|
+
'Property',
|
|
246
|
+
// Const is linkable so the value-receiver-owner bridge in
|
|
247
|
+
// `receiver-bound-calls.ts` Case 5 can translate the scope-resolution
|
|
248
|
+
// `Variable` def for `export const fooService = {...}` to the canonical
|
|
249
|
+
// `Const:filePath:name` graph node id, against which object-literal
|
|
250
|
+
// method symbols register their `ownerId` (PR #1718 / issue #1358).
|
|
251
|
+
'Const',
|
|
252
|
+
// Macro nodes are linkable so a macro invocation (`log!(…)`) resolved
|
|
253
|
+
// via `MacroRegistry` can bridge its scope-resolution `Macro` def to
|
|
254
|
+
// the legacy `@definition.macro` graph node and emit the `USES` edge
|
|
255
|
+
// (Rust #1934 F72; also covers C/C++ `#define` macro defs).
|
|
256
|
+
'Macro',
|
|
257
|
+
]);
|
|
212
258
|
export function isLinkableLabel(label) {
|
|
213
|
-
return (label
|
|
214
|
-
label === 'Method' ||
|
|
215
|
-
label === 'Constructor' ||
|
|
216
|
-
// Program-like module declarations are provider-gated callable-value
|
|
217
|
-
// targets and need the same def→graph bridge.
|
|
218
|
-
label === 'Module' ||
|
|
219
|
-
label === 'Class' ||
|
|
220
|
-
label === 'Interface' ||
|
|
221
|
-
label === 'Struct' ||
|
|
222
|
-
label === 'Enum' ||
|
|
223
|
-
// Trait nodes are linkable so MRO builders can bridge PHP/Rust trait
|
|
224
|
-
// defs between scope-resolution DefIds and the graph's node ids.
|
|
225
|
-
// IMPLEMENTS edges from classes to traits are otherwise invisible to
|
|
226
|
-
// the scope-resolution MRO pass.
|
|
227
|
-
label === 'Trait' ||
|
|
228
|
-
// Variable / Property are linkable too — receiver-bound write/read
|
|
229
|
-
// ACCESSES edges target field nodes (e.g. `user.name = "x"` →
|
|
230
|
-
// ACCESSES edge to User's `name` Variable/Property node).
|
|
231
|
-
label === 'Variable' ||
|
|
232
|
-
label === 'Property' ||
|
|
233
|
-
// Const is linkable so the value-receiver-owner bridge in
|
|
234
|
-
// `receiver-bound-calls.ts` Case 5 can translate the scope-resolution
|
|
235
|
-
// `Variable` def for `export const fooService = {...}` to the canonical
|
|
236
|
-
// `Const:filePath:name` graph node id, against which object-literal
|
|
237
|
-
// method symbols register their `ownerId` (PR #1718 / issue #1358).
|
|
238
|
-
label === 'Const' ||
|
|
239
|
-
// Macro nodes are linkable so a macro invocation (`log!(…)`) resolved
|
|
240
|
-
// via `MacroRegistry` can bridge its scope-resolution `Macro` def to
|
|
241
|
-
// the legacy `@definition.macro` graph node and emit the `USES` edge
|
|
242
|
-
// (Rust #1934 F72; also covers C/C++ `#define` macro defs).
|
|
243
|
-
label === 'Macro');
|
|
259
|
+
return LINKABLE_LABELS.has(label);
|
|
244
260
|
}
|
|
@@ -14,8 +14,8 @@
|
|
|
14
14
|
import fs from 'fs/promises';
|
|
15
15
|
import { createWriteStream } from 'fs';
|
|
16
16
|
import path from 'path';
|
|
17
|
-
import {
|
|
18
|
-
import { parseRelationSchemaPairs, RelPairRouter } from './rel-pair-routing.js';
|
|
17
|
+
import { RELATION_SCHEMA } from './schema.js';
|
|
18
|
+
import { VALID_NODE_TABLES, parseRelationSchemaPairs, RelPairRouter } from './rel-pair-routing.js';
|
|
19
19
|
import { parseTruthyEnv } from '../ingestion/utils/env.js';
|
|
20
20
|
import { SYMBOL_NODE_LABELS } from '../ingestion/utils/symbol-labels.js';
|
|
21
21
|
import { applyCjkSegmentationIfEnabled } from '../search/cjk-segmentation.js';
|
|
@@ -662,11 +662,11 @@ export const streamAllCSVsToDisk = async (graph, repoPath, csvDir, onNodePhaseCo
|
|
|
662
662
|
// read once instead of twice. The router applies the SAME label-derivation +
|
|
663
663
|
// validTables filter as the legacy splitRelCsvByLabelPair, so the per-pair
|
|
664
664
|
// files are byte-identical (asserted by the differential test).
|
|
665
|
-
const relRouter = new RelPairRouter(csvDir, REL_CSV_HEADER,
|
|
665
|
+
const relRouter = new RelPairRouter(csvDir, REL_CSV_HEADER, VALID_NODE_TABLES, DECLARED_RELATION_PAIRS);
|
|
666
666
|
try {
|
|
667
667
|
let emitted = 0;
|
|
668
668
|
for (const rel of orderedRelationships(graph, sortOutput)) {
|
|
669
|
-
const pending = relRouter.route(rel.sourceId, rel.targetId, buildRelRow(rel));
|
|
669
|
+
const pending = relRouter.route(rel.sourceId, rel.targetId, buildRelRow(rel), rel.type);
|
|
670
670
|
if (pending)
|
|
671
671
|
await pending;
|
|
672
672
|
// Periodically hand the event loop back so the overlapped node COPY and
|
|
@@ -81,7 +81,6 @@ export declare class GraphEmitSink implements KnowledgeGraph, GraphEmitControl {
|
|
|
81
81
|
private readonly real;
|
|
82
82
|
private readonly csvDir;
|
|
83
83
|
private readonly chunkRows;
|
|
84
|
-
private readonly validTables;
|
|
85
84
|
private readonly relWriters;
|
|
86
85
|
/**
|
|
87
86
|
* Ids of relationships already streamed. `KnowledgeGraph.addRelationship`
|
|
@@ -85,17 +85,17 @@
|
|
|
85
85
|
* ## Correctness contract
|
|
86
86
|
*
|
|
87
87
|
* Structural sibling of {@link PdgEmitSink}, and reuses its row builder
|
|
88
|
-
* (`buildRelRow`), header (`REL_CSV_HEADER`)
|
|
89
|
-
*
|
|
90
|
-
* whole-graph emit's and the bulk COPY loads
|
|
88
|
+
* (`buildRelRow`), header (`REL_CSV_HEADER`) and pair classification
|
|
89
|
+
* (`relPairKeyFor`, which is also what `RelPairRouter` routes and skips by), so
|
|
90
|
+
* the streamed row SET equals the whole-graph emit's and the bulk COPY loads
|
|
91
|
+
* the same rows. Set-level, not
|
|
91
92
|
* byte-level: rows stream in emit order and are not re-sorted under
|
|
92
93
|
* `GITNEXUS_SORT_GRAPH_OUTPUT`.
|
|
93
94
|
*/
|
|
94
95
|
import fs from 'fs';
|
|
95
96
|
import path from 'path';
|
|
96
97
|
import { DECLARED_RELATION_PAIRS, REL_CSV_HEADER, buildRelRow } from './csv-generator.js';
|
|
97
|
-
import { assertDeclaredPair,
|
|
98
|
-
import { NODE_TABLES } from './schema.js';
|
|
98
|
+
import { VALID_NODE_TABLES, assertDeclaredPair, relPairKeyFor, splitRelPairKey, } from './rel-pair-routing.js';
|
|
99
99
|
import { DEFAULT_EMIT_CHUNK_ROWS, SyncCsvWriter } from './sync-csv-writer.js';
|
|
100
100
|
/**
|
|
101
101
|
* Relationship types that MUST stay in the in-memory graph because a phase
|
|
@@ -200,7 +200,6 @@ export class GraphEmitSink {
|
|
|
200
200
|
real;
|
|
201
201
|
csvDir;
|
|
202
202
|
chunkRows;
|
|
203
|
-
validTables;
|
|
204
203
|
relWriters = new Map();
|
|
205
204
|
/**
|
|
206
205
|
* Ids of relationships already streamed. `KnowledgeGraph.addRelationship`
|
|
@@ -272,7 +271,6 @@ export class GraphEmitSink {
|
|
|
272
271
|
this.real = real;
|
|
273
272
|
this.csvDir = csvDir;
|
|
274
273
|
this.chunkRows = chunkRows;
|
|
275
|
-
this.validTables = new Set(NODE_TABLES);
|
|
276
274
|
// Own directory, distinct from the PDG sink's: PdgEmitSink wipes and
|
|
277
275
|
// recreates its dir on construction and opens with O_EXCL, so a shared dir
|
|
278
276
|
// would destroy the other sink's manifest on a combined --pdg run.
|
|
@@ -367,16 +365,18 @@ export class GraphEmitSink {
|
|
|
367
365
|
return;
|
|
368
366
|
}
|
|
369
367
|
// Mirror KnowledgeGraph.addRelationship's first-writer-wins dedup.
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
//
|
|
373
|
-
//
|
|
374
|
-
|
|
368
|
+
// Classify + skip via the SHARED `relPairKeyFor`, not a local copy of its
|
|
369
|
+
// three lines, so the streamed set cannot drift from the whole-graph set
|
|
370
|
+
// `RelPairRouter` produces. `undefined` = an endpoint label is not a node
|
|
371
|
+
// table, so the edge is dropped exactly as the router drops it.
|
|
372
|
+
const pairKey = relPairKeyFor(relationship.sourceId, relationship.targetId, VALID_NODE_TABLES);
|
|
373
|
+
if (pairKey === undefined)
|
|
375
374
|
return;
|
|
376
|
-
|
|
377
|
-
assertDeclaredPair(pairKey, DECLARED_RELATION_PAIRS);
|
|
375
|
+
assertDeclaredPair(pairKey, DECLARED_RELATION_PAIRS, relationship.type, relationship.sourceId, relationship.targetId);
|
|
378
376
|
let writer = this.relWriters.get(pairKey);
|
|
379
377
|
if (writer === undefined) {
|
|
378
|
+
// Cold: once per pair, so decoding the key back into its labels is free.
|
|
379
|
+
const [fromLabel, toLabel] = splitRelPairKey(pairKey);
|
|
380
380
|
try {
|
|
381
381
|
writer = new SyncCsvWriter(path.join(this.csvDir, `rel_${fromLabel}_${toLabel}.csv`), REL_CSV_HEADER, this.chunkRows);
|
|
382
382
|
}
|
|
@@ -24,8 +24,8 @@
|
|
|
24
24
|
* `storage/parsedfile-store.ts`.
|
|
25
25
|
*
|
|
26
26
|
* Byte-identity (issue acceptance): the sink reuses the SAME shared row
|
|
27
|
-
* builders (`buildBasicBlockRow`, `buildRelRow`) and
|
|
28
|
-
* (`
|
|
27
|
+
* builders (`buildBasicBlockRow`, `buildRelRow`) and pair classification
|
|
28
|
+
* (`relPairKeyFor`) as `streamAllCSVsToDisk`, so the streamed CSV line SET is
|
|
29
29
|
* identical to the whole-graph emit's, and the bulk COPY loads the same rows →
|
|
30
30
|
* the persisted graph is SET-identical and DB-identical. The guarantee is
|
|
31
31
|
* set-level, not byte-level on the CSV file: the sink streams rows in emit
|
|
@@ -76,7 +76,6 @@ export declare class PdgEmitSink implements KnowledgeGraph {
|
|
|
76
76
|
private readonly real;
|
|
77
77
|
private readonly pdgCsvDir;
|
|
78
78
|
private readonly chunkRows;
|
|
79
|
-
private readonly validTables;
|
|
80
79
|
private bbWriter;
|
|
81
80
|
/** pairKey (`From|To`) → writer. PDG edges are all `BasicBlock|BasicBlock`,
|
|
82
81
|
* but the map keeps the sink general and the manifest pair-keyed. */
|
|
@@ -24,8 +24,8 @@
|
|
|
24
24
|
* `storage/parsedfile-store.ts`.
|
|
25
25
|
*
|
|
26
26
|
* Byte-identity (issue acceptance): the sink reuses the SAME shared row
|
|
27
|
-
* builders (`buildBasicBlockRow`, `buildRelRow`) and
|
|
28
|
-
* (`
|
|
27
|
+
* builders (`buildBasicBlockRow`, `buildRelRow`) and pair classification
|
|
28
|
+
* (`relPairKeyFor`) as `streamAllCSVsToDisk`, so the streamed CSV line SET is
|
|
29
29
|
* identical to the whole-graph emit's, and the bulk COPY loads the same rows →
|
|
30
30
|
* the persisted graph is SET-identical and DB-identical. The guarantee is
|
|
31
31
|
* set-level, not byte-level on the CSV file: the sink streams rows in emit
|
|
@@ -45,9 +45,8 @@
|
|
|
45
45
|
import fs from 'fs';
|
|
46
46
|
import path from 'path';
|
|
47
47
|
import { BASICBLOCK_CSV_HEADER, DECLARED_RELATION_PAIRS, REL_CSV_HEADER, buildBasicBlockRow, buildRelRow, } from './csv-generator.js';
|
|
48
|
-
import { assertDeclaredPair,
|
|
48
|
+
import { VALID_NODE_TABLES, assertDeclaredPair, relPairKeyFor, splitRelPairKey, } from './rel-pair-routing.js';
|
|
49
49
|
import { DEFAULT_EMIT_CHUNK_ROWS, SyncCsvWriter } from './sync-csv-writer.js';
|
|
50
|
-
import { NODE_TABLES } from './schema.js';
|
|
51
50
|
/**
|
|
52
51
|
* PDG edge types streamed per-file (all intra-block BasicBlock→BasicBlock).
|
|
53
52
|
* `TAINT_PATH` is intentionally excluded — it is the whole-program M4 edge
|
|
@@ -76,7 +75,6 @@ export class PdgEmitSink {
|
|
|
76
75
|
real;
|
|
77
76
|
pdgCsvDir;
|
|
78
77
|
chunkRows;
|
|
79
|
-
validTables;
|
|
80
78
|
bbWriter;
|
|
81
79
|
/** pairKey (`From|To`) → writer. PDG edges are all `BasicBlock|BasicBlock`,
|
|
82
80
|
* but the map keeps the sink general and the manifest pair-keyed. */
|
|
@@ -105,7 +103,6 @@ export class PdgEmitSink {
|
|
|
105
103
|
this.real = real;
|
|
106
104
|
this.pdgCsvDir = pdgCsvDir;
|
|
107
105
|
this.chunkRows = chunkRows;
|
|
108
|
-
this.validTables = new Set(NODE_TABLES);
|
|
109
106
|
// Clear any streamed CSVs left by a previous (possibly crashed) run so a
|
|
110
107
|
// later COPY never picks up stale rows.
|
|
111
108
|
fs.rmSync(pdgCsvDir, { recursive: true, force: true });
|
|
@@ -130,16 +127,18 @@ export class PdgEmitSink {
|
|
|
130
127
|
}
|
|
131
128
|
addRelationship(relationship) {
|
|
132
129
|
if (PDG_EDGE_TYPES.has(relationship.type)) {
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
//
|
|
136
|
-
//
|
|
137
|
-
|
|
130
|
+
// Classify + skip via the SHARED `relPairKeyFor`, not a local copy of its
|
|
131
|
+
// three lines, so the streamed set cannot drift from the whole-graph set
|
|
132
|
+
// `RelPairRouter` produces. `undefined` = an endpoint label is not a node
|
|
133
|
+
// table, so the edge is dropped exactly as the router drops it.
|
|
134
|
+
const pairKey = relPairKeyFor(relationship.sourceId, relationship.targetId, VALID_NODE_TABLES);
|
|
135
|
+
if (pairKey === undefined)
|
|
138
136
|
return;
|
|
139
|
-
|
|
140
|
-
assertDeclaredPair(pairKey, DECLARED_RELATION_PAIRS);
|
|
137
|
+
assertDeclaredPair(pairKey, DECLARED_RELATION_PAIRS, relationship.type, relationship.sourceId, relationship.targetId);
|
|
141
138
|
let writer = this.relWriters.get(pairKey);
|
|
142
139
|
if (writer === undefined) {
|
|
140
|
+
// Cold: once per pair, so decoding the key back into labels is free.
|
|
141
|
+
const [fromLabel, toLabel] = splitRelPairKey(pairKey);
|
|
143
142
|
try {
|
|
144
143
|
writer = new SyncCsvWriter(path.join(this.pdgCsvDir, `rel_${fromLabel}_${toLabel}.csv`), REL_CSV_HEADER, this.chunkRows);
|
|
145
144
|
}
|
|
@@ -1,6 +1,22 @@
|
|
|
1
1
|
import { type WriteStream } from 'fs';
|
|
2
2
|
/** Injectable for tests (backpressure/error simulation), mirroring split. */
|
|
3
3
|
export type WriteStreamFactory = (filePath: string) => WriteStream;
|
|
4
|
+
/**
|
|
5
|
+
* Every label LadybugDB has a node table for — the filter that decides whether
|
|
6
|
+
* an edge is routable at all.
|
|
7
|
+
*
|
|
8
|
+
* ONE shared instance, deliberately. `RelPairRouter`, `GraphEmitSink` and
|
|
9
|
+
* `PdgEmitSink` each used to build their own `new Set(NODE_TABLES)`; three
|
|
10
|
+
* copies of the same immutable set are three chances to seed one of them from
|
|
11
|
+
* a different source. Typed `ReadonlySet` because that — not `Object.freeze`,
|
|
12
|
+
* which does not touch a Set's internal slots — is what actually stops a
|
|
13
|
+
* consumer mutating the shared instance.
|
|
14
|
+
*
|
|
15
|
+
* Imported straight from `gitnexus-shared` rather than `./schema.js`: schema.ts
|
|
16
|
+
* imports `parseRelationSchemaPairs` from this module, so the reverse import
|
|
17
|
+
* would close a cycle.
|
|
18
|
+
*/
|
|
19
|
+
export declare const VALID_NODE_TABLES: ReadonlySet<string>;
|
|
4
20
|
/**
|
|
5
21
|
* Derive a node's table label from its graph id. Matches the legacy
|
|
6
22
|
* `getNodeLabel` that lived inline in `loadGraphToLbug`:
|
|
@@ -9,18 +25,125 @@ export type WriteStreamFactory = (filePath: string) => WriteStream;
|
|
|
9
25
|
* - otherwise the prefix before the first `:` (e.g. `Function:…` → Function)
|
|
10
26
|
*/
|
|
11
27
|
export declare const getNodeLabel: (nodeId: string) => string;
|
|
28
|
+
/**
|
|
29
|
+
* Classify one edge into its `From|To` pair key, or `undefined` when the edge
|
|
30
|
+
* must be SKIPPED because an endpoint's label is not a real node table.
|
|
31
|
+
*
|
|
32
|
+
* THE single definition of "which pair does this edge belong to, and is it
|
|
33
|
+
* routable at all". `RelPairRouter.route`, `GraphEmitSink.addRelationship`,
|
|
34
|
+
* `PdgEmitSink.addRelationship` and the `structural-pair-coverage` corpus guard
|
|
35
|
+
* each used to inline the same three lines (label both ends → drop if either
|
|
36
|
+
* label is not a node table → join with `|`). The corpus guard's docblock said
|
|
37
|
+
* it "mirrors `RelPairRouter.route`" — a mirror is a drift marker: change the
|
|
38
|
+
* skip rule here and the guard would keep classifying by the old one, report
|
|
39
|
+
* green, and let `analyze` abort on a pair it had already declared covered.
|
|
40
|
+
*
|
|
41
|
+
* HOT PATH — called once per edge (~1M on a large repo). Returns the key
|
|
42
|
+
* string (which every caller needs anyway for its own Map lookup) rather than
|
|
43
|
+
* a `{ pairKey, fromLabel, toLabel }` object or a tuple, so the success path
|
|
44
|
+
* allocates nothing beyond what `getNodeLabel` already did. Callers that need
|
|
45
|
+
* the two labels back — only when opening a new pair's CSV, once per pair —
|
|
46
|
+
* decode the key with {@link splitRelPairKey}.
|
|
47
|
+
*/
|
|
48
|
+
export declare const relPairKeyFor: (fromId: string, toId: string, validTables: ReadonlySet<string>) => string | undefined;
|
|
49
|
+
/**
|
|
50
|
+
* Decode a `From|To` pair key back into its two labels.
|
|
51
|
+
*
|
|
52
|
+
* Safe because `|` cannot occur inside a node label: every label is a
|
|
53
|
+
* `NODE_TABLES` identifier (`[A-Za-z][A-Za-z0-9_]*`), so the FIRST `|` is
|
|
54
|
+
* always the separator. That invariant was documented in one comment and
|
|
55
|
+
* enforced nowhere while every consumer re-derived it with a bare
|
|
56
|
+
* `key.split('|')`.
|
|
57
|
+
*
|
|
58
|
+
* DECODE ONLY — there is deliberately no matching `encode` helper. The key is
|
|
59
|
+
* built once per edge inside {@link relPairKeyFor} (~1M edges on a large
|
|
60
|
+
* repo), where a function call is a real regression risk; every decode site is
|
|
61
|
+
* cold by construction (once per pair when its CSV is opened, or on the
|
|
62
|
+
* throw path of {@link assertDeclaredPair}).
|
|
63
|
+
*/
|
|
64
|
+
export declare const splitRelPairKey: (key: string) => readonly [from: string, to: string];
|
|
65
|
+
/**
|
|
66
|
+
* Build a fresh matcher for the `FROM <label> TO <label>` clauses of a
|
|
67
|
+
* relationship DDL. Capture group 1 is the FROM label, group 2 the TO label;
|
|
68
|
+
* backticks quote schema labels and are not part of the graph label.
|
|
69
|
+
*
|
|
70
|
+
* THE SINGLE SOURCE OF TRUTH for that pattern. `parseRelationSchemaPairs`
|
|
71
|
+
* below builds its pair set from it, and `test/unit/schema-pair-coverage.test.ts`
|
|
72
|
+
* counts raw `FROM…TO` occurrences with it to catch a pair DUPLICATED in the
|
|
73
|
+
* DDL (a duplicate makes LadybugDB reject `CREATE REL TABLE`, killing every
|
|
74
|
+
* `analyze` — strictly worse than one missing pair). That guard used to inline
|
|
75
|
+
* its own copy of the regex: the two matched identically, so it worked, but any
|
|
76
|
+
* widening here (dotted identifiers, `IF NOT EXISTS`, a multi-target
|
|
77
|
+
* `FROM x TO y, z` form) would have silently degraded it to the tautology
|
|
78
|
+
* `declared.size === declared.size`. Consume this factory instead of
|
|
79
|
+
* re-inlining a copy.
|
|
80
|
+
*
|
|
81
|
+
* A FACTORY, not a shared `RegExp`: a module-level `/g` regex carries
|
|
82
|
+
* `lastIndex` between calls, so one consumer's `exec`/`test` would corrupt
|
|
83
|
+
* everyone else's next match. Each call returns a private instance.
|
|
84
|
+
*/
|
|
85
|
+
export declare const createRelationPairMatcher: () => RegExp;
|
|
12
86
|
/**
|
|
13
87
|
* Extract the FROM→TO pairs accepted by a relationship DDL.
|
|
14
88
|
*
|
|
15
89
|
* This belongs at the routing boundary: schema.ts owns the DDL, while the CSV
|
|
16
90
|
* router owns the fail-fast check that prevents writing a pair LadybugDB cannot
|
|
17
|
-
* COPY.
|
|
91
|
+
* COPY.
|
|
18
92
|
*/
|
|
19
93
|
export declare const parseRelationSchemaPairs: (relationSchema: string) => ReadonlySet<string>;
|
|
20
94
|
export interface RelPairMeta {
|
|
21
95
|
csvPath: string;
|
|
22
96
|
rows: number;
|
|
23
97
|
}
|
|
98
|
+
/**
|
|
99
|
+
* An edge whose endpoint-label pair is absent from the relationship DDL.
|
|
100
|
+
*
|
|
101
|
+
* Carries the context the emit call site already has — relationship type, both
|
|
102
|
+
* node ids, and the source file derived from them — so a user whose `analyze`
|
|
103
|
+
* just died mid-run can see WHICH of their files produced the edge and file a
|
|
104
|
+
* bug report that names the missing pair. The abstract label pair alone is
|
|
105
|
+
* unactionable outside GitNexus's own source (#2789).
|
|
106
|
+
*
|
|
107
|
+
* Classify by TYPE (`err instanceof UndeclaredRelationPairError`, or
|
|
108
|
+
* {@link findUndeclaredRelationPairError} when the error may be wrapped in a
|
|
109
|
+
* phase `cause` chain) — the repo norm from #2385 — never by message text.
|
|
110
|
+
*
|
|
111
|
+
* THE MESSAGE IS THE ONLY RENDERING. It carries the five context fields AND
|
|
112
|
+
* the two actionable next steps (report the pair; `.gitnexusignore` the file to
|
|
113
|
+
* finish the rest of the index), because `gitnexus serve` forwards nothing but
|
|
114
|
+
* `err.message` over worker IPC — anything a consumer re-renders from the
|
|
115
|
+
* structured fields instead is invisible to a serve-hosted user. The CLI
|
|
116
|
+
* branch in `cli/analyze.ts` therefore prints this message indented and adds
|
|
117
|
+
* only the machine-readable `cliError` fields, the same idiom `LbugWipeError`
|
|
118
|
+
* uses there. It used to re-render the five fields with its own wording; the
|
|
119
|
+
* two copies had already drifted on the pair separator, the no-file text and
|
|
120
|
+
* the closing sentence within a single PR, and each had its own pinning test.
|
|
121
|
+
*/
|
|
122
|
+
export declare class UndeclaredRelationPairError extends Error {
|
|
123
|
+
/** `From|To` label pair, exactly as keyed against the declared-pair set. */
|
|
124
|
+
readonly pairKey: string;
|
|
125
|
+
/** Relationship type of the edge that could not be routed (e.g. `CALLS`). */
|
|
126
|
+
readonly relationType: string;
|
|
127
|
+
readonly fromId: string;
|
|
128
|
+
readonly toId: string;
|
|
129
|
+
/** Source file derived from the node ids; `undefined` for synthetic ids. */
|
|
130
|
+
readonly sourceFile: string | undefined;
|
|
131
|
+
constructor(pairKey: string, relationType: string, fromId: string, toId: string);
|
|
132
|
+
}
|
|
133
|
+
/**
|
|
134
|
+
* Find an {@link UndeclaredRelationPairError} in `err` or its `cause` chain.
|
|
135
|
+
*
|
|
136
|
+
* The guard throws deep inside an ingestion phase, and the phase runner rewraps
|
|
137
|
+
* every phase failure as `new Error("Phase 'X' failed: …", { cause })` — so a
|
|
138
|
+
* bare `instanceof` at the CLI boundary would miss it and fall through to the
|
|
139
|
+
* generic stack dump.
|
|
140
|
+
*
|
|
141
|
+
* The traversal and its depth bound come from `lib/utils.ts` rather than being
|
|
142
|
+
* re-rolled here: this was the fourth hand-written copy in the repo and the
|
|
143
|
+
* only one that used `depth <= MAX` (six levels) while claiming to mirror
|
|
144
|
+
* `cli/analyze.ts`'s `depth < 5`.
|
|
145
|
+
*/
|
|
146
|
+
export declare const findUndeclaredRelationPairError: (err: unknown) => UndeclaredRelationPairError | undefined;
|
|
24
147
|
/**
|
|
25
148
|
* Fail fast on an endpoint-label pair absent from the relationship DDL, the
|
|
26
149
|
* same guard `RelPairRouter.route` applies to the whole-graph emit. Exported
|
|
@@ -29,14 +152,19 @@ export interface RelPairMeta {
|
|
|
29
152
|
* the bulk insert, and is silently dropped by the per-edge fallback instead
|
|
30
153
|
* of failing loudly like the non-streaming path does.
|
|
31
154
|
*
|
|
32
|
-
* Takes the already-built `From|To` pairKey
|
|
33
|
-
* caller needs that same key immediately
|
|
34
|
-
* and this is on the per-edge hot path,
|
|
35
|
-
*
|
|
36
|
-
*
|
|
37
|
-
*
|
|
155
|
+
* Takes the already-built `From|To` pairKey (from {@link relPairKeyFor})
|
|
156
|
+
* rather than the two labels — every caller needs that same key immediately
|
|
157
|
+
* after for its own Map/stream lookup, and this is on the per-edge hot path,
|
|
158
|
+
* so building it twice would be a needless allocation per edge. The error
|
|
159
|
+
* splits it back apart with {@link splitRelPairKey}, which only the throw path
|
|
160
|
+
* reaches. The edge context is passed POSITIONALLY for the same reason: a
|
|
161
|
+
* `{ relationType, fromId, toId }` context object would allocate on every
|
|
162
|
+
* edge, including the ~1M that never fail.
|
|
163
|
+
*
|
|
164
|
+
* The success path must stay allocation-free: no object literal, no template
|
|
165
|
+
* string, no closure, no `Error` constructed before the failure branch.
|
|
38
166
|
*/
|
|
39
|
-
export declare const assertDeclaredPair: (pairKey: string, declaredPairs: ReadonlySet<string
|
|
167
|
+
export declare const assertDeclaredPair: (pairKey: string, declaredPairs: ReadonlySet<string>, relationType: string, fromId: string, toId: string) => void;
|
|
40
168
|
/**
|
|
41
169
|
* Routes already-escaped relationship CSV rows to per-FROM→TO-label-pair
|
|
42
170
|
* files. Filters edges whose endpoint labels are not valid node tables
|
|
@@ -55,7 +183,7 @@ export declare class RelPairRouter {
|
|
|
55
183
|
total: number;
|
|
56
184
|
private streamError;
|
|
57
185
|
private readonly abort;
|
|
58
|
-
constructor(csvDir: string, header: string, validTables:
|
|
186
|
+
constructor(csvDir: string, header: string, validTables: ReadonlySet<string>, declaredPairs: ReadonlySet<string>, wsFactory?: WriteStreamFactory);
|
|
59
187
|
private markError;
|
|
60
188
|
/**
|
|
61
189
|
* The first stream error observed, if any. Lets the emit caller rethrow the
|
|
@@ -69,8 +197,13 @@ export declare class RelPairRouter {
|
|
|
69
197
|
* Returns `void` on the synchronous hot path; a `Promise<void>` only when a
|
|
70
198
|
* stream signals backpressure (or a new pair's header does) — the caller
|
|
71
199
|
* awaits the promise before routing the next edge.
|
|
200
|
+
*
|
|
201
|
+
* `relType` is not used for routing — it is carried purely so an undeclared
|
|
202
|
+
* pair can name the offending relationship in its error (the row is already
|
|
203
|
+
* CSV-escaped by then, so the type is not recoverable from it).
|
|
72
204
|
*/
|
|
73
|
-
route(fromId: string, toId: string, row: string): void | Promise<void>;
|
|
205
|
+
route(fromId: string, toId: string, row: string, relType: string): void | Promise<void>;
|
|
206
|
+
/** Cold: runs once per pair, so decoding the key back is free here. */
|
|
74
207
|
private openAndWrite;
|
|
75
208
|
/** Flush + close every pair stream. Rejects if any stream errored. */
|
|
76
209
|
close(): Promise<void>;
|