gitnexus 1.6.10-rc.152 → 1.6.10-rc.154
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/core/group/service.js +10 -1
- package/dist/core/ingestion/cfg/callee-cell-format.d.ts +42 -0
- package/dist/core/ingestion/cfg/callee-cell-format.js +42 -0
- package/dist/core/ingestion/cfg/emit.d.ts +8 -23
- package/dist/core/ingestion/cfg/emit.js +9 -23
- package/dist/core/lbug/lbug-adapter.js +6 -0
- package/dist/core/run-analyze.js +3 -2
- package/dist/mcp/local/pdg-impact.d.ts +131 -9
- package/dist/mcp/local/pdg-impact.js +341 -77
- package/dist/mcp/tools.js +1 -1
- package/package.json +1 -1
- package/scripts/cross-platform-shard.ts +136 -0
- package/scripts/cross-platform-tests.ts +27 -0
- package/scripts/run-cross-platform.ts +16 -5
|
@@ -9,7 +9,10 @@ import { loadMeta } from '../../storage/repo-manager.js';
|
|
|
9
9
|
import { GroupNotFoundError, loadGroupConfig } from './config-parser.js';
|
|
10
10
|
import { fileMatchesServicePrefix, normalizeServicePrefix, repoInSubgroup, } from './group-path-utils.js';
|
|
11
11
|
import { getDefaultGitnexusDir, getGroupDir, listGroups, readContractRegistry } from './storage.js';
|
|
12
|
-
|
|
12
|
+
// `./sync.js` is imported LAZILY in `groupSync` — see the comment at its call
|
|
13
|
+
// site. It statically pulls the six contract extractors and, through them, the
|
|
14
|
+
// native tree-sitter binding; a static import here puts all of that on MCP
|
|
15
|
+
// server startup, which never syncs.
|
|
13
16
|
import { logger } from '../logger.js';
|
|
14
17
|
function isStoredContract(raw) {
|
|
15
18
|
if (!raw || typeof raw !== 'object')
|
|
@@ -166,6 +169,12 @@ export class GroupService {
|
|
|
166
169
|
return { error: `Group "${name}" not found. Run group_list to see configured groups.` };
|
|
167
170
|
throw err;
|
|
168
171
|
}
|
|
172
|
+
// Lazy: `sync.js` reaches the six contract extractors and the native
|
|
173
|
+
// tree-sitter binding. `groupSync` is the ONLY consumer — the other seven
|
|
174
|
+
// group tools never need it — so deferring it here keeps that closure off
|
|
175
|
+
// MCP server startup entirely and off every non-sync group call. The CLI
|
|
176
|
+
// already does exactly this at `cli/group.ts`'s sync command.
|
|
177
|
+
const { syncGroup } = await import('./sync.js');
|
|
169
178
|
const result = await syncGroup(config, {
|
|
170
179
|
groupDir,
|
|
171
180
|
exactOnly: Boolean(params.exactOnly),
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Wire format of the `BasicBlock.callees` / `BasicBlock.calleeIds` cells.
|
|
3
|
+
*
|
|
4
|
+
* A LEAF module on purpose: it declares two string constants and imports
|
|
5
|
+
* nothing. `cfg/emit.ts` produces those cells and `mcp/local/pdg-impact.ts`
|
|
6
|
+
* parses them, but `emit.ts` is analyze-only and drags the whole CFG closure
|
|
7
|
+
* (reaching-defs, control-dependence, post-dominators, synthetic-escape,
|
|
8
|
+
* call-site-harvest) behind it — 8 modules evaluated at every MCP server start
|
|
9
|
+
* just to read two strings, since ESM evaluates a module to import any binding
|
|
10
|
+
* from it (#2802 review). Splitting the format constants out deletes that cost
|
|
11
|
+
* rather than deferring it, which is the same bar #2802 held its own proposals
|
|
12
|
+
* to.
|
|
13
|
+
*
|
|
14
|
+
* `emit.ts` RE-EXPORTS both names, so every existing importer keeps working and
|
|
15
|
+
* the producer/consumer pair still resolves to one definition — the drift this
|
|
16
|
+
* shared constant exists to prevent stays impossible.
|
|
17
|
+
*/
|
|
18
|
+
/**
|
|
19
|
+
* Reserved token placed in `BasicBlock.callees` when a statement's call sites
|
|
20
|
+
* were truncated at the per-statement site cap: the recorded callee list is then
|
|
21
|
+
* INCOMPLETE, so over-cap callees are absent. `*` is not a valid identifier
|
|
22
|
+
* leaf, so it cannot collide with a real callee name. The impact bridge treats a
|
|
23
|
+
* slice containing this sentinel as "callees unknown" and keeps reach
|
|
24
|
+
* callgraph-equal (proven), rather than falsely labeling an absent-but-real
|
|
25
|
+
* callee `unproven-bridge`.
|
|
26
|
+
*/
|
|
27
|
+
export declare const CALLEES_TRUNCATED_SENTINEL = "*";
|
|
28
|
+
/**
|
|
29
|
+
* Inner separator for the `BasicBlock.calleeIds` cell (resolved callee symbol
|
|
30
|
+
* ids). A TAB is used — NOT a space — because resolved ids embed `filePath` and
|
|
31
|
+
* C++ overload shape tags with multi-word primitive types (e.g. `unsigned char`,
|
|
32
|
+
* `long double`), so an id can legitimately contain a space; a space-joined cell
|
|
33
|
+
* then fragments on read and silently drops inter-procedural reach to that
|
|
34
|
+
* callee (#2227 tri-review). A tab cannot appear in a tree-sitter-derived id
|
|
35
|
+
* token (paths/identifiers/type tokens are tab-free) and round-trips intact
|
|
36
|
+
* through `escapeCSVField` (tab is in its preserved set) and the RFC-4180 COPY
|
|
37
|
+
* reader (every cell is quoted). Producer (`calleeIdsOfBlock`) and consumer
|
|
38
|
+
* (`splitCalleeIds`) both resolve to this single constant so they cannot drift.
|
|
39
|
+
* The sibling `callees` (leaf-name) cell stays space-joined — leaf names are
|
|
40
|
+
* bare identifiers and never contain a space.
|
|
41
|
+
*/
|
|
42
|
+
export declare const CALLEE_ID_SEP = "\t";
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Wire format of the `BasicBlock.callees` / `BasicBlock.calleeIds` cells.
|
|
3
|
+
*
|
|
4
|
+
* A LEAF module on purpose: it declares two string constants and imports
|
|
5
|
+
* nothing. `cfg/emit.ts` produces those cells and `mcp/local/pdg-impact.ts`
|
|
6
|
+
* parses them, but `emit.ts` is analyze-only and drags the whole CFG closure
|
|
7
|
+
* (reaching-defs, control-dependence, post-dominators, synthetic-escape,
|
|
8
|
+
* call-site-harvest) behind it — 8 modules evaluated at every MCP server start
|
|
9
|
+
* just to read two strings, since ESM evaluates a module to import any binding
|
|
10
|
+
* from it (#2802 review). Splitting the format constants out deletes that cost
|
|
11
|
+
* rather than deferring it, which is the same bar #2802 held its own proposals
|
|
12
|
+
* to.
|
|
13
|
+
*
|
|
14
|
+
* `emit.ts` RE-EXPORTS both names, so every existing importer keeps working and
|
|
15
|
+
* the producer/consumer pair still resolves to one definition — the drift this
|
|
16
|
+
* shared constant exists to prevent stays impossible.
|
|
17
|
+
*/
|
|
18
|
+
/**
|
|
19
|
+
* Reserved token placed in `BasicBlock.callees` when a statement's call sites
|
|
20
|
+
* were truncated at the per-statement site cap: the recorded callee list is then
|
|
21
|
+
* INCOMPLETE, so over-cap callees are absent. `*` is not a valid identifier
|
|
22
|
+
* leaf, so it cannot collide with a real callee name. The impact bridge treats a
|
|
23
|
+
* slice containing this sentinel as "callees unknown" and keeps reach
|
|
24
|
+
* callgraph-equal (proven), rather than falsely labeling an absent-but-real
|
|
25
|
+
* callee `unproven-bridge`.
|
|
26
|
+
*/
|
|
27
|
+
export const CALLEES_TRUNCATED_SENTINEL = '*';
|
|
28
|
+
/**
|
|
29
|
+
* Inner separator for the `BasicBlock.calleeIds` cell (resolved callee symbol
|
|
30
|
+
* ids). A TAB is used — NOT a space — because resolved ids embed `filePath` and
|
|
31
|
+
* C++ overload shape tags with multi-word primitive types (e.g. `unsigned char`,
|
|
32
|
+
* `long double`), so an id can legitimately contain a space; a space-joined cell
|
|
33
|
+
* then fragments on read and silently drops inter-procedural reach to that
|
|
34
|
+
* callee (#2227 tri-review). A tab cannot appear in a tree-sitter-derived id
|
|
35
|
+
* token (paths/identifiers/type tokens are tab-free) and round-trips intact
|
|
36
|
+
* through `escapeCSVField` (tab is in its preserved set) and the RFC-4180 COPY
|
|
37
|
+
* reader (every cell is quoted). Producer (`calleeIdsOfBlock`) and consumer
|
|
38
|
+
* (`splitCalleeIds`) both resolve to this single constant so they cannot drift.
|
|
39
|
+
* The sibling `callees` (leaf-name) cell stays space-joined — leaf names are
|
|
40
|
+
* bare identifiers and never contain a space.
|
|
41
|
+
*/
|
|
42
|
+
export const CALLEE_ID_SEP = '\t';
|
|
@@ -22,30 +22,15 @@ import type { KnowledgeGraph } from '../../graph/types.js';
|
|
|
22
22
|
import { type ReachingDefsSolver } from './reaching-defs.js';
|
|
23
23
|
import type { BasicBlockData, BindingEntry, FunctionCfg } from './types.js';
|
|
24
24
|
/**
|
|
25
|
-
*
|
|
26
|
-
*
|
|
27
|
-
*
|
|
28
|
-
*
|
|
29
|
-
*
|
|
30
|
-
*
|
|
31
|
-
*
|
|
25
|
+
* Cell-format constants live in the LEAF module `callee-cell-format.ts` and are
|
|
26
|
+
* re-exported here so every existing importer keeps working. The consumer side
|
|
27
|
+
* (`mcp/local/pdg-impact.ts`) imports them from the leaf directly: importing any
|
|
28
|
+
* binding from THIS module evaluates it, and with it the whole analyze-only CFG
|
|
29
|
+
* closure — 8 modules on every MCP server start to read two strings (#2802
|
|
30
|
+
* review). Producer and consumer still resolve to one definition, so the drift
|
|
31
|
+
* these shared constants exist to prevent stays impossible.
|
|
32
32
|
*/
|
|
33
|
-
export
|
|
34
|
-
/**
|
|
35
|
-
* Inner separator for the `BasicBlock.calleeIds` cell (resolved callee symbol
|
|
36
|
-
* ids). A TAB is used — NOT a space — because resolved ids embed `filePath` and
|
|
37
|
-
* C++ overload shape tags with multi-word primitive types (e.g. `unsigned char`,
|
|
38
|
-
* `long double`), so an id can legitimately contain a space; a space-joined cell
|
|
39
|
-
* then fragments on read and silently drops inter-procedural reach to that
|
|
40
|
-
* callee (#2227 tri-review). A tab cannot appear in a tree-sitter-derived id
|
|
41
|
-
* token (paths/identifiers/type tokens are tab-free) and round-trips intact
|
|
42
|
-
* through `escapeCSVField` (tab is in its preserved set) and the RFC-4180 COPY
|
|
43
|
-
* reader (every cell is quoted). Producer ({@link calleeIdsOfBlock}) and
|
|
44
|
-
* consumer (`splitCalleeIds`) import this single constant so they cannot drift.
|
|
45
|
-
* The sibling `callees` (leaf-name) cell stays space-joined — leaf names are
|
|
46
|
-
* bare identifiers and never contain a space.
|
|
47
|
-
*/
|
|
48
|
-
export declare const CALLEE_ID_SEP = "\t";
|
|
33
|
+
export { CALLEES_TRUNCATED_SENTINEL, CALLEE_ID_SEP } from './callee-cell-format.js';
|
|
49
34
|
/**
|
|
50
35
|
* Default per-function CFG edge cap. A pathological generated function could
|
|
51
36
|
* otherwise emit an unbounded edge set; the cap bounds graph growth and is
|
|
@@ -6,31 +6,17 @@ import { augmentForPostDom } from './synthetic-escape.js';
|
|
|
6
6
|
import { DEFAULT_PDG_MAX_SITES_PER_STATEMENT } from './visitors/call-site-harvest.js';
|
|
7
7
|
import { calleeIdPosKey } from '../scope-resolution/graph-bridge/callee-id-sink.js';
|
|
8
8
|
import { encodeReachingDefReasonPairs } from './reaching-def-reason-codec.js';
|
|
9
|
+
import { CALLEES_TRUNCATED_SENTINEL, CALLEE_ID_SEP } from './callee-cell-format.js';
|
|
9
10
|
/**
|
|
10
|
-
*
|
|
11
|
-
*
|
|
12
|
-
*
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
-
*
|
|
16
|
-
*
|
|
11
|
+
* Cell-format constants live in the LEAF module `callee-cell-format.ts` and are
|
|
12
|
+
* re-exported here so every existing importer keeps working. The consumer side
|
|
13
|
+
* (`mcp/local/pdg-impact.ts`) imports them from the leaf directly: importing any
|
|
14
|
+
* binding from THIS module evaluates it, and with it the whole analyze-only CFG
|
|
15
|
+
* closure — 8 modules on every MCP server start to read two strings (#2802
|
|
16
|
+
* review). Producer and consumer still resolve to one definition, so the drift
|
|
17
|
+
* these shared constants exist to prevent stays impossible.
|
|
17
18
|
*/
|
|
18
|
-
export
|
|
19
|
-
/**
|
|
20
|
-
* Inner separator for the `BasicBlock.calleeIds` cell (resolved callee symbol
|
|
21
|
-
* ids). A TAB is used — NOT a space — because resolved ids embed `filePath` and
|
|
22
|
-
* C++ overload shape tags with multi-word primitive types (e.g. `unsigned char`,
|
|
23
|
-
* `long double`), so an id can legitimately contain a space; a space-joined cell
|
|
24
|
-
* then fragments on read and silently drops inter-procedural reach to that
|
|
25
|
-
* callee (#2227 tri-review). A tab cannot appear in a tree-sitter-derived id
|
|
26
|
-
* token (paths/identifiers/type tokens are tab-free) and round-trips intact
|
|
27
|
-
* through `escapeCSVField` (tab is in its preserved set) and the RFC-4180 COPY
|
|
28
|
-
* reader (every cell is quoted). Producer ({@link calleeIdsOfBlock}) and
|
|
29
|
-
* consumer (`splitCalleeIds`) import this single constant so they cannot drift.
|
|
30
|
-
* The sibling `callees` (leaf-name) cell stays space-joined — leaf names are
|
|
31
|
-
* bare identifiers and never contain a space.
|
|
32
|
-
*/
|
|
33
|
-
export const CALLEE_ID_SEP = '\t';
|
|
19
|
+
export { CALLEES_TRUNCATED_SENTINEL, CALLEE_ID_SEP } from './callee-cell-format.js';
|
|
34
20
|
/**
|
|
35
21
|
* Default per-function CFG edge cap. A pathological generated function could
|
|
36
22
|
* otherwise emit an unbounded edge set; the cap bounds graph growth and is
|
|
@@ -10,6 +10,12 @@ import { escapeCypherString } from './cypher-escape.js';
|
|
|
10
10
|
import { withConnLock } from './conn-lock.js';
|
|
11
11
|
import { isWalDriverActive } from './wal-driver-state.js';
|
|
12
12
|
import { NODE_TABLES, REL_TABLE_NAME, SCHEMA_QUERIES, EMBEDDING_TABLE_NAME, CREATE_VECTOR_INDEX_QUERY, STALE_HASH_SENTINEL, } from './schema.js';
|
|
13
|
+
// Analyze-only, but reached from MCP startup via `pool-adapter.js`. #2802
|
|
14
|
+
// proposed lazy-importing it; rejected — `core/search/bm25-index.ts` statically
|
|
15
|
+
// imports `normalizeFtsText` from `csv-generator.js`, and `local-backend.ts`
|
|
16
|
+
// dynamically imports bm25-index on the FTS query path, so deferring here
|
|
17
|
+
// relocates the startup cost to first query rather than removing it. The
|
|
18
|
+
// measured figures live in #2802; they were environment-bound, this is not.
|
|
13
19
|
import { streamAllCSVsToDisk } from './csv-generator.js';
|
|
14
20
|
import { getNodeLabel as deriveNodeLabel } from './rel-pair-routing.js';
|
|
15
21
|
import { EMBEDDABLE_LABELS } from '../embeddings/types.js';
|
package/dist/core/run-analyze.js
CHANGED
|
@@ -382,8 +382,9 @@ export const pdgModeMismatch = (recorded, options) => {
|
|
|
382
382
|
// different runs would always be `!==`, tripping pdgModeMismatch on every
|
|
383
383
|
// re-analyze and forcing a needless full writeback. e.g. do NOT change
|
|
384
384
|
// `hasCallSummary: true` to a per-language object like `{ ts: true, ... }`; keep
|
|
385
|
-
//
|
|
386
|
-
//
|
|
385
|
+
// any diagnostic refinement in the impact CONSUMER (see pdg-impact.ts
|
|
386
|
+
// assemblePdgImpactResult, which reports empty ascent from the persisted
|
|
387
|
+
// CALL_SUMMARY data), not in this version discriminator.
|
|
387
388
|
for (const key of new Set([...Object.keys(reqRecord), ...Object.keys(recRecord)])) {
|
|
388
389
|
if (reqRecord[key] !== recRecord[key])
|
|
389
390
|
return true;
|
|
@@ -19,15 +19,10 @@ import { loadMeta } from '../../storage/repo-manager.js';
|
|
|
19
19
|
*/
|
|
20
20
|
export declare function fnLineOf(id: string): number;
|
|
21
21
|
/**
|
|
22
|
-
*
|
|
23
|
-
*
|
|
24
|
-
*
|
|
25
|
-
*
|
|
26
|
-
*
|
|
27
|
-
* Extracted here (U1) so the two callers — `LocalBackend.calleeIdsOfBlocks` (the
|
|
28
|
-
* statement-precise bridge key) and the inter-procedural descent's
|
|
29
|
-
* `calleeIdsFromCalleeRows` — cannot diverge on the split-and-drop-sentinel
|
|
30
|
-
* logic. Both consume rows of `BasicBlock.calleeIds`; this is the single source.
|
|
22
|
+
* Ids-only view of {@link parseCalleeIdsCell}. Exported (U1) so the two callers —
|
|
23
|
+
* `LocalBackend.calleeIdsOfBlocks` (the statement-precise bridge key) and the
|
|
24
|
+
* inter-procedural descent — cannot diverge on the split-and-drop-sentinel logic.
|
|
25
|
+
* Both consume rows of `BasicBlock.calleeIds`; this is the single source.
|
|
31
26
|
*/
|
|
32
27
|
export declare function splitCalleeIds(raw: unknown): string[];
|
|
33
28
|
/**
|
|
@@ -76,6 +71,127 @@ export interface PdgImpactParityFields {
|
|
|
76
71
|
affected_modules: unknown[];
|
|
77
72
|
}
|
|
78
73
|
export type PdgImpactEvidence = 'local-dependence' | 'owner-projection' | 'callgraph-bridge' | 'unproven-bridge' | 'degraded';
|
|
74
|
+
/**
|
|
75
|
+
* WHY the callee set the descent EXAMINED for `CALL_SUMMARY` return-flows is a
|
|
76
|
+
* strict PREFIX of the slice's real callee list. A STRUCTURED vocabulary, in the
|
|
77
|
+
* spirit of `truncatedByReasons: readonly ('depth'|'limit')[]` — the codes are the
|
|
78
|
+
* contract, the English phrasing is a rendering of them
|
|
79
|
+
* ({@link ASCENT_INCOMPLETE_PHRASE}). A caller branches on the code; only the note
|
|
80
|
+
* reads the phrase, so a rewording can never break a consumer.
|
|
81
|
+
* - `'traversal-truncated'` — the traversal stopped at its depth/size budget, so
|
|
82
|
+
* a callee that DOES carry a return-flow can sit in a hop never reached (the
|
|
83
|
+
* same fact the result's `truncated`/`truncatedByReasons` report, read at the
|
|
84
|
+
* ascent's granularity).
|
|
85
|
+
* - `'callee-list-capped'` — a slice block's `calleeIds` cell was capped at emit;
|
|
86
|
+
* `parseCalleeIdsCell` strips the sentinel, so the dropped ids are invisible to
|
|
87
|
+
* BOTH the summary scan and the counters.
|
|
88
|
+
* - `'callee-ids-unrecorded'` — a slice block records CALL SITES (a non-empty
|
|
89
|
+
* `callees` name cell) but NO resolved callee ids. An empty `calleeIds` cell
|
|
90
|
+
* carries no sentinel, so those call sites are invisible to the summary scan
|
|
91
|
+
* without raising `'callee-list-capped'`. Distinct from the cap: nothing was
|
|
92
|
+
* dropped at emit — the ids were never recorded.
|
|
93
|
+
*
|
|
94
|
+
* THREE producer paths yield it, and the consumer cannot tell them apart —
|
|
95
|
+
* do not read this code as naming any one of them (`cfg/emit.ts`,
|
|
96
|
+
* `calleeIdsOfBlock`):
|
|
97
|
+
* 1. the file's resolved-id map is absent entirely (`fileMap === undefined`);
|
|
98
|
+
* 2. a call site has no position anchor;
|
|
99
|
+
* 3. a call site's position IS in the map but did not RESOLVE.
|
|
100
|
+
* (3) is the ordinary one — it is exactly the receiver-resolution gaps this
|
|
101
|
+
* repo pins (e.g. #2807's inference-typed field receivers, where `calleeIds`
|
|
102
|
+
* empties while `calleesOfBlock` still writes the leaf names). So on a real
|
|
103
|
+
* index this fires broadly and is driven by resolution quality, NOT by a
|
|
104
|
+
* missing `--pdg` layer: "re-run analyze --pdg" is the wrong remedy for it,
|
|
105
|
+
* and `examinedComplete: false` here is a statement about how much of the
|
|
106
|
+
* call graph resolved, not about the traversal giving up.
|
|
107
|
+
*
|
|
108
|
+
* Consequence worth knowing before branching on it: because (3) is common,
|
|
109
|
+
* `examinedComplete: true` is the strong, rare signal and `false` is close to
|
|
110
|
+
* the default on a large repo. Distinguishing the three needs a marker at
|
|
111
|
+
* emit time, which would move the persisted cell format — deliberately out of
|
|
112
|
+
* scope here, and tracked separately.
|
|
113
|
+
*/
|
|
114
|
+
export type PdgAscentIncompleteReason = 'traversal-truncated' | 'callee-list-capped' | 'callee-ids-unrecorded';
|
|
115
|
+
/**
|
|
116
|
+
* Return-value-ascent coverage, published on {@link PdgImpactEvidenceSummary} so a
|
|
117
|
+
* consumer can answer "was the ascent complete, and if not why" WITHOUT parsing
|
|
118
|
+
* the result `note`. This is MCP output read by agents: the same four facts are
|
|
119
|
+
* also narrated in the note (see {@link AscentCoverage} for the canonical
|
|
120
|
+
* rationale, and `assemblePdgImpactResult` for the single site that renders both
|
|
121
|
+
* from one computation), and the prose is the human surface, not the contract.
|
|
122
|
+
*
|
|
123
|
+
* Present iff the inter-procedural descent RAN (a downstream slice that reached
|
|
124
|
+
* the assembly path). Absent ⇒ nothing was scanned — deliberately not a zeroed
|
|
125
|
+
* object, which would read as "we looked and found nothing".
|
|
126
|
+
*/
|
|
127
|
+
export interface PdgAscentCoverage {
|
|
128
|
+
/**
|
|
129
|
+
* Count of DISTINCT callee ids scanned for a `CALL_SUMMARY` — a distinct-callee
|
|
130
|
+
* tally, NOT a call-site count: two slice blocks invoking the same callee
|
|
131
|
+
* contribute 1, and one block invoking it twice also contributes 1. That is the
|
|
132
|
+
* correct population for the claim `returnFlowFound` makes, because a
|
|
133
|
+
* `CALL_SUMMARY` is a property of the CALLEE, not of the call site.
|
|
134
|
+
*
|
|
135
|
+
* Callee granularity, NOT "callees resolved to a body": a cell's ids that
|
|
136
|
+
* `resolveCalleeSpans` never matches (out-of-repo target, interface method, the
|
|
137
|
+
* `Class:` id a `new` expression contributes) are scanned all the same. See
|
|
138
|
+
* {@link AscentCoverage} POPULATION. (The field NAME is historical — the
|
|
139
|
+
* published shape is versioned by `pdgResultVersion`, so it is kept while the
|
|
140
|
+
* prose on both surfaces says "distinct callees".)
|
|
141
|
+
*/
|
|
142
|
+
referencesScanned: number;
|
|
143
|
+
/**
|
|
144
|
+
* Whether ANY scanned callee carried a DECODED non-empty return-flow — i.e.
|
|
145
|
+
* whether the ascent FIRED anywhere in this slice. `false` with
|
|
146
|
+
* `referencesScanned > 0` is the structural counterpart of the note's
|
|
147
|
+
* "no return-value ascent in this slice" sentence.
|
|
148
|
+
*/
|
|
149
|
+
returnFlowFound: boolean;
|
|
150
|
+
/**
|
|
151
|
+
* Scanned callees whose `CALL_SUMMARY` the codec could not decode (version skew
|
|
152
|
+
* / corruption / NULL reason). Each withholds the ascent exactly like an empty
|
|
153
|
+
* summary, so a non-zero count means `returnFlowFound: false` is NOT a statement
|
|
154
|
+
* about what the persisted summaries record — remedy: re-run `analyze --pdg`.
|
|
155
|
+
*/
|
|
156
|
+
undecodableSummaryCount: number;
|
|
157
|
+
/**
|
|
158
|
+
* Whether {@link referencesScanned} ranges over EVERY callee the descent's slice
|
|
159
|
+
* blocks recorded a resolved id for. `false` ⇒ the counters above range over a
|
|
160
|
+
* strict subset, so `returnFlowFound: false` is not a whole-slice claim. Reasons
|
|
161
|
+
* in {@link incompleteReasons}.
|
|
162
|
+
*
|
|
163
|
+
* SCOPE — the population is "callees the INDEX recorded resolved ids for on the
|
|
164
|
+
* blocks the DESCENT visited", never "every call the source text makes". Two
|
|
165
|
+
* gaps are named structurally rather than assumed away: a block whose ids the
|
|
166
|
+
* emitter capped raises `'callee-list-capped'`, and a block that records call
|
|
167
|
+
* sites but no ids at all raises `'callee-ids-unrecorded'`. What is NOT modelled
|
|
168
|
+
* (and cannot be, from the persisted graph) is a call the CFG never materialised
|
|
169
|
+
* as a call site — so `true` means "nothing the index recorded was skipped", not
|
|
170
|
+
* "the program makes no other calls".
|
|
171
|
+
*/
|
|
172
|
+
examinedComplete: boolean;
|
|
173
|
+
/** Empty iff {@link examinedComplete}; otherwise every mechanism that fired. */
|
|
174
|
+
incompleteReasons: readonly PdgAscentIncompleteReason[];
|
|
175
|
+
/**
|
|
176
|
+
* Whether the index carries the `CALL_SUMMARY` layer at all. `false` ⇒ a PRE-FU-C
|
|
177
|
+
* (v3) `--pdg` index, where the scan COULD NOT have found a return-flow — without
|
|
178
|
+
* this a consumer would read `returnFlowFound: false` as "these callees record no
|
|
179
|
+
* return-flow" when the truth is "the layer that records it does not exist here"
|
|
180
|
+
* (the note distinguishes the two in prose; this keeps the structured surface
|
|
181
|
+
* from being false-safe). Remedy: re-run `analyze --pdg`.
|
|
182
|
+
*
|
|
183
|
+
* READING IT WITH THE OTHER FIELDS. `{referencesScanned: N>0, returnFlowFound:
|
|
184
|
+
* false, callSummaryLayerPresent: false}` is SELF-CONSISTENT and expected on a
|
|
185
|
+
* v3 index, not a contradiction: the scan really did run over N callees and
|
|
186
|
+
* really did find nothing, because there was no layer in which a return-flow
|
|
187
|
+
* could be recorded. Read this field FIRST — while it is `false`,
|
|
188
|
+
* `returnFlowFound` and `undecodableSummaryCount` say nothing about the callees
|
|
189
|
+
* themselves and must not be used to conclude "no callee returns a
|
|
190
|
+
* slice-dependent value". `examinedComplete` is orthogonal to all of this: it
|
|
191
|
+
* reports coverage of the callee POPULATION, never the presence of the layer.
|
|
192
|
+
*/
|
|
193
|
+
callSummaryLayerPresent: boolean;
|
|
194
|
+
}
|
|
79
195
|
export interface PdgImpactEvidenceSummary {
|
|
80
196
|
statements?: PdgImpactEvidence;
|
|
81
197
|
localSymbols?: PdgImpactEvidence;
|
|
@@ -84,6 +200,12 @@ export interface PdgImpactEvidenceSummary {
|
|
|
84
200
|
unresolvedBlockCount?: number;
|
|
85
201
|
ambiguousProjectionCount?: number;
|
|
86
202
|
interproceduralEvidenceCounts?: Partial<Record<PdgImpactEvidence, number>>;
|
|
203
|
+
/**
|
|
204
|
+
* Return-value-ascent coverage — the same "counts + classification" kind as the
|
|
205
|
+
* three counters above, scoped to the U-C4 ascent. Optional because the descent
|
|
206
|
+
* does not run for every slice; see {@link PdgAscentCoverage}.
|
|
207
|
+
*/
|
|
208
|
+
ascent?: PdgAscentCoverage;
|
|
87
209
|
}
|
|
88
210
|
export interface PdgInterproceduralImpact {
|
|
89
211
|
engine: 'symbol-graph';
|
|
@@ -8,12 +8,15 @@
|
|
|
8
8
|
import path from 'path';
|
|
9
9
|
import { loadMeta } from '../../storage/repo-manager.js';
|
|
10
10
|
import { IMPACT_MAX_DEPTH, PDG_QUERY_DEFAULT_LIMIT, PDG_QUERY_MAX_LIMIT } from '../tools.js';
|
|
11
|
-
|
|
11
|
+
// Imported from the LEAF `callee-cell-format.js`, NOT from `cfg/emit.js` which
|
|
12
|
+
// re-exports them: ESM evaluates a module to import any binding from it, and
|
|
13
|
+
// `emit.ts` drags the analyze-only CFG closure (reaching-defs, control-
|
|
14
|
+
// dependence, post-dominators, synthetic-escape, call-site-harvest) with it —
|
|
15
|
+
// 8 modules on every MCP server start, to read two strings (#2802 review).
|
|
16
|
+
import { CALLEES_TRUNCATED_SENTINEL, CALLEE_ID_SEP, } from '../../core/ingestion/cfg/callee-cell-format.js';
|
|
12
17
|
import { toDisplayLine } from './line-display.js';
|
|
13
18
|
import { decodeCallSummary } from '../../core/ingestion/taint/call-summary-codec.js';
|
|
14
19
|
import { decodeReachingDefReason } from '../../core/ingestion/cfg/reaching-def-reason-codec.js';
|
|
15
|
-
import { getProviderForFile } from '../../core/ingestion/languages/index.js';
|
|
16
|
-
import { SupportedLanguages } from '../../_shared/index.js';
|
|
17
20
|
/**
|
|
18
21
|
* Parse the `<fnLine>` segment out of a `BasicBlock` id (1-based function start
|
|
19
22
|
* line). The id template is
|
|
@@ -60,26 +63,40 @@ const INTERPROC_DEPTH_BUDGET = 3;
|
|
|
60
63
|
*/
|
|
61
64
|
const INTERPROC_NODE_BUDGET = 5000;
|
|
62
65
|
/**
|
|
63
|
-
*
|
|
64
|
-
* symbol ids
|
|
65
|
-
*
|
|
66
|
-
*
|
|
66
|
+
* Parse a tab-joined ({@link CALLEE_ID_SEP}) `BasicBlock.calleeIds` cell in ONE
|
|
67
|
+
* pass into its resolved callee symbol ids plus whether the cell was CAPPED at
|
|
68
|
+
* emit. The truncation sentinel is NOT a resolved symbol id and must never enter
|
|
69
|
+
* a `has(realId)` set, so it is dropped from `ids` — which is exactly why
|
|
70
|
+
* `truncated` has to come back alongside them: without it a capped block is
|
|
71
|
+
* indistinguishable from a complete one and the dropped callees are invisible to
|
|
72
|
+
* every consumer that only reads `ids` (they reach neither the `CALL_SUMMARY`
|
|
73
|
+
* scan nor the ascent counters). Empty/whitespace cells yield no ids.
|
|
67
74
|
*
|
|
68
|
-
*
|
|
69
|
-
*
|
|
70
|
-
* `calleeIdsFromCalleeRows` — cannot diverge on the split-and-drop-sentinel
|
|
71
|
-
* logic. Both consume rows of `BasicBlock.calleeIds`; this is the single source.
|
|
75
|
+
* The callgraph bridge already treats a capped block as callee-INCOMPLETE (see
|
|
76
|
+
* `classifyPdgBridgeEvidence`); this is the same fact, read at the descent side.
|
|
72
77
|
*/
|
|
73
|
-
|
|
74
|
-
const
|
|
78
|
+
function parseCalleeIdsCell(raw) {
|
|
79
|
+
const ids = [];
|
|
80
|
+
let truncated = false;
|
|
75
81
|
// Split on the SHARED CALLEE_ID_SEP (tab) — ids embed file paths / multi-word
|
|
76
82
|
// C++ type tokens that can contain a space, so a space split would fragment
|
|
77
83
|
// them. Producer (calleeIdsOfBlock) joins with the same constant.
|
|
78
84
|
for (const id of String(raw ?? '').split(CALLEE_ID_SEP)) {
|
|
79
|
-
if (id
|
|
80
|
-
|
|
85
|
+
if (id === CALLEES_TRUNCATED_SENTINEL)
|
|
86
|
+
truncated = true;
|
|
87
|
+
else if (id)
|
|
88
|
+
ids.push(id);
|
|
81
89
|
}
|
|
82
|
-
return
|
|
90
|
+
return { ids, truncated };
|
|
91
|
+
}
|
|
92
|
+
/**
|
|
93
|
+
* Ids-only view of {@link parseCalleeIdsCell}. Exported (U1) so the two callers —
|
|
94
|
+
* `LocalBackend.calleeIdsOfBlocks` (the statement-precise bridge key) and the
|
|
95
|
+
* inter-procedural descent — cannot diverge on the split-and-drop-sentinel logic.
|
|
96
|
+
* Both consume rows of `BasicBlock.calleeIds`; this is the single source.
|
|
97
|
+
*/
|
|
98
|
+
export function splitCalleeIds(raw) {
|
|
99
|
+
return parseCalleeIdsCell(raw).ids;
|
|
83
100
|
}
|
|
84
101
|
/**
|
|
85
102
|
* Contract version of the mode:'pdg' impact result shape. A stable discriminator
|
|
@@ -474,6 +491,56 @@ export function makePdgLayerDegradedResult(input) {
|
|
|
474
491
|
...emptyPdgParityFields(),
|
|
475
492
|
};
|
|
476
493
|
}
|
|
494
|
+
/**
|
|
495
|
+
* Render table for {@link PdgAscentIncompleteReason} — the ONLY place a code
|
|
496
|
+
* becomes English. Keeping the mapping here (rather than building sentences at
|
|
497
|
+
* the point the mechanism is detected) is what lets the published vocabulary and
|
|
498
|
+
* the note's wording move independently: a reworded phrase is invisible to every
|
|
499
|
+
* consumer branching on the code, and a new code cannot silently change the
|
|
500
|
+
* joiner the existing sentence uses.
|
|
501
|
+
*/
|
|
502
|
+
const ASCENT_INCOMPLETE_PHRASE = {
|
|
503
|
+
'traversal-truncated': 'the traversal stopped at its depth/size budget',
|
|
504
|
+
'callee-list-capped': "a slice block's call-site list was capped at emit",
|
|
505
|
+
'callee-ids-unrecorded': 'a slice block records call sites but no resolved callee ids',
|
|
506
|
+
};
|
|
507
|
+
/**
|
|
508
|
+
* The ONE place {@link AscentCoverage} plus the result-level truncation flag
|
|
509
|
+
* become the published {@link PdgAscentIncompleteReason} codes. Extracted so the
|
|
510
|
+
* two exits that publish coverage — `assemblePdgImpactResult` (the slice result,
|
|
511
|
+
* which also renders the codes into the note) and `runImpactPDG`'s empty-slice
|
|
512
|
+
* return — cannot classify the same descent differently.
|
|
513
|
+
*
|
|
514
|
+
* Emission order is the array order below and is part of what the note renders,
|
|
515
|
+
* so a new code appends rather than inserts.
|
|
516
|
+
*/
|
|
517
|
+
function ascentIncompleteReasonsOf(input) {
|
|
518
|
+
const reasons = [];
|
|
519
|
+
if (input.truncated)
|
|
520
|
+
reasons.push('traversal-truncated');
|
|
521
|
+
if (input.coverage?.listTruncated === true)
|
|
522
|
+
reasons.push('callee-list-capped');
|
|
523
|
+
if (input.coverage?.idlessCallSites === true)
|
|
524
|
+
reasons.push('callee-ids-unrecorded');
|
|
525
|
+
return reasons;
|
|
526
|
+
}
|
|
527
|
+
/**
|
|
528
|
+
* Project the descent's {@link AscentCoverage} onto the published
|
|
529
|
+
* {@link PdgAscentCoverage}. Shared by both exits that publish `pdgEvidence.ascent`
|
|
530
|
+
* so the contract sentence "present iff the inter-procedural descent ran" holds on
|
|
531
|
+
* BOTH — a descent that ran and scanned callees must not go unreported merely
|
|
532
|
+
* because the slice happened to reach no DISTINCT downstream block.
|
|
533
|
+
*/
|
|
534
|
+
function publishedAscentCoverage(input) {
|
|
535
|
+
return {
|
|
536
|
+
referencesScanned: input.coverage.references,
|
|
537
|
+
returnFlowFound: input.coverage.anyReturnFlow,
|
|
538
|
+
undecodableSummaryCount: input.coverage.undecodable,
|
|
539
|
+
examinedComplete: input.incompleteReasons.length === 0,
|
|
540
|
+
incompleteReasons: input.incompleteReasons,
|
|
541
|
+
callSummaryLayerPresent: input.callSummaryAvailable,
|
|
542
|
+
};
|
|
543
|
+
}
|
|
477
544
|
/**
|
|
478
545
|
* Assemble the consumer-safe PDG impact result (U4 / KTD8 parity matrix).
|
|
479
546
|
*
|
|
@@ -537,6 +604,20 @@ function assemblePdgImpactResult(input) {
|
|
|
537
604
|
const impactedCount = resolvedUids.size;
|
|
538
605
|
const byDepth = items.length > 0 ? { 1: items } : {};
|
|
539
606
|
const byDepthCounts = { 1: items.length };
|
|
607
|
+
// ── Ascent coverage: ONE computation, TWO surfaces ─────────────────────────
|
|
608
|
+
// The empty-ascent sentence quantifies UNIVERSALLY over the callees the descent
|
|
609
|
+
// actually EXAMINED, and three mechanisms can make that set a strict
|
|
610
|
+
// subset of the slice's real callee list (rationale: AscentCoverage
|
|
611
|
+
// INCOMPLETENESS, vocabulary: PdgAscentIncompleteReason). Classified ONCE by the
|
|
612
|
+
// shared `ascentIncompleteReasonsOf` and consumed twice — by `pdgEvidence.ascent`
|
|
613
|
+
// (structured, the contract) and by the note's qualifier clause (prose, rendered
|
|
614
|
+
// through ASCENT_INCOMPLETE_PHRASE). Deriving both from one array is what stops
|
|
615
|
+
// an agent branching on the codes and a human reading the note from ever
|
|
616
|
+
// disagreeing.
|
|
617
|
+
const ascentIncompleteReasons = ascentIncompleteReasonsOf({
|
|
618
|
+
truncated: input.truncated,
|
|
619
|
+
coverage: input.ascentCoverage,
|
|
620
|
+
});
|
|
540
621
|
const noteParts = statementMode
|
|
541
622
|
? [
|
|
542
623
|
`mode:'pdg' — intra-procedural slice from line ${input.criterionLine} of ` +
|
|
@@ -579,19 +660,57 @@ function assemblePdgImpactResult(input) {
|
|
|
579
660
|
`CALL_SUMMARY edges and enable it.`);
|
|
580
661
|
}
|
|
581
662
|
else if (input.callSummaryAvailable === true) {
|
|
582
|
-
// The CALL_SUMMARY layer is present, but
|
|
583
|
-
//
|
|
584
|
-
//
|
|
585
|
-
//
|
|
586
|
-
//
|
|
587
|
-
//
|
|
588
|
-
//
|
|
589
|
-
|
|
590
|
-
|
|
591
|
-
|
|
592
|
-
|
|
593
|
-
|
|
594
|
-
|
|
663
|
+
// The CALL_SUMMARY layer is present, but that only means the index CAN
|
|
664
|
+
// carry return-flow summaries — not that the callees in THIS slice have
|
|
665
|
+
// one. When none of them does, the ascent is structurally empty and the
|
|
666
|
+
// note says so, rather than letting the omission read as "ascent ran and
|
|
667
|
+
// found nothing". Sound — never claims the ascent fired. Keyed on the
|
|
668
|
+
// OBSERVED summaries, never on the criterion's language (#2802) — see
|
|
669
|
+
// {@link AscentCoverage} for why, and for the population the sentence below
|
|
670
|
+
// quantifies over: DISTINCT CALLEES (a `Set` of callee ids — two call sites
|
|
671
|
+
// to the same callee count once), NOT callees resolved to a body. Hence the
|
|
672
|
+
// "distinct callee(s)" wording; a call-SITE count would over-state the set.
|
|
673
|
+
const coverage = input.ascentCoverage;
|
|
674
|
+
const references = coverage?.references ?? 0;
|
|
675
|
+
const undecodable = coverage?.undecodable ?? 0;
|
|
676
|
+
// When the examined set is a strict prefix (the codes computed once above)
|
|
677
|
+
// the claim is qualified: the note may describe what was examined, never
|
|
678
|
+
// assert a property of the whole slice the traversal did not establish. The
|
|
679
|
+
// codes are mapped to phrases HERE — the note is a rendering of the same
|
|
680
|
+
// vocabulary `pdgEvidence.ascent.incompleteReasons` publishes.
|
|
681
|
+
const examinedIncomplete = ascentIncompleteReasons.length > 0;
|
|
682
|
+
const incompleteClause = examinedIncomplete
|
|
683
|
+
? ` (${ascentIncompleteReasons
|
|
684
|
+
.map((reason) => ASCENT_INCOMPLETE_PHRASE[reason])
|
|
685
|
+
.join(' and ')}, so callees past the examined set were not checked)`
|
|
686
|
+
: '';
|
|
687
|
+
if (references > 0 && coverage?.anyReturnFlow !== true) {
|
|
688
|
+
// ONE head for both arms — a shared gate and a shared opening sentence, so
|
|
689
|
+
// the two cannot drift on wording or pluralization. When at least one
|
|
690
|
+
// summary could not be DECODED the note must not assert what the persisted
|
|
691
|
+
// summaries record: an undecodable `reason` may well encode a return-flow
|
|
692
|
+
// this reader cannot unpack (`decodeCallSummary` never throws, so a
|
|
693
|
+
// version-skewed / corrupt / NULL reason is otherwise indistinguishable
|
|
694
|
+
// from a cleanly-decoded empty one). The ascent is withheld either way;
|
|
695
|
+
// only the tail that explains it changes.
|
|
696
|
+
noteParts.push(`no return-value ascent in this slice: none of the ${references} distinct ` +
|
|
697
|
+
`${references === 1 ? 'callee carries' : 'callees carry'} a ` +
|
|
698
|
+
`${undecodable > 0 ? 'decodable ' : ''}CALL_SUMMARY return-flow${incompleteClause}` +
|
|
699
|
+
(undecodable > 0
|
|
700
|
+
? `, and ${undecodable} callee ` +
|
|
701
|
+
`${undecodable === 1 ? 'summary' : 'summaries'} could not be decoded (version ` +
|
|
702
|
+
`skew or corruption) — re-run gitnexus analyze --pdg to rebuild them. A caller ` +
|
|
703
|
+
`statement depending on a callee's RETURN value is not in the slice; descent and ` +
|
|
704
|
+
`the intra slice are unaffected.`
|
|
705
|
+
: `. So a caller statement depending on a callee's RETURN value is ` +
|
|
706
|
+
`not in the slice. ` +
|
|
707
|
+
(examinedIncomplete
|
|
708
|
+
? `Every summary examined decoded, so this is a property of those summaries, not `
|
|
709
|
+
: `Every summary in this slice decoded, so this is a property of the persisted ` +
|
|
710
|
+
`summaries, not `) +
|
|
711
|
+
`of the criterion's language — a callee whose producer records no formal index and ` +
|
|
712
|
+
`one with genuinely no return-flow are indistinguishable here. Descent and the ` +
|
|
713
|
+
`intra slice are unaffected.`));
|
|
595
714
|
}
|
|
596
715
|
}
|
|
597
716
|
}
|
|
@@ -625,6 +744,28 @@ function assemblePdgImpactResult(input) {
|
|
|
625
744
|
localSymbolCount: impactedCount,
|
|
626
745
|
unresolvedBlockCount: unresolvedCount,
|
|
627
746
|
ambiguousProjectionCount: ambiguousCount,
|
|
747
|
+
// Structured ascent coverage — the note's facts, published so a caller never
|
|
748
|
+
// has to regex prose to learn whether the ascent was complete. Emitted iff
|
|
749
|
+
// the DESCENT RAN (`ascentCoverage` present), which is exactly the contract
|
|
750
|
+
// sentence on {@link PdgAscentCoverage}: absent ⇒ nothing was scanned because
|
|
751
|
+
// the descent never ran (an upstream slice), present ⇒ it ran and these are
|
|
752
|
+
// its counts.
|
|
753
|
+
//
|
|
754
|
+
// A zeroed-but-PRESENT record is therefore a real, honest reading — "the
|
|
755
|
+
// descent ran and the slice's blocks recorded no callee ids to scan" — not a
|
|
756
|
+
// placeholder. What used to make that reading unsafe was a block carrying
|
|
757
|
+
// call sites the index left id-less, which vanished from the population with
|
|
758
|
+
// no signal; that case now raises `'callee-ids-unrecorded'`, so a zero here
|
|
759
|
+
// with `examinedComplete: true` really does mean there was nothing to scan.
|
|
760
|
+
...(input.ascentCoverage
|
|
761
|
+
? {
|
|
762
|
+
ascent: publishedAscentCoverage({
|
|
763
|
+
coverage: input.ascentCoverage,
|
|
764
|
+
incompleteReasons: ascentIncompleteReasons,
|
|
765
|
+
callSummaryAvailable: input.callSummaryAvailable === true,
|
|
766
|
+
}),
|
|
767
|
+
}
|
|
768
|
+
: {}),
|
|
628
769
|
},
|
|
629
770
|
// Statement-level slice: the dependent source statements (line + text) the
|
|
630
771
|
// change reaches. This is the primary useful output of statement mode; the
|
|
@@ -936,66 +1077,88 @@ async function bfsReachableBlocks(input) {
|
|
|
936
1077
|
return { reachable, depthReached, truncatedByDepth, truncatedByLimit };
|
|
937
1078
|
}
|
|
938
1079
|
/**
|
|
939
|
-
* Gather the resolved callee
|
|
940
|
-
*
|
|
941
|
-
*
|
|
942
|
-
*
|
|
943
|
-
*
|
|
944
|
-
|
|
945
|
-
|
|
946
|
-
|
|
947
|
-
|
|
948
|
-
|
|
949
|
-
|
|
950
|
-
|
|
951
|
-
|
|
952
|
-
|
|
953
|
-
|
|
954
|
-
*
|
|
955
|
-
* `
|
|
956
|
-
*
|
|
957
|
-
*
|
|
958
|
-
*
|
|
959
|
-
*
|
|
960
|
-
*
|
|
961
|
-
*
|
|
1080
|
+
* Gather the resolved callee ids (`BasicBlock.calleeIds`) invoked across a set of
|
|
1081
|
+
* slice blocks, keeping the CALL block → callees association rather than
|
|
1082
|
+
* flattening it: the return-value ascent (U-C4) re-seeds the caller's intra
|
|
1083
|
+
* closure FROM the specific call block whose callee's `CALL_SUMMARY` licenses the
|
|
1084
|
+
* ascent, so a flat id set is insufficient. Reuses the SHARED
|
|
1085
|
+
* {@link parseCalleeIdsCell} so the split/drop-sentinel logic cannot diverge from
|
|
1086
|
+
* `LocalBackend.calleeIdsOfBlocks`. A block with no callee ids (empty/whitespace
|
|
1087
|
+
* cell, or a pre-namespace-v4 index with no `calleeIds` column) yields an empty
|
|
1088
|
+
* `calleeIds` — skipped by the consumer, so such an index degrades cleanly to
|
|
1089
|
+
* intra-only (no inter-procedural hop).
|
|
1090
|
+
*
|
|
1091
|
+
* `calleeListTruncated` reports whether ANY of the queried blocks carried the
|
|
1092
|
+
* emit-time cap sentinel. It is read from the RAW cell, so a block whose entire
|
|
1093
|
+
* list was capped away (sentinel only ⇒ no ids ⇒ not emitted as a `BlockCallees`
|
|
1094
|
+
* row) still raises it.
|
|
1095
|
+
*
|
|
1096
|
+
* `idlessCallSites` is the OTHER way a block's call sites leave the population
|
|
1097
|
+
* unannounced: `calleeIdsOfBlock` emits an EMPTY `calleeIds` cell for a whole file
|
|
1098
|
+
* whose resolved-id map is absent, and an empty cell carries no sentinel, so the
|
|
1099
|
+
* cap flag cannot see it. The sibling `callees` (leaf NAMES) cell is read purely
|
|
1100
|
+
* to tell that case apart from a block that genuinely calls nothing — names
|
|
1101
|
+
* present + ids absent means the index recorded call sites it could not resolve.
|
|
1102
|
+
* The name cell is never used for the descent itself (the resolved id is the sound
|
|
1103
|
+
* key); it only keeps the coverage claim honest.
|
|
962
1104
|
*/
|
|
963
1105
|
async function calleeIdsByBlock(lbugPath, blockIds, exec) {
|
|
964
1106
|
if (blockIds.length === 0)
|
|
965
|
-
return [];
|
|
966
|
-
const rows = await exec(lbugPath, `MATCH (b:BasicBlock) WHERE b.id IN $ids
|
|
1107
|
+
return { blocks: [], calleeListTruncated: false, idlessCallSites: false };
|
|
1108
|
+
const rows = await exec(lbugPath, `MATCH (b:BasicBlock) WHERE b.id IN $ids
|
|
1109
|
+
RETURN b.id AS id, b.calleeIds AS calleeIds, b.callees AS callees`, { ids: blockIds });
|
|
967
1110
|
const out = [];
|
|
1111
|
+
let calleeListTruncated = false;
|
|
1112
|
+
let idlessCallSites = false;
|
|
968
1113
|
// Narrow the awaited rows ONCE at the boundary to a typed record shape; read
|
|
969
1114
|
// the aliased cells via bracket access — no per-field `as any`.
|
|
970
1115
|
for (const r of rows) {
|
|
971
1116
|
const blockId = String(r['id'] ?? '');
|
|
972
1117
|
if (!blockId)
|
|
973
1118
|
continue;
|
|
974
|
-
|
|
1119
|
+
// ONE pass over the cell classifies BOTH facts — a second full split just to
|
|
1120
|
+
// re-test the sentinel doubled the per-row parse cost.
|
|
1121
|
+
const { ids: calleeIds, truncated } = parseCalleeIdsCell(r['calleeIds']);
|
|
1122
|
+
if (truncated)
|
|
1123
|
+
calleeListTruncated = true;
|
|
1124
|
+
// Ids absent while NAMES are present ⇒ recorded call sites with no resolved
|
|
1125
|
+
// id. Gated on `!truncated` so a capped-to-nothing cell keeps reporting the
|
|
1126
|
+
// cap (the more specific mechanism) rather than both.
|
|
1127
|
+
// `!idlessCallSites` first: the flag is sticky, so once it is set the string
|
|
1128
|
+
// allocation below is pure waste on every remaining row of every later hop.
|
|
1129
|
+
if (!idlessCallSites &&
|
|
1130
|
+
calleeIds.length === 0 &&
|
|
1131
|
+
!truncated &&
|
|
1132
|
+
String(r['callees'] ?? '').trim().length > 0) {
|
|
1133
|
+
idlessCallSites = true;
|
|
1134
|
+
}
|
|
975
1135
|
if (calleeIds.length > 0)
|
|
976
1136
|
out.push({ blockId, calleeIds });
|
|
977
1137
|
}
|
|
978
|
-
return out;
|
|
1138
|
+
return { blocks: out, calleeListTruncated, idlessCallSites };
|
|
979
1139
|
}
|
|
980
1140
|
/**
|
|
981
|
-
*
|
|
982
|
-
*
|
|
983
|
-
* (
|
|
984
|
-
* side of the producer's per-callee summary (see `call-summary-codec.ts`).
|
|
1141
|
+
* Scan the persisted `CALL_SUMMARY` self-loops of a set of resolved callee
|
|
1142
|
+
* symbol ids. This is the FU-C consumer side of the producer's per-callee
|
|
1143
|
+
* summary (see `call-summary-codec.ts`).
|
|
985
1144
|
*
|
|
986
1145
|
* The summary is a self-loop on the Function/Method/Constructor node:
|
|
987
1146
|
* `(c)-[r:CodeRelation {type:'CALL_SUMMARY'}]->(c) WHERE c.id IN $ids`. The
|
|
988
1147
|
* `reason` carries the param→return bitset; `decodeCallSummary` unpacks it and
|
|
989
|
-
* NEVER throws
|
|
990
|
-
*
|
|
991
|
-
*
|
|
992
|
-
* ascent
|
|
993
|
-
*
|
|
1148
|
+
* NEVER throws. Three outcomes, per {@link CalleeReturnFlowScan}: a non-empty
|
|
1149
|
+
* decoded return-flow, a cleanly-decoded EMPTY (`r:0`) summary, and an
|
|
1150
|
+
* UNDECODABLE reason. Only the first licenses an ascent — the other two yield no
|
|
1151
|
+
* ascent (the sound default: never claim a false return-flow) but are reported
|
|
1152
|
+
* separately so the note never states a fact about summaries it could not read.
|
|
1153
|
+
* A PRE-FU-C (v3) `--pdg` index has NO `CALL_SUMMARY` edges, so both sets come
|
|
1154
|
+
* back empty and the ascent is a clean no-op (the intra slice is unchanged — the
|
|
1155
|
+
* documented "re-index for CALL_SUMMARY" degradation).
|
|
994
1156
|
*/
|
|
995
1157
|
async function calleesWithReturnFlow(lbugPath, calleeIds, exec) {
|
|
996
|
-
const
|
|
1158
|
+
const returnFlowing = new Set();
|
|
1159
|
+
const undecodable = new Set();
|
|
997
1160
|
if (calleeIds.length === 0)
|
|
998
|
-
return
|
|
1161
|
+
return { returnFlowing, undecodable };
|
|
999
1162
|
const rows = await exec(lbugPath, `MATCH (c)-[r:CodeRelation]->(c)
|
|
1000
1163
|
WHERE r.type = 'CALL_SUMMARY' AND c.id IN $ids
|
|
1001
1164
|
RETURN c.id AS id, r.reason AS reason`, { ids: calleeIds });
|
|
@@ -1004,6 +1167,13 @@ async function calleesWithReturnFlow(lbugPath, calleeIds, exec) {
|
|
|
1004
1167
|
if (!id)
|
|
1005
1168
|
continue;
|
|
1006
1169
|
const decoded = decodeCallSummary(r['reason']);
|
|
1170
|
+
// A typed decode failure is NOT an empty summary — record it separately and
|
|
1171
|
+
// withhold the ascent all the same (the codec's contract: a decode failure
|
|
1172
|
+
// means "no usable ascent fact"). Only the note's wording depends on this.
|
|
1173
|
+
if (!decoded.ok) {
|
|
1174
|
+
undecodable.add(id);
|
|
1175
|
+
continue;
|
|
1176
|
+
}
|
|
1007
1177
|
// ARG→FORMAL trace precision: the conservative-but-sound default — ascend if
|
|
1008
1178
|
// ANY formal is return-flowing (the call site's argument is, by construction
|
|
1009
1179
|
// of the descent, in the slice: the call block is itself a slice block). A
|
|
@@ -1012,10 +1182,10 @@ async function calleesWithReturnFlow(lbugPath, calleeIds, exec) {
|
|
|
1012
1182
|
// per-arg list), so this never drops a real ascent; it may over-include
|
|
1013
1183
|
// (bounded — the result still flows to a slice statement). See the descent
|
|
1014
1184
|
// doc-comment + the result `note` caveat.
|
|
1015
|
-
if (decoded.
|
|
1016
|
-
|
|
1185
|
+
if (decoded.returnFlowParams.length > 0)
|
|
1186
|
+
returnFlowing.add(id);
|
|
1017
1187
|
}
|
|
1018
|
-
return
|
|
1188
|
+
return { returnFlowing, undecodable };
|
|
1019
1189
|
}
|
|
1020
1190
|
/**
|
|
1021
1191
|
* Batch-resolve resolved callee symbol ids → their `{id,filePath,startLine,endLine}`
|
|
@@ -1095,13 +1265,35 @@ async function interproceduralDescent(input) {
|
|
|
1095
1265
|
// U-C4 return-value ascent: CALL blocks whose callee has a non-empty
|
|
1096
1266
|
// CALL_SUMMARY return-flow → the call's result depends on the slice.
|
|
1097
1267
|
const ascentBlocks = new Set();
|
|
1268
|
+
// Ascent-coverage accumulators (rationale: {@link AscentCoverage}). Sets so a
|
|
1269
|
+
// callee invoked from two hops is tallied once; the `Seen` suffix marks them as
|
|
1270
|
+
// accumulators whose `.size` — not the set — is what gets returned.
|
|
1271
|
+
const calleeReferencesSeen = new Set();
|
|
1272
|
+
const calleesUndecodableSeen = new Set();
|
|
1273
|
+
// Sticky across hops. `anyReturnFlow` is the cross-hop union being non-empty,
|
|
1274
|
+
// which holds iff SOME hop's return-flowing set was — so the flag is set inside
|
|
1275
|
+
// the hop's existing non-empty branch rather than accumulating another Set.
|
|
1276
|
+
let anyReturnFlow = false;
|
|
1277
|
+
let calleeListTruncated = false;
|
|
1278
|
+
let idlessCallSites = false;
|
|
1098
1279
|
hopLoop: for (let hop = 0; hop < depthBudget; hop++) {
|
|
1099
1280
|
if (sliceBlocks.length === 0)
|
|
1100
1281
|
break;
|
|
1282
|
+
// Blocks this hop newly reached — the NEXT hop's slice, and therefore the set
|
|
1283
|
+
// whose `calleeIds` cells the next hop gathers. Declared BEFORE the U-C4
|
|
1284
|
+
// ascent below so the ascent's own newly-reached blocks land in it: they are
|
|
1285
|
+
// slice blocks (they are unioned into `reachable` and published in
|
|
1286
|
+
// `reachableBlocks`), so their call sites must reach the CALL_SUMMARY scan and
|
|
1287
|
+
// the coverage counters exactly like a descent-reached block's.
|
|
1288
|
+
const hopReached = new Set();
|
|
1101
1289
|
// Keep the CALL block → callee association (U-C4 needs it to re-seed the
|
|
1102
1290
|
// caller's intra closure FROM the specific call block the ascent licenses);
|
|
1103
1291
|
// the flattened id set still drives the descent's fresh-callee bookkeeping.
|
|
1104
|
-
const blockCallees = await calleeIdsByBlock(lbugPath, sliceBlocks, exec);
|
|
1292
|
+
const { blocks: blockCallees, calleeListTruncated: hopCellCapped, idlessCallSites: hopIdless, } = await calleeIdsByBlock(lbugPath, sliceBlocks, exec);
|
|
1293
|
+
if (hopCellCapped)
|
|
1294
|
+
calleeListTruncated = true;
|
|
1295
|
+
if (hopIdless)
|
|
1296
|
+
idlessCallSites = true;
|
|
1105
1297
|
const calleeIds = new Set();
|
|
1106
1298
|
for (const { calleeIds: ids } of blockCallees)
|
|
1107
1299
|
for (const id of ids)
|
|
@@ -1114,8 +1306,17 @@ async function interproceduralDescent(input) {
|
|
|
1114
1306
|
// that consumes the result is captured. Monotone: only ADDS to `reachable`,
|
|
1115
1307
|
// reusing the shared `visited` set, so it stays bounded + terminating. A
|
|
1116
1308
|
// pre-v4 index (no CALL_SUMMARY) yields no return-flowing callees → no-op.
|
|
1117
|
-
const
|
|
1309
|
+
const summaryScan = await calleesWithReturnFlow(lbugPath, [...calleeIds], exec);
|
|
1310
|
+
const returnFlowing = summaryScan.returnFlowing;
|
|
1311
|
+
for (const id of calleeIds)
|
|
1312
|
+
calleeReferencesSeen.add(id);
|
|
1313
|
+
// An undecodable summary withholds the ascent exactly like an empty one; it
|
|
1314
|
+
// is tracked only so the note reports "could not read" rather than "records
|
|
1315
|
+
// no return-flow".
|
|
1316
|
+
for (const id of summaryScan.undecodable)
|
|
1317
|
+
calleesUndecodableSeen.add(id);
|
|
1118
1318
|
if (returnFlowing.size > 0) {
|
|
1319
|
+
anyReturnFlow = true;
|
|
1119
1320
|
for (const { blockId, calleeIds: ids } of blockCallees) {
|
|
1120
1321
|
// Bound the ascent re-seeds the same way the descent bounds its per-span
|
|
1121
1322
|
// BFS (line ~1496): a wide fan-out of return-flowing call blocks must not
|
|
@@ -1145,10 +1346,26 @@ async function interproceduralDescent(input) {
|
|
|
1145
1346
|
stepLimit,
|
|
1146
1347
|
probeLimit,
|
|
1147
1348
|
});
|
|
1349
|
+
// BOTH budgets, not just the row budget: the re-seed runs the SAME BFS
|
|
1350
|
+
// under the SAME depth clamp as the top-level intra pass, whose depth
|
|
1351
|
+
// exhaustion is result-level truncation.
|
|
1352
|
+
//
|
|
1353
|
+
// The depth fold here is a CONSISTENCY guard with no independent
|
|
1354
|
+
// observable, and deliberately so — do not go hunting for the test that
|
|
1355
|
+
// pins it. The re-seed shares the caller's `visited` set, so it can only
|
|
1356
|
+
// discover new ground past the depth budget when the traversal that
|
|
1357
|
+
// already covered this closure (the top-level intra BFS, or the callee's
|
|
1358
|
+
// own BFS at a later hop) was ITSELF cut short — which has already raised
|
|
1359
|
+
// one of these flags. Keeping it is what stops that reasoning from
|
|
1360
|
+
// silently becoming load-bearing if the sharing of `visited` ever changes.
|
|
1148
1361
|
if (ascent.truncatedByLimit)
|
|
1149
1362
|
truncatedByLimit = true;
|
|
1150
|
-
|
|
1363
|
+
if (ascent.truncatedByDepth)
|
|
1364
|
+
truncatedByDepth = true;
|
|
1365
|
+
for (const id of ascent.reachable) {
|
|
1151
1366
|
reachable.add(id);
|
|
1367
|
+
hopReached.add(id);
|
|
1368
|
+
}
|
|
1152
1369
|
}
|
|
1153
1370
|
}
|
|
1154
1371
|
const freshIds = [...calleeIds].filter((id) => !seededCalleeIds.has(id));
|
|
@@ -1202,7 +1419,6 @@ async function interproceduralDescent(input) {
|
|
|
1202
1419
|
stepLimit,
|
|
1203
1420
|
probeLimit,
|
|
1204
1421
|
})));
|
|
1205
|
-
const hopReached = new Set();
|
|
1206
1422
|
for (let si = 0; si < spans.length; si++) {
|
|
1207
1423
|
// Node budget is checked INSIDE the per-span MERGE (in span order) so the
|
|
1208
1424
|
// mid-hop short-circuit stays byte-identical: the cumulative reachable size
|
|
@@ -1232,6 +1448,14 @@ async function interproceduralDescent(input) {
|
|
|
1232
1448
|
continue;
|
|
1233
1449
|
if (bfs.truncatedByLimit)
|
|
1234
1450
|
truncatedByLimit = true;
|
|
1451
|
+
// A callee whose own dependence chain outruns `intraDepthBudget` is the SAME
|
|
1452
|
+
// kind of incompleteness the top-level intra BFS reports through this flag
|
|
1453
|
+
// (the budget is deliberately the same clamp — see `intraDepthBudget`), so
|
|
1454
|
+
// it folds into the same result-level signal. Without this the slice could
|
|
1455
|
+
// stop mid-callee while `truncated` stayed false and `examinedComplete`
|
|
1456
|
+
// published a false all-clear over the callees past the frontier.
|
|
1457
|
+
if (bfs.truncatedByDepth)
|
|
1458
|
+
truncatedByDepth = true;
|
|
1235
1459
|
// The per-callee BFS ran against a clone, so fold its discovered blocks
|
|
1236
1460
|
// into the shared `visited`/`reachable` here (the sequential path did this
|
|
1237
1461
|
// inside the BFS); Sets dedup, so order across siblings is irrelevant.
|
|
@@ -1247,9 +1471,12 @@ async function interproceduralDescent(input) {
|
|
|
1247
1471
|
}
|
|
1248
1472
|
sliceBlocks = [...hopReached];
|
|
1249
1473
|
}
|
|
1250
|
-
// Frontier of callees still expandable after the hop budget ⇒ depth
|
|
1251
|
-
// (Conservative: if the last hop reached blocks AND we used the full
|
|
1252
|
-
// deeper callees may exist.)
|
|
1474
|
+
// Frontier of callees still expandable after the FUNCTION-hop budget ⇒ depth
|
|
1475
|
+
// truncation. (Conservative: if the last hop reached blocks AND we used the full
|
|
1476
|
+
// budget, deeper callees may exist.) This is the hop-level source; the per-callee
|
|
1477
|
+
// and ascent BFS passes above fold their own block-hop depth exhaustion into the
|
|
1478
|
+
// same flag, so `truncatedByDepth` means "some dependence frontier was cut by a
|
|
1479
|
+
// depth budget", at either granularity.
|
|
1253
1480
|
if (hopsReached >= depthBudget && sliceBlocks.length > 0)
|
|
1254
1481
|
truncatedByDepth = true;
|
|
1255
1482
|
return {
|
|
@@ -1259,6 +1486,13 @@ async function interproceduralDescent(input) {
|
|
|
1259
1486
|
truncatedByLimit,
|
|
1260
1487
|
truncatedByNodeCap,
|
|
1261
1488
|
ascentBlocks,
|
|
1489
|
+
ascentCoverage: {
|
|
1490
|
+
references: calleeReferencesSeen.size,
|
|
1491
|
+
anyReturnFlow,
|
|
1492
|
+
undecodable: calleesUndecodableSeen.size,
|
|
1493
|
+
listTruncated: calleeListTruncated,
|
|
1494
|
+
idlessCallSites,
|
|
1495
|
+
},
|
|
1262
1496
|
};
|
|
1263
1497
|
}
|
|
1264
1498
|
export async function runImpactPDG(deps) {
|
|
@@ -1424,6 +1658,13 @@ export async function runImpactPDG(deps) {
|
|
|
1424
1658
|
// call lines (a coalesced call block spans several statements whose results
|
|
1425
1659
|
// chain through it — the statement-granularity realisation of the ascent).
|
|
1426
1660
|
let ascentBlocks = new Set();
|
|
1661
|
+
// Observed ascent inputs, plumbed to the note AND to `pdgEvidence.ascent`
|
|
1662
|
+
// (rationale: AscentCoverage). Left UNDEFINED when the descent never ran (an
|
|
1663
|
+
// upstream slice): "nothing was scanned" is a different fact from "we scanned
|
|
1664
|
+
// and found nothing", and a zeroed record would publish the second. The note's
|
|
1665
|
+
// ascent branch is gated on `interproceduralHops > 0`, which only a descent can
|
|
1666
|
+
// produce, so the prose is unaffected either way.
|
|
1667
|
+
let ascentCoverage;
|
|
1427
1668
|
if (direction === 'downstream') {
|
|
1428
1669
|
const interproc = await interproceduralDescent({
|
|
1429
1670
|
lbugPath: repo.lbugPath,
|
|
@@ -1450,6 +1691,7 @@ export async function runImpactPDG(deps) {
|
|
|
1450
1691
|
});
|
|
1451
1692
|
interproceduralHops = interproc.hopsReached;
|
|
1452
1693
|
ascentBlocks = interproc.ascentBlocks;
|
|
1694
|
+
ascentCoverage = interproc.ascentCoverage;
|
|
1453
1695
|
for (const id of interproc.reachable)
|
|
1454
1696
|
reachable.add(id);
|
|
1455
1697
|
if (interproc.truncatedByDepth)
|
|
@@ -1582,6 +1824,27 @@ export async function runImpactPDG(deps) {
|
|
|
1582
1824
|
...(truncated ? { truncated: true } : {}),
|
|
1583
1825
|
...(truncatedBy ? { truncatedBy } : {}),
|
|
1584
1826
|
...(truncatedByReasons ? { truncatedByReasons } : {}),
|
|
1827
|
+
// The descent may well have RUN and scanned callees before the slice turned
|
|
1828
|
+
// out to reach no DISTINCT downstream block (a seed line whose only callee
|
|
1829
|
+
// is invoked directly on it). `pdgEvidence.ascent` is contracted as "present
|
|
1830
|
+
// iff the inter-procedural descent ran", so it must be published here too —
|
|
1831
|
+
// omitting it made the documented "absent ⇒ nothing was scanned" reading
|
|
1832
|
+
// false on exactly this exit. Classified through the SAME shared helpers the
|
|
1833
|
+
// assembled slice result uses, so the two exits cannot disagree.
|
|
1834
|
+
...(ascentCoverage
|
|
1835
|
+
? {
|
|
1836
|
+
pdgEvidence: {
|
|
1837
|
+
ascent: publishedAscentCoverage({
|
|
1838
|
+
coverage: ascentCoverage,
|
|
1839
|
+
incompleteReasons: ascentIncompleteReasonsOf({
|
|
1840
|
+
truncated,
|
|
1841
|
+
coverage: ascentCoverage,
|
|
1842
|
+
}),
|
|
1843
|
+
callSummaryAvailable,
|
|
1844
|
+
}),
|
|
1845
|
+
},
|
|
1846
|
+
}
|
|
1847
|
+
: {}),
|
|
1585
1848
|
...emptyPdgParityFields(),
|
|
1586
1849
|
};
|
|
1587
1850
|
}
|
|
@@ -1611,6 +1874,7 @@ export async function runImpactPDG(deps) {
|
|
|
1611
1874
|
truncatedByReasons,
|
|
1612
1875
|
interproceduralHops,
|
|
1613
1876
|
callSummaryAvailable,
|
|
1877
|
+
ascentCoverage,
|
|
1614
1878
|
});
|
|
1615
1879
|
}
|
|
1616
1880
|
export function pdgBridgeEvidenceForImpact(input) {
|
package/dist/mcp/tools.js
CHANGED
|
@@ -405,7 +405,7 @@ MODE (opt-in): "callgraph" (default) walks symbol→symbol edges (CALLS/IMPORTS/
|
|
|
405
405
|
|
|
406
406
|
STATEMENT-ANCHORED PDG SLICE: with mode:'pdg', pass "line" (1-based source line within the target symbol) to seed the dependence slice on the statement at that line and return what depends on it in affectedStatements (line + text). Inter-procedural symbols are still reported through interproceduralByDepth/pdgInterprocedural and the compatibility byDepth bucket. Without "line", pdg returns whole-symbol inter-procedural reach plus local whole-symbol PDG diagnostics.
|
|
407
407
|
|
|
408
|
-
PDG OUTPUT CONTRACT: every mode:'pdg' result (success, empty, degraded, or error) carries pdgResultVersion:2 — a stable discriminator for external consumers that bumps on any breaking change to the PDG result shape (distinct from the DB schema version). Successful PDG results include mode:'pdg', a full target envelope (id/name/type/filePath), affectedStatements, affectedStatementCount, interproceduralByDepth/pdgInterprocedural for cross-function reach, compatibility byDepth/byDepthCounts, risk:'UNKNOWN', and a note describing the unified contract. Degraded PDG results (no-layer, sub-layer-missing, unknown) keep mode:'pdg', pdgResultVersion:2, target metadata when the target resolves, risk:'UNKNOWN', note/remediation, and empty byDepth parity fields — never a false-safe zero. If depth and limit both bound the slice, truncatedByReasons reports both causes while truncatedBy remains scalar.
|
|
408
|
+
PDG OUTPUT CONTRACT: every mode:'pdg' result (success, empty, degraded, or error) carries pdgResultVersion:2 — a stable discriminator for external consumers that bumps on any breaking change to the PDG result shape (distinct from the DB schema version). Successful PDG results include mode:'pdg', a full target envelope (id/name/type/filePath), affectedStatements, affectedStatementCount, interproceduralByDepth/pdgInterprocedural for cross-function reach, compatibility byDepth/byDepthCounts, risk:'UNKNOWN', and a note describing the unified contract. Degraded PDG results (no-layer, sub-layer-missing, unknown) keep mode:'pdg', pdgResultVersion:2, target metadata when the target resolves, risk:'UNKNOWN', note/remediation, and empty byDepth parity fields — never a false-safe zero. If depth and limit both bound the slice, truncatedByReasons reports both causes while truncatedBy remains scalar. Return-value-ascent coverage is published structurally at pdgEvidence.ascent — present iff the inter-procedural descent ran, including on an empty slice — with referencesScanned (DISTINCT callees scanned for a CALL_SUMMARY: a distinct-id tally, not a call-site count — two call sites to the same callee count once), returnFlowFound (whether the ascent fired anywhere in the slice), undecodableSummaryCount, examinedComplete (whether that scan covered every callee the index recorded a resolved id for on the visited blocks), incompleteReasons ('traversal-truncated' | 'callee-list-capped' | 'callee-ids-unrecorded'), and callSummaryLayerPresent. Read callSummaryLayerPresent FIRST: false ⇒ a pre-CALL_SUMMARY index, so {referencesScanned:N>0, returnFlowFound:false} is self-consistent and says nothing about the callees — the scan ran, but no layer existed in which a return-flow could be recorded (remedy: re-run gitnexus analyze --pdg). Branch on those fields; the note narrates the same facts in prose for humans and is not a stable contract.
|
|
409
409
|
|
|
410
410
|
WHEN TO USE: Before making code changes — especially refactoring, renaming, or modifying shared code. Shows what would break.
|
|
411
411
|
AFTER THIS: Review d=1 items (WILL BREAK). Use context() on high-risk symbols.
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "gitnexus",
|
|
3
|
-
"version": "1.6.10-rc.
|
|
3
|
+
"version": "1.6.10-rc.154",
|
|
4
4
|
"description": "Graph-powered code intelligence for AI agents. Index any codebase, query via MCP or CLI.",
|
|
5
5
|
"author": "Abhigyan Patwari",
|
|
6
6
|
"license": "PolyForm-Noncommercial-1.0.0",
|
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Weight-aware partitioning for the cross-platform test matrix.
|
|
3
|
+
*
|
|
4
|
+
* WHY THIS EXISTS. `run-cross-platform.ts` used to hand vitest the whole file
|
|
5
|
+
* list plus `--shard=i/n`, and vitest partitions by file COUNT. Runtime on this
|
|
6
|
+
* suite is wildly uneven — measured on the Windows runner, `cli-e2e` is 361 s
|
|
7
|
+
* and `worker-pool` 221 s, while most files are under a second — so a
|
|
8
|
+
* count-split routinely put several of the heaviest suites on one shard. That
|
|
9
|
+
* is #2449, and this file's sibling header has documented the symptom ("the
|
|
10
|
+
* heaviest spawn suites can cluster on one shard") since the watchdog was first
|
|
11
|
+
* raised from 15 to 20 minutes.
|
|
12
|
+
*
|
|
13
|
+
* It went from a latent hazard to a red matrix when three CHEAP files (the
|
|
14
|
+
* `dist/` module-load closure guards: 448 ms, 53 ms, sub-second) were added to
|
|
15
|
+
* `SPAWN_CLI`. They cost nothing to run, but a count-split re-partitions on
|
|
16
|
+
* every insertion, and the reshuffle happened to land `cli-e2e` + `cli-limit-e2e`
|
|
17
|
+
* + `analyze-heap-oom-e2e` together on shard 1/3 — 32 files against 26 and 29 —
|
|
18
|
+
* which blew the 20-minute budget with four files still queued. Nothing about
|
|
19
|
+
* the added files caused it; they were simply the perturbation.
|
|
20
|
+
*
|
|
21
|
+
* So the split is done HERE, by weight, and only the chosen shard's files are
|
|
22
|
+
* handed to vitest. Two properties follow, and both are pinned in
|
|
23
|
+
* `test/unit/cross-platform-shard.test.ts`:
|
|
24
|
+
*
|
|
25
|
+
* - the heaviest suites are spread across shards by construction, so the
|
|
26
|
+
* busiest shard tracks the ideal rather than the luck of the sort order;
|
|
27
|
+
* - adding or removing a CHEAP file cannot move a heavy one, so registering a
|
|
28
|
+
* new platform-sensitive test is no longer a CI-stability gamble. That is the
|
|
29
|
+
* property whose absence caused this.
|
|
30
|
+
*/
|
|
31
|
+
|
|
32
|
+
/**
|
|
33
|
+
* Measured wall-clock on the WINDOWS runner (the slowest platform, so it is the
|
|
34
|
+
* one that decides the budget), in seconds, from the last fully-green matrix run
|
|
35
|
+
* plus the timed files of the run that failed.
|
|
36
|
+
*
|
|
37
|
+
* Only files heavy enough to matter are listed; everything else is carried by
|
|
38
|
+
* {@link PER_FILE_OVERHEAD_SEC} alone. These are load-balancing hints, NOT
|
|
39
|
+
* assertions — no
|
|
40
|
+
* test asserts a runtime, and drift only makes the split slightly less even, so
|
|
41
|
+
* a stale entry is harmless and refreshing them is optional. Deliberately not
|
|
42
|
+
* auto-generated: a committed table is reviewable and works offline, and the
|
|
43
|
+
* alternative (timing files at CI runtime to decide the split) would make the
|
|
44
|
+
* partition depend on the very machine load it is trying to protect against.
|
|
45
|
+
*/
|
|
46
|
+
export const WINDOWS_WEIGHTS_SEC: Readonly<Record<string, number>> = {
|
|
47
|
+
'test/integration/cli-e2e.test.ts': 361,
|
|
48
|
+
'test/integration/worker-pool.test.ts': 222,
|
|
49
|
+
'test/unit/incremental-vector-extension-ordering.test.ts': 87,
|
|
50
|
+
'test/integration/cli-limit-e2e.test.ts': 75,
|
|
51
|
+
'test/unit/hooks.test.ts': 26,
|
|
52
|
+
'test/integration/analyze-heap-oom-e2e.test.ts': 23,
|
|
53
|
+
'test/unit/git-utils.test.ts': 18,
|
|
54
|
+
'test/integration/hooks-e2e.test.ts': 15,
|
|
55
|
+
'test/integration/tree-sitter-languages.test.ts': 9,
|
|
56
|
+
'test/unit/repo-manager.test.ts': 9,
|
|
57
|
+
'test/unit/detect-changes-worktree.test.ts': 9,
|
|
58
|
+
'test/integration/antigravity-hook-e2e.test.ts': 7,
|
|
59
|
+
'test/unit/index-lock.test.ts': 5,
|
|
60
|
+
'test/unit/setup.test.ts': 5,
|
|
61
|
+
};
|
|
62
|
+
|
|
63
|
+
/**
|
|
64
|
+
* Fixed cost every file pays regardless of what it asserts: a pool worker start,
|
|
65
|
+
* module graph evaluation, and (for most of this list) a native addon load.
|
|
66
|
+
*
|
|
67
|
+
* Added to EVERY file's weight, not just unmeasured ones, and that is the point.
|
|
68
|
+
* Calibrated against the last green Windows matrix: its busiest shard ran 736 s
|
|
69
|
+
* of wall clock over ~511 s of measured file time, so roughly 8 s per file is
|
|
70
|
+
* unattributed setup. Without this term the balancer treats a light file as
|
|
71
|
+
* nearly free and, having isolated the two monsters, piles every remaining file
|
|
72
|
+
* onto the other shards — trading a runtime imbalance for a file-count one that
|
|
73
|
+
* costs just as much. With it, the split balances runtime AND count together.
|
|
74
|
+
*/
|
|
75
|
+
const PER_FILE_OVERHEAD_SEC = 8;
|
|
76
|
+
|
|
77
|
+
/**
|
|
78
|
+
* Scheduling weight for `file`: its measured runtime (0 if it was fast enough
|
|
79
|
+
* that vitest printed no duration) plus the per-file floor above.
|
|
80
|
+
*/
|
|
81
|
+
export function weightOf(file: string): number {
|
|
82
|
+
return (WINDOWS_WEIGHTS_SEC[file] ?? 0) + PER_FILE_OVERHEAD_SEC;
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
/**
|
|
86
|
+
* Partition `files` into `total` shards and return the 1-based `index` one.
|
|
87
|
+
*
|
|
88
|
+
* Longest-processing-time first: sort by weight descending, then repeatedly give
|
|
89
|
+
* the next file to the lightest shard so far. LPT is the standard greedy for
|
|
90
|
+
* multiprocessor scheduling and is guaranteed within 4/3 of optimal — far more
|
|
91
|
+
* than enough here, where the goal is only "no shard gets two monsters".
|
|
92
|
+
*
|
|
93
|
+
* Ties break on the file path so the partition is DETERMINISTIC: every shard
|
|
94
|
+
* computes the same split independently, on a different machine, with no
|
|
95
|
+
* coordination — which is what lets each runner select its own slice.
|
|
96
|
+
*
|
|
97
|
+
* Returns files in the input list's original order, not weight order, so failure
|
|
98
|
+
* output and reruns stay readable.
|
|
99
|
+
*/
|
|
100
|
+
export function shardFiles(
|
|
101
|
+
files: readonly string[],
|
|
102
|
+
index: number,
|
|
103
|
+
total: number,
|
|
104
|
+
): readonly string[] {
|
|
105
|
+
if (!Number.isInteger(total) || total < 1) {
|
|
106
|
+
throw new Error(`shard total must be a positive integer, got ${total}`);
|
|
107
|
+
}
|
|
108
|
+
if (!Number.isInteger(index) || index < 1 || index > total) {
|
|
109
|
+
throw new Error(`shard index must be in 1..${total}, got ${index}`);
|
|
110
|
+
}
|
|
111
|
+
if (total === 1) return [...files];
|
|
112
|
+
|
|
113
|
+
const byWeightDesc = [...files].sort((a, b) => {
|
|
114
|
+
const diff = weightOf(b) - weightOf(a);
|
|
115
|
+
return diff !== 0 ? diff : a.localeCompare(b);
|
|
116
|
+
});
|
|
117
|
+
|
|
118
|
+
const loads = Array.from({ length: total }, () => 0);
|
|
119
|
+
const assigned = Array.from({ length: total }, () => new Set<string>());
|
|
120
|
+
for (const file of byWeightDesc) {
|
|
121
|
+
let lightest = 0;
|
|
122
|
+
for (let i = 1; i < total; i++) {
|
|
123
|
+
if (loads[i]! < loads[lightest]!) lightest = i;
|
|
124
|
+
}
|
|
125
|
+
assigned[lightest]!.add(file);
|
|
126
|
+
loads[lightest]! += weightOf(file);
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
const mine = assigned[index - 1]!;
|
|
130
|
+
return files.filter((f) => mine.has(f));
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
/** Total weight of a file set — the shard cost this balancer is minimising. */
|
|
134
|
+
export function shardWeight(files: readonly string[]): number {
|
|
135
|
+
return files.reduce((sum, f) => sum + weightOf(f), 0);
|
|
136
|
+
}
|
|
@@ -186,6 +186,33 @@ const SPAWN_CLI = [
|
|
|
186
186
|
// exposed a file-backend double-admit race here (#2658 review); the reclaim is
|
|
187
187
|
// now judgment-verified so a live holder is never displaced.
|
|
188
188
|
'test/integration/analyze-index-lock-concurrency.test.ts',
|
|
189
|
+
// The three `dist/` module-load closure guards, all built on the shared
|
|
190
|
+
// child-process probe in `test/helpers/module-load-probe.ts`. That probe IS
|
|
191
|
+
// the platform-varying part: it spawns `process.execPath` in array form,
|
|
192
|
+
// clears NODE_OPTIONS, addresses its target via `pathToFileURL` (Windows needs
|
|
193
|
+
// the `file:///C:/...` form — a bare absolute path is not a valid ESM
|
|
194
|
+
// specifier there), and renders every result through a `path.sep`→POSIX
|
|
195
|
+
// normalisation the anchors and offender regexes depend on. None of that is
|
|
196
|
+
// proven anywhere else.
|
|
197
|
+
//
|
|
198
|
+
// Cheap: measured on the Windows runner at 448 ms, 53 ms and sub-second. An
|
|
199
|
+
// earlier attempt to register them still turned the matrix red — not from
|
|
200
|
+
// their own cost, but because vitest sharded by file COUNT, so inserting any
|
|
201
|
+
// file re-partitioned the list and happened to cluster `cli-e2e` (361 s) with
|
|
202
|
+
// `cli-limit-e2e` (75 s) on one shard. The split is weight-aware now
|
|
203
|
+
// (`scripts/cross-platform-shard.ts`), so a cheap file can no longer move a
|
|
204
|
+
// heavy one.
|
|
205
|
+
//
|
|
206
|
+
// #2802: MCP startup must not eagerly load the analyze-only language
|
|
207
|
+
// provider registry or the group contract extractors.
|
|
208
|
+
'test/integration/mcp/startup-language-closure.test.ts',
|
|
209
|
+
// PR #1383: `cli/mcp.js`'s static-import closure must stay leaf-only so no
|
|
210
|
+
// native binding initialises before the stdout sentinel installs.
|
|
211
|
+
'test/integration/mcp/import-closure.test.ts',
|
|
212
|
+
// #2091/#2093/#2116: the scope-resolution registry must not load the optional
|
|
213
|
+
// tree-sitter grammars at import time. The offender regexes match grammar
|
|
214
|
+
// paths with either separator, which only the Windows runner proves.
|
|
215
|
+
'test/integration/optional-grammars/registry-import-closure.test.ts',
|
|
189
216
|
];
|
|
190
217
|
|
|
191
218
|
// Worker threads tests — exercise real worker_threads which have
|
|
@@ -15,6 +15,7 @@ import path from 'path';
|
|
|
15
15
|
import { fileURLToPath } from 'url';
|
|
16
16
|
import { ALL_CROSS_PLATFORM } from './cross-platform-tests.js';
|
|
17
17
|
import { parseShardArg } from './shard-arg.js';
|
|
18
|
+
import { shardFiles, shardWeight } from './cross-platform-shard.js';
|
|
18
19
|
|
|
19
20
|
const __dirname = path.dirname(fileURLToPath(import.meta.url));
|
|
20
21
|
const ROOT = path.resolve(__dirname, '..');
|
|
@@ -29,8 +30,10 @@ if (missing.length > 0) {
|
|
|
29
30
|
}
|
|
30
31
|
|
|
31
32
|
// Optional sharding (CI): `--shard=<i>/<n>` splits the fixed file list across
|
|
32
|
-
// parallel matrix shards
|
|
33
|
-
//
|
|
33
|
+
// parallel matrix shards. The split is computed HERE, by measured weight, and
|
|
34
|
+
// only this shard's files are handed to vitest — it is NOT passed through,
|
|
35
|
+
// because vitest partitions by file COUNT and this suite's runtimes span three
|
|
36
|
+
// orders of magnitude (see cross-platform-shard.ts). The
|
|
34
37
|
// Windows runner is ~5x slower than macOS/Linux on this spawn-heavy suite (~50
|
|
35
38
|
// CLI/worker process spawns), so a single shard was creeping past the watchdog
|
|
36
39
|
// below; sharding keeps each runner well under it (see ci-tests.yml matrix).
|
|
@@ -63,14 +66,22 @@ const timeoutMs =
|
|
|
63
66
|
? timeoutMinutes * 60 * 1000
|
|
64
67
|
: DEFAULT_TIMEOUT_MIN * 60 * 1000;
|
|
65
68
|
|
|
69
|
+
// Resolve the shard to an explicit file list. `--shard=i/n` is consumed here,
|
|
70
|
+
// never forwarded: forwarding it as well would re-partition this slice a second
|
|
71
|
+
// time and silently drop most of it.
|
|
72
|
+
const shardParts = shardArg?.replace('--shard=', '').split('/');
|
|
73
|
+
const shardIndex = shardParts ? Number(shardParts[0]) : 1;
|
|
74
|
+
const shardTotal = shardParts ? Number(shardParts[1]) : 1;
|
|
75
|
+
const files = shardFiles(ALL_CROSS_PLATFORM, shardIndex, shardTotal);
|
|
76
|
+
|
|
66
77
|
console.log(
|
|
67
|
-
`Running ${ALL_CROSS_PLATFORM.length} platform-sensitive tests` +
|
|
68
|
-
`${shardArg ? ` (${
|
|
78
|
+
`Running ${files.length} of ${ALL_CROSS_PLATFORM.length} platform-sensitive tests` +
|
|
79
|
+
`${shardArg ? ` (shard ${shardIndex}/${shardTotal}, ~${shardWeight(files)}s measured weight)` : ''}...\n`,
|
|
69
80
|
);
|
|
70
81
|
|
|
71
82
|
const startedAt = Date.now();
|
|
72
83
|
try {
|
|
73
|
-
execFileSync('npx', ['vitest', 'run', ...
|
|
84
|
+
execFileSync('npx', ['vitest', 'run', ...files], {
|
|
74
85
|
cwd: ROOT,
|
|
75
86
|
stdio: 'inherit',
|
|
76
87
|
timeout: timeoutMs,
|