@titan-design/code-graph 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +105 -0
- package/dist/index.d.ts +532 -0
- package/dist/index.js +2413 -0
- package/dist/index.js.map +1 -0
- package/package.json +45 -0
package/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Henry Jewkes
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
package/README.md
ADDED
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
# @titan-design/code-graph
|
|
2
|
+
|
|
3
|
+
The dependency graph of a TypeScript or Python tree: files, modules, exported and internal
|
|
4
|
+
symbols, external packages, and the import / re-export / reference edges between them, with
|
|
5
|
+
index-time source metrics, in one SQLite file built from `@titan-design/store-sqlite` kit
|
|
6
|
+
tables and refreshed incrementally.
|
|
7
|
+
|
|
8
|
+
Tier 2 of the titan-platform DAG. Depends on `store-sqlite`, `ts-morph`, and the tree-sitter
|
|
9
|
+
WASM grammars. Extracted from codewatch's `@codewatch/graph` (TP-9), split along the seam the
|
|
10
|
+
audit identified: that package did the job of both a store and a code graph.
|
|
11
|
+
|
|
12
|
+
```ts
|
|
13
|
+
import { indexPaths, listEdges, listNodes, openCodeGraph } from "@titan-design/code-graph";
|
|
14
|
+
|
|
15
|
+
const store = openCodeGraph(".codewatch/graph.db");
|
|
16
|
+
const { snapshotId, files, nodes, edges, reused } = await indexPaths(store, {
|
|
17
|
+
paths: ["packages", "products"],
|
|
18
|
+
ref: "head",
|
|
19
|
+
});
|
|
20
|
+
listNodes(store, snapshotId); // file / module / external nodes, symbols on request
|
|
21
|
+
listEdges(store, snapshotId); // imports / re-exports, references on request
|
|
22
|
+
```
|
|
23
|
+
|
|
24
|
+
## What was extracted, and what was not
|
|
25
|
+
|
|
26
|
+
In: the parser (tree-sitter WASM for TypeScript, TSX and Python), the walk, the ts-morph
|
|
27
|
+
extractor and its symbol layer, role classification, generated-file detection, id aliasing
|
|
28
|
+
across git renames, the three-tier incremental reuse, and the metrics computed at index time
|
|
29
|
+
(degree, utilization, loc, cyclomatic, cognitive, nesting, class count, lcom4, per-symbol
|
|
30
|
+
complexity). `lcom.ts` came along despite being an analysis: `source-metrics.ts` calls it
|
|
31
|
+
directly and lcom4 is a pure function of a file's bytes, so it belongs with the metrics that
|
|
32
|
+
carry forward under reuse.
|
|
33
|
+
|
|
34
|
+
Deferred, all of it still in codewatch, all of it a follow-up on this package rather than a
|
|
35
|
+
change to it:
|
|
36
|
+
|
|
37
|
+
- The rules engine (`check*.ts`) that turns a snapshot into pass/fail against a config.
|
|
38
|
+
- Git-history mining: churn, ownership, change coupling, symbol coupling, test coverage
|
|
39
|
+
linking. `buildIndexerMetrics` used to fold these into the same pass; here it computes only
|
|
40
|
+
what a file's own bytes and the assembled graph determine.
|
|
41
|
+
- Graph analyses over a finished snapshot: communities, pagerank, partition quality,
|
|
42
|
+
relevance, conventions, coverage overlay, dead code, growth risk, patterns, prune,
|
|
43
|
+
test-linker, diff, reuse-delta reporting, embeddings.
|
|
44
|
+
|
|
45
|
+
Python support is new here rather than ported. codewatch walked TypeScript only; the parser
|
|
46
|
+
already had the grammar. The Python extractor is deliberately narrower than the ts-morph one:
|
|
47
|
+
no type checker, so imports resolve by dotted path against the tree and symbols come from the
|
|
48
|
+
same tree-sitter declaration walk that feeds complexity.
|
|
49
|
+
|
|
50
|
+
## The id scheme
|
|
51
|
+
|
|
52
|
+
Preserved exactly from codewatch, because this repo's own `dag:check` consumes it through
|
|
53
|
+
codewatch's CLI:
|
|
54
|
+
|
|
55
|
+
- A **file** id is its path relative to the git toplevel, in posix form:
|
|
56
|
+
`packages/registry/src/index.ts`. Ids root at the git toplevel even when you walk a
|
|
57
|
+
subtree, so importers across subtrees share one id space.
|
|
58
|
+
- A **module** id is the file id minus its extension: `packages/registry/src/index`. Its
|
|
59
|
+
parent is the directory above it.
|
|
60
|
+
- A **symbol** id hangs under its declaring file as `<fileId>#<name>`. `#` is legal in
|
|
61
|
+
neither a posix path nor a JS identifier, so the first one is the split.
|
|
62
|
+
- An **external** id is `npm:<package>` (scope-aware) or the `node:` builtin verbatim.
|
|
63
|
+
|
|
64
|
+
`NodeKind`, `EdgeKind`, and the `role` vocabulary are unchanged. So is the property the DAG
|
|
65
|
+
check rests on: an import of a workspace package by its published name resolves to that
|
|
66
|
+
package's source file, not to an `npm:` external, by remapping the `dist/*.d.ts` entry
|
|
67
|
+
ts-morph resolves back onto `src/`. That remap needs the target package built, which is why
|
|
68
|
+
`pnpm build` precedes both `pnpm test` and `dag:check`.
|
|
69
|
+
|
|
70
|
+
## The three reuse tiers
|
|
71
|
+
|
|
72
|
+
Every run writes a fingerprint per file: a content hash and a comment/whitespace-insensitive
|
|
73
|
+
hash of its parse structure. The next run diffs against the most recent snapshot carrying the
|
|
74
|
+
same `INDEX_VERSION` and sorts each file into one tier.
|
|
75
|
+
|
|
76
|
+
| Tier | Trigger | Work skipped |
|
|
77
|
+
|---|---|---|
|
|
78
|
+
| reuse | content hash matches | parse and extract both; nodes, edges and source metrics carried forward verbatim |
|
|
79
|
+
| cosmetic | content changed, structural hash matches | the ts-morph extract; edges come from the basis, symbol line spans are refreshed from the fresh parse |
|
|
80
|
+
| full | structural hash changed, or the file is new | nothing |
|
|
81
|
+
|
|
82
|
+
A file-membership delta (a file added or removed) forces the files whose imports it
|
|
83
|
+
re-resolves back to full extraction even when they are byte-identical. Degree metrics are
|
|
84
|
+
always recomputed over the whole assembled graph, so a heavily-reused run and a
|
|
85
|
+
`incremental: false` run produce the same snapshot; `indexer.test.ts` asserts that.
|
|
86
|
+
|
|
87
|
+
## Store layout
|
|
88
|
+
|
|
89
|
+
`openCodeGraph` opens the database with the kit's pragmas and runs three migrations.
|
|
90
|
+
|
|
91
|
+
Kit tables: `snapshot` (the snapshot registry), `node` (`entity_snap`, keyed
|
|
92
|
+
`(snapshot_id, id)`), `blob_cache` (content-addressed, for the embedding and summary caches
|
|
93
|
+
an analysis layer will want). Migration 2 adds four columns the kit shapes do not carry but
|
|
94
|
+
every consumer reads directly: `node.language`, `node.role`, `snapshot.commit_hash`,
|
|
95
|
+
`snapshot.index_version`.
|
|
96
|
+
|
|
97
|
+
Domain tables in `schema.ts`: `edge` (snapshot-scoped, keyed
|
|
98
|
+
`(snapshot_id, src_id, dst_id, kind)` — the kit's own edge table is bi-temporal, which is the
|
|
99
|
+
wrong time model for a population re-indexed all at once), `metric`, `id_alias`, and
|
|
100
|
+
`file_fingerprint`.
|
|
101
|
+
|
|
102
|
+
The symbol layer is hidden by default. `listNodes` drops `symbol` nodes and `listEdges` drops
|
|
103
|
+
`references` edges unless you ask for them, so a caller reasoning about module structure sees
|
|
104
|
+
the graph it expects and does not have one import of thirty names read as thirty
|
|
105
|
+
dependencies.
|
package/dist/index.d.ts
ADDED
|
@@ -0,0 +1,532 @@
|
|
|
1
|
+
import { Db, Migration } from '@titan-design/store-sqlite';
|
|
2
|
+
import { Tree } from 'web-tree-sitter';
|
|
3
|
+
import { Project } from 'ts-morph';
|
|
4
|
+
|
|
5
|
+
type NodeKind = "package" | "module" | "file" | "symbol" | "external";
|
|
6
|
+
type EdgeKind = "imports" | "re-exports" | "calls" | "extends" | "implements" | "references" | "depends-on";
|
|
7
|
+
type IdAliasReason = "rename" | "move" | "merge";
|
|
8
|
+
type NodeRole = "test" | "fixture" | "barrel" | "types" | "config" | "script" | "entry" | "generated" | "source";
|
|
9
|
+
interface GraphNode {
|
|
10
|
+
id: string;
|
|
11
|
+
kind: NodeKind;
|
|
12
|
+
name: string;
|
|
13
|
+
parentId?: string;
|
|
14
|
+
language?: string;
|
|
15
|
+
role?: NodeRole;
|
|
16
|
+
attrs?: Record<string, unknown>;
|
|
17
|
+
}
|
|
18
|
+
interface GraphEdge {
|
|
19
|
+
srcId: string;
|
|
20
|
+
dstId: string;
|
|
21
|
+
kind: EdgeKind;
|
|
22
|
+
attrs?: Record<string, unknown>;
|
|
23
|
+
}
|
|
24
|
+
interface GraphMetric {
|
|
25
|
+
nodeId: string;
|
|
26
|
+
name: string;
|
|
27
|
+
value: number | null;
|
|
28
|
+
unit?: string;
|
|
29
|
+
}
|
|
30
|
+
interface GraphFragment {
|
|
31
|
+
nodes: GraphNode[];
|
|
32
|
+
edges: GraphEdge[];
|
|
33
|
+
}
|
|
34
|
+
interface SnapshotRow {
|
|
35
|
+
id: number;
|
|
36
|
+
ref: string;
|
|
37
|
+
commitHash: string | null;
|
|
38
|
+
takenAt: string;
|
|
39
|
+
indexVersion: string;
|
|
40
|
+
attrs: Record<string, unknown>;
|
|
41
|
+
}
|
|
42
|
+
interface IdAlias {
|
|
43
|
+
oldId: string;
|
|
44
|
+
newId: string;
|
|
45
|
+
reason: IdAliasReason;
|
|
46
|
+
}
|
|
47
|
+
interface FileFingerprint {
|
|
48
|
+
fileId: string;
|
|
49
|
+
contentHash: string;
|
|
50
|
+
/**
|
|
51
|
+
* Comment/whitespace-insensitive parse-structure hash (C-18). Absent on
|
|
52
|
+
* snapshots written before C-18 (they can only reuse whole unchanged files).
|
|
53
|
+
*/
|
|
54
|
+
structuralHash?: string;
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
interface SnapshotInsert {
|
|
58
|
+
ref: string;
|
|
59
|
+
commitHash?: string;
|
|
60
|
+
indexVersion: string;
|
|
61
|
+
attrs?: Record<string, unknown>;
|
|
62
|
+
}
|
|
63
|
+
/**
|
|
64
|
+
* The code graph's query surface over store-sqlite's snapshot-scoped kit tables
|
|
65
|
+
* plus this package's domain tables. Replaces codewatch's `GraphDatabase`: the
|
|
66
|
+
* connection, pragmas, and migration runner are the kit's; what is left here is
|
|
67
|
+
* the code-specific statements the indexer and its readers need.
|
|
68
|
+
*/
|
|
69
|
+
declare class CodeGraphStore {
|
|
70
|
+
readonly db: Db;
|
|
71
|
+
private readonly statements;
|
|
72
|
+
constructor(db: Db);
|
|
73
|
+
createSnapshot(input: SnapshotInsert): number;
|
|
74
|
+
insertNodes(snapshotId: number, nodes: readonly GraphNode[]): void;
|
|
75
|
+
insertEdges(snapshotId: number, edges: readonly GraphEdge[]): void;
|
|
76
|
+
insertMetrics(snapshotId: number, metrics: readonly GraphMetric[]): void;
|
|
77
|
+
insertAliases(snapshotId: number, aliases: readonly IdAlias[]): void;
|
|
78
|
+
insertFingerprints(snapshotId: number, fingerprints: readonly FileFingerprint[]): void;
|
|
79
|
+
getSnapshot(id: number): SnapshotRow | null;
|
|
80
|
+
listSnapshots(opts?: {
|
|
81
|
+
ref?: string;
|
|
82
|
+
limit?: number;
|
|
83
|
+
}): SnapshotRow[];
|
|
84
|
+
getLatestSnapshotByRef(ref: string): SnapshotRow | null;
|
|
85
|
+
getNode(snapshotId: number, id: string): GraphNode | null;
|
|
86
|
+
/**
|
|
87
|
+
* File-level structural graph by default: the per-symbol layer is excluded so
|
|
88
|
+
* consumers that reason about module structure see the graph they expect. Pass
|
|
89
|
+
* `includeSymbols` for the symbol layer (the reuse basis needs it).
|
|
90
|
+
*/
|
|
91
|
+
listNodes(snapshotId: number, opts?: {
|
|
92
|
+
includeSymbols?: boolean;
|
|
93
|
+
}): GraphNode[];
|
|
94
|
+
/** See {@link listNodes}: `references` edges are the symbol layer, excluded by default. */
|
|
95
|
+
listEdges(snapshotId: number, opts?: {
|
|
96
|
+
includeReferences?: boolean;
|
|
97
|
+
}): GraphEdge[];
|
|
98
|
+
listMetrics(snapshotId: number): GraphMetric[];
|
|
99
|
+
listAliases(snapshotId: number): IdAlias[];
|
|
100
|
+
listFingerprints(snapshotId: number): FileFingerprint[];
|
|
101
|
+
close(): void;
|
|
102
|
+
}
|
|
103
|
+
/** Open (creating if absent) a code graph database and bring it to the current schema. */
|
|
104
|
+
declare function openCodeGraph(dbPath: string): CodeGraphStore;
|
|
105
|
+
|
|
106
|
+
/**
|
|
107
|
+
* Kit table names. The code graph is snapshot-scoped end to end: a whole repo is
|
|
108
|
+
* re-indexed together, so `snapshotTableDdl` + `entitySnapTableDdl` are the right
|
|
109
|
+
* time model and the interval `edgeTableDdl` is not — one unchanged import would
|
|
110
|
+
* otherwise write a row per file per snapshot. The edge table below is therefore
|
|
111
|
+
* the snapshot-scoped counterpart, owned here.
|
|
112
|
+
*/
|
|
113
|
+
declare const KIT: {
|
|
114
|
+
readonly snapshot: "snapshot";
|
|
115
|
+
readonly entitySnap: "node";
|
|
116
|
+
readonly cacheBlob: "blob_cache";
|
|
117
|
+
};
|
|
118
|
+
/**
|
|
119
|
+
* Domain tables. `edge` is the snapshot-scoped edge shape (the kit's own edge
|
|
120
|
+
* table is bi-temporal); `metric` holds index-time measurements keyed by node;
|
|
121
|
+
* `id_alias` carries a node id across a rename so two snapshots line up; and
|
|
122
|
+
* `file_fingerprint` is the reuse basis the incremental indexer diffs against.
|
|
123
|
+
*/
|
|
124
|
+
declare const DOMAIN_DDL = "\n CREATE TABLE IF NOT EXISTS edge (\n snapshot_id INTEGER NOT NULL,\n src_id TEXT NOT NULL,\n dst_id TEXT NOT NULL,\n kind TEXT NOT NULL,\n attrs TEXT,\n PRIMARY KEY (snapshot_id, src_id, dst_id, kind)\n );\n CREATE INDEX IF NOT EXISTS idx_edge_dst ON edge(snapshot_id, dst_id);\n CREATE INDEX IF NOT EXISTS idx_edge_kind ON edge(snapshot_id, kind);\n\n CREATE TABLE IF NOT EXISTS metric (\n snapshot_id INTEGER NOT NULL,\n node_id TEXT NOT NULL,\n name TEXT NOT NULL,\n value REAL,\n unit TEXT,\n PRIMARY KEY (snapshot_id, node_id, name)\n );\n CREATE INDEX IF NOT EXISTS idx_metric_name ON metric(snapshot_id, name);\n\n CREATE TABLE IF NOT EXISTS id_alias (\n snapshot_id INTEGER NOT NULL,\n old_id TEXT NOT NULL,\n new_id TEXT NOT NULL,\n reason TEXT NOT NULL,\n PRIMARY KEY (snapshot_id, old_id, new_id)\n );\n\n CREATE TABLE IF NOT EXISTS file_fingerprint (\n snapshot_id INTEGER NOT NULL,\n file_id TEXT NOT NULL,\n content_hash TEXT NOT NULL,\n structural_hash TEXT,\n PRIMARY KEY (snapshot_id, file_id)\n );\n";
|
|
125
|
+
declare const MIGRATIONS: Migration[];
|
|
126
|
+
|
|
127
|
+
/**
|
|
128
|
+
* Bumping this invalidates every reuse basis: a snapshot written by a different
|
|
129
|
+
* index version is never reused, so a change to node/edge shape can never be
|
|
130
|
+
* carried forward from an incompatible graph.
|
|
131
|
+
*/
|
|
132
|
+
declare const INDEX_VERSION = "0.11.0";
|
|
133
|
+
interface IndexOptions {
|
|
134
|
+
/** Roots to walk. Node ids are still rooted at the git toplevel, so importers across roots share an id space. */
|
|
135
|
+
paths: string[];
|
|
136
|
+
ref?: string;
|
|
137
|
+
commitHash?: string;
|
|
138
|
+
tsConfig?: string;
|
|
139
|
+
detectRenames?: boolean;
|
|
140
|
+
computeMetrics?: boolean;
|
|
141
|
+
/**
|
|
142
|
+
* Reuse the prior snapshot for byte-identical files: skip their tree-sitter
|
|
143
|
+
* parse + ts-morph extract and carry their nodes/edges/source metrics forward.
|
|
144
|
+
* Defaults to `true`; the reuse path is equivalent to a full index, so the only
|
|
145
|
+
* reason to disable it is to rebuild from scratch.
|
|
146
|
+
*/
|
|
147
|
+
incremental?: boolean;
|
|
148
|
+
}
|
|
149
|
+
interface IndexResult {
|
|
150
|
+
snapshotId: number;
|
|
151
|
+
files: number;
|
|
152
|
+
nodes: number;
|
|
153
|
+
edges: number;
|
|
154
|
+
aliases: number;
|
|
155
|
+
metrics: number;
|
|
156
|
+
/** Files whose parse + extract were skipped by reusing the prior snapshot. */
|
|
157
|
+
reused: number;
|
|
158
|
+
/** Files parsed + extracted this run (new, changed, or full index). */
|
|
159
|
+
reparsed: number;
|
|
160
|
+
/** Parsed files whose extract was skipped as cosmetic; a subset of `reparsed`. */
|
|
161
|
+
cosmetic: number;
|
|
162
|
+
nodesByKind: Record<string, number>;
|
|
163
|
+
edgesByKind: Record<string, number>;
|
|
164
|
+
}
|
|
165
|
+
/**
|
|
166
|
+
* Index one or more roots into a new snapshot: walk, read, parse what changed,
|
|
167
|
+
* extract, annotate roles, compute metrics, and persist. Files unchanged since
|
|
168
|
+
* the prior snapshot are carried forward rather than re-parsed, so the result
|
|
169
|
+
* matches a full index regardless of how much was reused.
|
|
170
|
+
*/
|
|
171
|
+
declare function indexPaths(store: CodeGraphStore, options: IndexOptions): Promise<IndexResult>;
|
|
172
|
+
|
|
173
|
+
interface ParsedFile {
|
|
174
|
+
tree: Tree;
|
|
175
|
+
content: string;
|
|
176
|
+
filePath: string;
|
|
177
|
+
language: string;
|
|
178
|
+
}
|
|
179
|
+
interface Extractor<T> {
|
|
180
|
+
name: string;
|
|
181
|
+
extract(file: ParsedFile): T[];
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
declare function parseFile(content: string, filePath: string, language: string): Promise<ParsedFile>;
|
|
185
|
+
declare function getSupportedLanguages(): string[];
|
|
186
|
+
|
|
187
|
+
declare function shouldIncludeFile(filePath: string, languages: string[]): boolean;
|
|
188
|
+
declare function getLanguageFromPath(filePath: string): string | null;
|
|
189
|
+
|
|
190
|
+
interface LanguageExtractorOptions {
|
|
191
|
+
repoRoot: string;
|
|
192
|
+
tsConfigPath?: string;
|
|
193
|
+
}
|
|
194
|
+
/**
|
|
195
|
+
* One extractor over every supported language: each delegate already returns an
|
|
196
|
+
* empty fragment list for files it does not own, so dispatch is a fold. Keeps
|
|
197
|
+
* `assembleFragments` language-agnostic.
|
|
198
|
+
*/
|
|
199
|
+
declare class LanguageExtractor implements Extractor<GraphFragment> {
|
|
200
|
+
readonly name = "language-graph";
|
|
201
|
+
private readonly delegates;
|
|
202
|
+
constructor(options: LanguageExtractorOptions);
|
|
203
|
+
extract(file: ParsedFile): GraphFragment[];
|
|
204
|
+
}
|
|
205
|
+
|
|
206
|
+
/**
|
|
207
|
+
* Tree-sitter Python extraction. Deliberately narrower than the ts-morph side:
|
|
208
|
+
* no type checker, so imports resolve by dotted path against the tree rather
|
|
209
|
+
* than by symbol resolution, and symbols are the declarations the same
|
|
210
|
+
* tree-sitter walk already finds for complexity. Enough to place Python files in
|
|
211
|
+
* the graph with their dependencies; a semantic Python layer is a follow-up.
|
|
212
|
+
*/
|
|
213
|
+
declare class PythonGraphExtractor implements Extractor<GraphFragment> {
|
|
214
|
+
private readonly repoRoot;
|
|
215
|
+
readonly name = "python-graph";
|
|
216
|
+
constructor(repoRoot: string);
|
|
217
|
+
extract(file: ParsedFile): GraphFragment[];
|
|
218
|
+
private collectImports;
|
|
219
|
+
/**
|
|
220
|
+
* Resolve `a.b.c` (or a leading-dot relative import) to an in-repo file id by
|
|
221
|
+
* walking the dotted path from the importing file's package and then from the
|
|
222
|
+
* repo root, which covers both intra-package and root-relative layouts.
|
|
223
|
+
*/
|
|
224
|
+
private resolveDotted;
|
|
225
|
+
}
|
|
226
|
+
|
|
227
|
+
interface TsMorphGraphExtractorOptions {
|
|
228
|
+
repoRoot: string;
|
|
229
|
+
tsConfigPath?: string;
|
|
230
|
+
project?: Project;
|
|
231
|
+
}
|
|
232
|
+
declare class TsMorphGraphExtractor implements Extractor<GraphFragment> {
|
|
233
|
+
readonly name = "ts-morph-graph";
|
|
234
|
+
private readonly repoRoot;
|
|
235
|
+
private readonly tsConfigPath?;
|
|
236
|
+
private project?;
|
|
237
|
+
constructor(options: TsMorphGraphExtractorOptions);
|
|
238
|
+
extract(file: ParsedFile): GraphFragment[];
|
|
239
|
+
private ensureProject;
|
|
240
|
+
private loadSourceFile;
|
|
241
|
+
private buildFileAndModuleNodes;
|
|
242
|
+
private collectEdges;
|
|
243
|
+
private recordImportEdge;
|
|
244
|
+
/**
|
|
245
|
+
* Split an import into per-export `references` edges targeting `symbol` nodes,
|
|
246
|
+
* so per-symbol utilization (which exports are hot) falls out of the same
|
|
247
|
+
* inbound-weight sum that powers file utilization (C-53). Extracted *forward*
|
|
248
|
+
* from the importing file — source-local, exactly like the file-level import
|
|
249
|
+
* weight — which sidesteps the reuse-breaking reverse `findReferences` query:
|
|
250
|
+
* an unchanged file's outbound reference edges are carried forward verbatim.
|
|
251
|
+
*/
|
|
252
|
+
private recordSymbolReferences;
|
|
253
|
+
/**
|
|
254
|
+
* Resolve an imported name to the id of the `symbol` node for the export that
|
|
255
|
+
* actually *declares* it — following re-exports through barrels via ts-morph's
|
|
256
|
+
* own `getExportedDeclarations`, so `import { x } from "./barrel"` credits the
|
|
257
|
+
* origin file's `x`, not the barrel. When ts-morph can't resolve the specifier
|
|
258
|
+
* (e.g. an extensionless relative import), falls back to a filesystem-based
|
|
259
|
+
* re-export walk that still sees through barrel hops (C-70). Returns null for
|
|
260
|
+
* external / unresolved targets.
|
|
261
|
+
* A name that resolves to no local declaration (an aliased re-export the
|
|
262
|
+
* origin doesn't export under this name) yields a dangling edge, pruned after
|
|
263
|
+
* assembly (`pruneDanglingReferences`).
|
|
264
|
+
*/
|
|
265
|
+
private resolveSymbolTarget;
|
|
266
|
+
private recordReExportEdge;
|
|
267
|
+
/**
|
|
268
|
+
* Resolve an import/export specifier to a graph node id (an in-repo file id or
|
|
269
|
+
* an `external` node, creating the external node on first sight), or `null`
|
|
270
|
+
* when it should not produce an edge.
|
|
271
|
+
*/
|
|
272
|
+
private resolveTarget;
|
|
273
|
+
private resolveInternal;
|
|
274
|
+
/**
|
|
275
|
+
* Resolve a relative specifier ts-morph failed to link by walking the
|
|
276
|
+
* filesystem from the importing file. ts-morph's NodeNext resolution only
|
|
277
|
+
* links relative imports that carry an explicit `.js` extension, so
|
|
278
|
+
* extensionless bundler-style imports (`../types`, common in code outside the
|
|
279
|
+
* tsconfig project such as `dashboard/`) resolve to nothing and would
|
|
280
|
+
* otherwise fall through to the `npm:` external bucket (C-44). Resolves
|
|
281
|
+
* through the ts-morph filesystem host so it honours in-memory test fixtures.
|
|
282
|
+
*/
|
|
283
|
+
private resolveRelativeInternal;
|
|
284
|
+
private inRepoFileId;
|
|
285
|
+
}
|
|
286
|
+
|
|
287
|
+
/**
|
|
288
|
+
* Build the `file` + `module` nodes for a source file from its path alone.
|
|
289
|
+
* Path-derived and deterministic, so the incremental indexer can reconstruct an
|
|
290
|
+
* unchanged file's nodes without re-parsing it — they are byte-for-byte the same
|
|
291
|
+
* nodes the extractor would emit. Keep this the single source of truth for
|
|
292
|
+
* file/module node shape.
|
|
293
|
+
*/
|
|
294
|
+
declare function buildFileModuleNodes(repoRoot: string, absPath: string): GraphNode[];
|
|
295
|
+
|
|
296
|
+
declare function fileId(repoRoot: string, absPath: string): string;
|
|
297
|
+
declare function moduleId(repoRoot: string, absPath: string): string;
|
|
298
|
+
declare function parentModuleId(id: string): string | null;
|
|
299
|
+
declare function packageId(name: string): string;
|
|
300
|
+
declare const SYMBOL_ID_SEP = "#";
|
|
301
|
+
declare function symbolId(fileId: string, exportName: string): string;
|
|
302
|
+
/**
|
|
303
|
+
* Inverse of {@link symbolId}: split a `<fileId>#<name>` symbol id back into its
|
|
304
|
+
* declaring file and export name. Returns null for an id with no separator (a
|
|
305
|
+
* plain file id), so callers can filter the symbol layer cleanly. `#` is illegal
|
|
306
|
+
* in both posix paths and JS identifiers, so the first occurrence is the split.
|
|
307
|
+
*/
|
|
308
|
+
declare function parseSymbolId(id: string): {
|
|
309
|
+
fileId: string;
|
|
310
|
+
name: string;
|
|
311
|
+
} | null;
|
|
312
|
+
declare function externalId(specifier: string): string;
|
|
313
|
+
|
|
314
|
+
/** SHA-256 of file content. The fingerprint that gates per-file reuse. */
|
|
315
|
+
declare function hashContent(content: string): string;
|
|
316
|
+
/**
|
|
317
|
+
* Comment/whitespace-insensitive hash of a file's parse structure (C-18). Walks
|
|
318
|
+
* the tree-sitter tree emitting each node's type plus leaf text, skipping comment
|
|
319
|
+
* nodes; whitespace and positions are absent from the tree, so a pure reformat or
|
|
320
|
+
* comment edit yields the SAME signature while any token change yields a new one.
|
|
321
|
+
* Two files with an equal signature therefore produce identical edges and AST
|
|
322
|
+
* metrics and differ only in line spans (and loc) — the COSMETIC reuse class.
|
|
323
|
+
*/
|
|
324
|
+
declare function structuralSignature(file: ParsedFile): string;
|
|
325
|
+
/** The languages that have an extractor; anything else is skipped by the walk. */
|
|
326
|
+
type SourceLanguage = "typescript" | "python";
|
|
327
|
+
interface ReadFile {
|
|
328
|
+
filePath: string;
|
|
329
|
+
language: SourceLanguage;
|
|
330
|
+
content: string;
|
|
331
|
+
hash: string;
|
|
332
|
+
}
|
|
333
|
+
/** Read + hash every source file. Cheap I/O; the parse/extract it gates is not. */
|
|
334
|
+
declare function readSourceFiles(filePaths: readonly string[]): Promise<ReadFile[]>;
|
|
335
|
+
/**
|
|
336
|
+
* Everything from a prior snapshot needed to rebuild an unchanged file's
|
|
337
|
+
* contribution to the graph without re-parsing it: its content fingerprint,
|
|
338
|
+
* the node table (to look up external-node names), outbound edges grouped by
|
|
339
|
+
* source, and the source-content metrics (loc, complexity, lcom4, ...).
|
|
340
|
+
*/
|
|
341
|
+
interface ReuseBasis {
|
|
342
|
+
snapshotId: number;
|
|
343
|
+
fingerprints: Map<string, string>;
|
|
344
|
+
/** Per-file structural signature (C-18); empty for a pre-C-18 basis. */
|
|
345
|
+
structuralHashes: Map<string, string>;
|
|
346
|
+
nodesById: Map<string, GraphNode>;
|
|
347
|
+
edgesBySrc: Map<string, GraphEdge[]>;
|
|
348
|
+
sourceMetricsByFile: Map<string, GraphMetric[]>;
|
|
349
|
+
/** Symbol nodes grouped by their declaring file id, to carry forward for reused files (C-53). */
|
|
350
|
+
symbolsByFile: Map<string, GraphNode[]>;
|
|
351
|
+
}
|
|
352
|
+
|
|
353
|
+
/**
|
|
354
|
+
* Recursively collect the source files under `rootDirs` that pass the ingest
|
|
355
|
+
* filter for `languages`, deduped by absolute path across roots (walk order
|
|
356
|
+
* preserved). Extracted from the indexer (C-61): as a function nested inside the
|
|
357
|
+
* per-root loop, the recursive walker carried a cognitive-complexity nesting
|
|
358
|
+
* bonus that made it the indexer's single most complex function (cx 19, above
|
|
359
|
+
* the exported entry point). Hoisting it here — one recursion helper, one
|
|
360
|
+
* directory branch — drops it well under budget and makes the walker unit-
|
|
361
|
+
* testable in isolation.
|
|
362
|
+
*/
|
|
363
|
+
declare function walkSourceFiles(rootDirs: readonly string[], languages: readonly string[]): Promise<string[]>;
|
|
364
|
+
|
|
365
|
+
/** 1-based inclusive line span of a declaration, for coverage range-attribution (C-63). */
|
|
366
|
+
interface LineSpan {
|
|
367
|
+
startLine: number;
|
|
368
|
+
endLine: number;
|
|
369
|
+
}
|
|
370
|
+
/**
|
|
371
|
+
* Every function/method/class a file DECLARES, mapped to its 1-based line span
|
|
372
|
+
* (the model-B symbol surface, C-64, now carrying spans for C-63 coverage
|
|
373
|
+
* range-containment). A superset of the file's exports: internal helpers like
|
|
374
|
+
* `mergeFragments` are included so they get a `symbol` node (and, by name match,
|
|
375
|
+
* their complexity + coverage) even though nothing imports them. Names come from
|
|
376
|
+
* the same tree-sitter walk that computes complexity, so they never drift.
|
|
377
|
+
* Anonymous declarations (a default-exported arrow, inline callbacks) contribute
|
|
378
|
+
* no name and are skipped. A name declared more than once keeps its last span
|
|
379
|
+
* (overloads / same-named methods are rare; range lookup still resolves most).
|
|
380
|
+
*/
|
|
381
|
+
declare function collectDeclaredSpans(file: ParsedFile): Map<string, LineSpan>;
|
|
382
|
+
/** Declared names only — the Set view over {@link collectDeclaredSpans}. */
|
|
383
|
+
declare function collectDeclaredNames(file: ParsedFile): Set<string>;
|
|
384
|
+
|
|
385
|
+
/** Edge weight (C-51 reference count), floored at 1 for unweighted edges. */
|
|
386
|
+
declare function edgeWeight(e: GraphEdge): number;
|
|
387
|
+
/**
|
|
388
|
+
* Drop `references` edges (C-53) whose target `symbol` node doesn't exist — an
|
|
389
|
+
* aliased re-export chain, or a barrel that changed under a reused file, can
|
|
390
|
+
* resolve a name to an origin that doesn't declare it. Only `references` can
|
|
391
|
+
* dangle (imports/re-exports resolve to always-emitted file/external nodes);
|
|
392
|
+
* metrics already guard unknown ids, so this just keeps the persisted edge set
|
|
393
|
+
* clean. Mutates `edges` in place.
|
|
394
|
+
*/
|
|
395
|
+
declare function pruneDanglingReferences(nodes: ReadonlyMap<string, GraphNode>, edges: Map<string, GraphEdge>): void;
|
|
396
|
+
/**
|
|
397
|
+
* Resolve edges that land on a barrel (`role="barrel"` — a bare `index.*`
|
|
398
|
+
* re-export file) onto the files the barrel actually re-exports from, so
|
|
399
|
+
* downstream signals measure the real dependency surface instead of the
|
|
400
|
+
* re-export plumbing. A cross-package `import … from "@codewatch/graph"`
|
|
401
|
+
* resolves to the barrel; without this every such import piles onto the barrel
|
|
402
|
+
* as an artificial hub while the module that truly does the work is
|
|
403
|
+
* under-credited.
|
|
404
|
+
*
|
|
405
|
+
* An inbound edge `F → B` (weight w) is split across B's outbound re-export
|
|
406
|
+
* targets `t_i` in proportion to each target's re-export weight `r_i` (C-51:
|
|
407
|
+
* the count of names B forwards from `t_i`). This CONSERVES w — no magnitude
|
|
408
|
+
* inflation, unlike a uniform fan-out that turns one import into N edges — and
|
|
409
|
+
* attributes it by how much of the barrel each target supplies. Resolution
|
|
410
|
+
* recurses through barrel chains (a barrel re-exporting a barrel) with a
|
|
411
|
+
* visited guard against re-export cycles; a barrel with no resolvable
|
|
412
|
+
* re-exports is left as its own target so the dependency is never dropped.
|
|
413
|
+
*
|
|
414
|
+
* Pure over the assembled node/edge set, so it is deterministic regardless of
|
|
415
|
+
* how much of an incremental index was reused.
|
|
416
|
+
*/
|
|
417
|
+
declare function resolveBarrelEdges(nodes: readonly GraphNode[], edges: readonly GraphEdge[]): GraphEdge[];
|
|
418
|
+
|
|
419
|
+
declare const ALL_ROLES: readonly NodeRole[];
|
|
420
|
+
interface RoleHints {
|
|
421
|
+
/** File begins with a `#!` shebang, i.e. it is an executable entry point. */
|
|
422
|
+
hasShebang?: boolean;
|
|
423
|
+
/** File is codegen output (`.gitattributes linguist-generated` or heuristic). */
|
|
424
|
+
isGenerated?: boolean;
|
|
425
|
+
}
|
|
426
|
+
declare function classifyRole(id: string, hints?: RoleHints): NodeRole;
|
|
427
|
+
interface AnnotateRolesOptions {
|
|
428
|
+
/** Node ids whose source begins with a `#!` shebang. */
|
|
429
|
+
shebangIds?: ReadonlySet<string>;
|
|
430
|
+
/** Node ids detected as generated (codegen output). */
|
|
431
|
+
generatedIds?: ReadonlySet<string>;
|
|
432
|
+
}
|
|
433
|
+
interface ReadFileLike {
|
|
434
|
+
filePath: string;
|
|
435
|
+
content: string;
|
|
436
|
+
}
|
|
437
|
+
/**
|
|
438
|
+
* Derive per-file role hints for a batch of read files in one pass: the shebang
|
|
439
|
+
* set (executable entries) and the generated set (codegen output, detected via
|
|
440
|
+
* `.gitattributes linguist-generated` under `idRoot` plus filename/path
|
|
441
|
+
* heuristics). Lives here so the indexer stays at its file-size ceiling.
|
|
442
|
+
*/
|
|
443
|
+
declare function computeRoleHints(readFiles: readonly ReadFileLike[], idRoot: string, toId: (root: string, filePath: string) => string): AnnotateRolesOptions;
|
|
444
|
+
declare function annotateRoles(nodes: readonly GraphNode[], options?: AnnotateRolesOptions): GraphNode[];
|
|
445
|
+
|
|
446
|
+
/** True when a path looks generated by filename/dir convention alone. */
|
|
447
|
+
declare function isGeneratedByHeuristic(id: string): boolean;
|
|
448
|
+
/**
|
|
449
|
+
* A file is generated when it matches a repo-declared `linguist-generated`
|
|
450
|
+
* pattern or the filename/path heuristic. `patterns` come from
|
|
451
|
+
* {@link parseGeneratedPatterns}; pass `[]` when no `.gitattributes` exists.
|
|
452
|
+
*/
|
|
453
|
+
declare function isGeneratedFile(id: string, patterns: readonly RegExp[]): boolean;
|
|
454
|
+
/** Read `<rootDir>/.gitattributes` and compile its linguist-generated patterns. */
|
|
455
|
+
declare function loadGeneratedPatterns(rootDir: string): RegExp[];
|
|
456
|
+
|
|
457
|
+
declare function canonicalMetricName(name: string): string;
|
|
458
|
+
declare function canonicalRole(name: string): NodeRole;
|
|
459
|
+
declare function canonicalEdgeKind(name: string): EdgeKind;
|
|
460
|
+
|
|
461
|
+
interface RenamePair {
|
|
462
|
+
oldPath: string;
|
|
463
|
+
newPath: string;
|
|
464
|
+
similarity: number;
|
|
465
|
+
}
|
|
466
|
+
interface DetectRenamesOptions {
|
|
467
|
+
repoRoot: string;
|
|
468
|
+
fromCommit: string;
|
|
469
|
+
toCommit?: string;
|
|
470
|
+
similarityThreshold?: number;
|
|
471
|
+
}
|
|
472
|
+
declare function detectGitHead(repoRoot: string): string | null;
|
|
473
|
+
declare function isInsideGitRepo(repoRoot: string): boolean;
|
|
474
|
+
declare function detectGitToplevel(cwd: string): string | null;
|
|
475
|
+
declare function detectRenames(options: DetectRenamesOptions): RenamePair[];
|
|
476
|
+
declare function buildAliases(repoRoot: string, pairs: readonly RenamePair[]): IdAlias[];
|
|
477
|
+
|
|
478
|
+
declare function computeMetrics(nodes: readonly GraphNode[], edges: readonly GraphEdge[]): GraphMetric[];
|
|
479
|
+
|
|
480
|
+
/**
|
|
481
|
+
* Names of every metric `computeSourceMetrics` can emit. These are pure
|
|
482
|
+
* functions of a file's content, so the incremental indexer can carry them
|
|
483
|
+
* forward unchanged for a byte-identical file instead of re-parsing it. If a
|
|
484
|
+
* new source metric is added above, add its name here — the incremental
|
|
485
|
+
* round-trip test will fail loudly if this set drifts out of sync.
|
|
486
|
+
*/
|
|
487
|
+
declare const SOURCE_METRIC_NAMES: ReadonlySet<string>;
|
|
488
|
+
/**
|
|
489
|
+
* Per-file source metrics. `symbolNamesByFile` maps a file id to the names of
|
|
490
|
+
* the `symbol` nodes it declares; when supplied, per-function complexity is
|
|
491
|
+
* emitted on those symbol nodes (C-58; all declared functions since C-64). It
|
|
492
|
+
* defaults to empty, so existing callers/tests that don't thread the symbol
|
|
493
|
+
* layer keep emitting only file-level metrics.
|
|
494
|
+
*/
|
|
495
|
+
declare function computeSourceMetrics(files: readonly ParsedFile[], fileIdOf: (filePath: string) => string, symbolNamesByFile?: ReadonlyMap<string, ReadonlySet<string>>): GraphMetric[];
|
|
496
|
+
|
|
497
|
+
interface IndexerMetricsInput {
|
|
498
|
+
nodes: Map<string, GraphNode>;
|
|
499
|
+
edges: Map<string, GraphEdge>;
|
|
500
|
+
/** Files (re)parsed this run — source metrics are computed fresh for these. */
|
|
501
|
+
parsedFiles: ParsedFile[];
|
|
502
|
+
/** Source metrics carried forward verbatim for reused (unchanged) files. */
|
|
503
|
+
reusedSourceMetrics: GraphMetric[];
|
|
504
|
+
idRoot: string;
|
|
505
|
+
}
|
|
506
|
+
/**
|
|
507
|
+
* Assemble the metric set for a snapshot: graph-wide degree metrics over the
|
|
508
|
+
* full node/edge set, freshly-computed source metrics for (re)parsed files, and
|
|
509
|
+
* reused source metrics carried forward for unchanged files. Everything but the
|
|
510
|
+
* reused source metrics is recomputed over the full set, so the result matches a
|
|
511
|
+
* full index regardless of how much was reused.
|
|
512
|
+
*
|
|
513
|
+
* Git-history metrics (churn, ownership, change coupling, test coverage) are
|
|
514
|
+
* deliberately absent here; they are a separate analysis layer, deferred with
|
|
515
|
+
* the rest of codewatch's analyses.
|
|
516
|
+
*/
|
|
517
|
+
declare function buildIndexerMetrics(input: IndexerMetricsInput): GraphMetric[];
|
|
518
|
+
|
|
519
|
+
/**
|
|
520
|
+
* The free-function reading surface named in the package API. Each delegates to
|
|
521
|
+
* the store's method of the same name, so a caller that only reads a snapshot
|
|
522
|
+
* never has to hold on to the store's shape.
|
|
523
|
+
*/
|
|
524
|
+
declare function listNodes(store: CodeGraphStore, snapshotId: number, opts?: {
|
|
525
|
+
includeSymbols?: boolean;
|
|
526
|
+
}): GraphNode[];
|
|
527
|
+
declare function listEdges(store: CodeGraphStore, snapshotId: number, opts?: {
|
|
528
|
+
includeReferences?: boolean;
|
|
529
|
+
}): GraphEdge[];
|
|
530
|
+
declare function listMetrics(store: CodeGraphStore, snapshotId: number): GraphMetric[];
|
|
531
|
+
|
|
532
|
+
export { ALL_ROLES, CodeGraphStore, DOMAIN_DDL, type EdgeKind, type Extractor, type FileFingerprint, type GraphEdge, type GraphFragment, type GraphMetric, type GraphNode, INDEX_VERSION, type IdAlias, type IdAliasReason, type IndexOptions, type IndexResult, KIT, LanguageExtractor, type LanguageExtractorOptions, type LineSpan, MIGRATIONS, type NodeKind, type NodeRole, type ParsedFile, PythonGraphExtractor, type ReadFile, type ReuseBasis, SOURCE_METRIC_NAMES, SYMBOL_ID_SEP, type SnapshotInsert, type SnapshotRow, type SourceLanguage, TsMorphGraphExtractor, type TsMorphGraphExtractorOptions, annotateRoles, buildAliases, buildFileModuleNodes, buildIndexerMetrics, canonicalEdgeKind, canonicalMetricName, canonicalRole, classifyRole, collectDeclaredNames, collectDeclaredSpans, computeMetrics, computeRoleHints, computeSourceMetrics, detectGitHead, detectGitToplevel, detectRenames, edgeWeight, externalId, fileId, getLanguageFromPath, getSupportedLanguages, hashContent, indexPaths, isGeneratedByHeuristic, isGeneratedFile, isInsideGitRepo, listEdges, listMetrics, listNodes, loadGeneratedPatterns, moduleId, openCodeGraph, packageId, parentModuleId, parseFile, parseSymbolId, pruneDanglingReferences, readSourceFiles, resolveBarrelEdges, shouldIncludeFile, structuralSignature, symbolId, walkSourceFiles };
|