kirograph 0.16.1 → 0.17.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. package/README.md +225 -8
  2. package/dist/bin/commands/affected.js +40 -5
  3. package/dist/bin/commands/affected.js.map +2 -2
  4. package/dist/bin/commands/data.js +665 -0
  5. package/dist/bin/commands/data.js.map +7 -0
  6. package/dist/bin/commands/help.js +5 -0
  7. package/dist/bin/commands/help.js.map +2 -2
  8. package/dist/bin/installer/config-prompt.js +28 -1
  9. package/dist/bin/installer/config-prompt.js.map +2 -2
  10. package/dist/bin/installer/index.js +28 -1
  11. package/dist/bin/installer/index.js.map +2 -2
  12. package/dist/bin/installer/steering.js +33 -0
  13. package/dist/bin/installer/steering.js.map +2 -2
  14. package/dist/bin/installer/targets/index.js.map +1 -1
  15. package/dist/bin/installer/targets/kiro.js +2 -2
  16. package/dist/bin/installer/targets/kiro.js.map +2 -2
  17. package/dist/bin/kirograph.js +3 -1
  18. package/dist/bin/kirograph.js.map +3 -3
  19. package/dist/bin/progress.js +30 -0
  20. package/dist/bin/progress.js.map +2 -2
  21. package/dist/compression/naive-cost.js +25 -0
  22. package/dist/compression/naive-cost.js.map +2 -2
  23. package/dist/compression/tracker.js +11 -2
  24. package/dist/compression/tracker.js.map +2 -2
  25. package/dist/compression/types.js.map +1 -1
  26. package/dist/config.js +37 -1
  27. package/dist/config.js.map +2 -2
  28. package/dist/core/pipeline.js +56 -2
  29. package/dist/core/pipeline.js.map +2 -2
  30. package/dist/data/filters.js +105 -0
  31. package/dist/data/filters.js.map +7 -0
  32. package/dist/data/indexer.js +225 -0
  33. package/dist/data/indexer.js.map +7 -0
  34. package/dist/data/linker.js +150 -0
  35. package/dist/data/linker.js.map +7 -0
  36. package/dist/data/lint.js +109 -0
  37. package/dist/data/lint.js.map +7 -0
  38. package/dist/data/parsers/csv.js +132 -0
  39. package/dist/data/parsers/csv.js.map +7 -0
  40. package/dist/data/parsers/excel.js +88 -0
  41. package/dist/data/parsers/excel.js.map +7 -0
  42. package/dist/data/parsers/index.js +79 -0
  43. package/dist/data/parsers/index.js.map +7 -0
  44. package/dist/data/parsers/json-array.js +89 -0
  45. package/dist/data/parsers/json-array.js.map +7 -0
  46. package/dist/data/parsers/jsonl.js +84 -0
  47. package/dist/data/parsers/jsonl.js.map +7 -0
  48. package/dist/data/parsers/parquet.js +95 -0
  49. package/dist/data/parsers/parquet.js.map +7 -0
  50. package/dist/data/profiler.js +182 -0
  51. package/dist/data/profiler.js.map +7 -0
  52. package/dist/data/queries.js +512 -0
  53. package/dist/data/queries.js.map +7 -0
  54. package/dist/data/types.js +17 -0
  55. package/dist/data/types.js.map +7 -0
  56. package/dist/db/data-schema.sql +61 -0
  57. package/dist/db/database.js +11 -0
  58. package/dist/db/database.js.map +2 -2
  59. package/dist/mcp/tool-names.js +9 -1
  60. package/dist/mcp/tool-names.js.map +2 -2
  61. package/dist/mcp/tools.js +423 -0
  62. package/dist/mcp/tools.js.map +2 -2
  63. package/dist/types.js.map +1 -1
  64. package/package.json +4 -2
package/README.md CHANGED
@@ -31,7 +31,7 @@ The index is kept fresh automatically via Kiro hooks when using the Kiro integra
31
31
 
32
32
  ## How Indexing Works
33
33
 
34
- Indexing has three layers: **structural** (always on), **semantic** (opt-in), and **architecture** (opt-in).
34
+ Indexing has five layers: **structural** (always on), **semantic** (opt-in), **architecture** (opt-in), **documentation** (opt-in), and **data** (opt-in).
35
35
 
36
36
  ### Structural indexing
37
37
 
@@ -123,6 +123,27 @@ Enable via `kirograph install` or directly in `.kirograph/config.json`:
123
123
 
124
124
  See the [Documentation](#documentation-requires-enabledocs-true) section below for full details.
125
125
 
126
+ ### Data indexing (opt-in)
127
+
128
+ When `enableData: true` is set, KiroGraph indexes tabular data files (CSV, TSV, JSONL, JSON, Excel, Parquet) that live alongside your code — test fixtures, seed data, configuration tables, sample datasets. Inspired by [jDataMunch-MCP](https://github.com/jgravelle/jdatamunch-mcp) by [J. Gravelle](https://www.linkedin.com/in/j-gravelle-2778223/).
129
+
130
+ - **Streaming parser**: never loads full files into memory. Processes line-by-line (CSV/JSONL) or in chunks (Excel/Parquet)
131
+ - **Column profiling**: type inference, cardinality, null percentages, min/max, sample values
132
+ - **Server-side computation**: filters, aggregations, and joins run in SQLite. Only results enter the context window
133
+ - **Incremental**: content hash (SHA-256) skips unchanged files on re-index
134
+ - **Token savings**: 95–99% reduction vs reading raw data files (tracked in `kirograph_gain`)
135
+ - **Optional format deps**: CSV/TSV/JSONL/JSON are built-in (zero deps). Excel requires `xlsx`, Parquet requires `parquetjs-lite`
136
+
137
+ Enable via `kirograph install` or directly in `.kirograph/config.json`:
138
+
139
+ ```json
140
+ {
141
+ "enableData": true
142
+ }
143
+ ```
144
+
145
+ See the [Data](#data-requires-enabledata-true) section below for full details.
146
+
126
147
  ## Installation
127
148
 
128
149
  ### From npm (not yet available on npm registry)
@@ -247,7 +268,12 @@ Registers the KiroGraph MCP server. Used by both the IDE and the CLI agent:
247
268
  "kirograph_hotspots", "kirograph_surprising", "kirograph_diff",
248
269
  "kirograph_exec", "kirograph_gain"
249
270
  "kirograph_mem_search", "kirograph_mem_store",
250
- "kirograph_mem_timeline", "kirograph_mem_status"
271
+ "kirograph_mem_timeline", "kirograph_mem_status",
272
+ "kirograph_docs_toc", "kirograph_docs_search",
273
+ "kirograph_docs_section", "kirograph_docs_outline", "kirograph_docs_refs",
274
+ "kirograph_data_list", "kirograph_data_describe",
275
+ "kirograph_data_query", "kirograph_data_aggregate", "kirograph_data_search",
276
+ "kirograph_data_join", "kirograph_data_correlations", "kirograph_data_quality"
251
277
  ]
252
278
  }
253
279
  }
@@ -256,12 +282,13 @@ Registers the KiroGraph MCP server. Used by both the IDE and the CLI agent:
256
282
 
257
283
  ### IDE Hooks (`.kiro/hooks/`)
258
284
 
259
- Up to two hooks are installed (`.kiro.hook` extension):
285
+ Up to three hooks are installed (`.kiro.hook` extension):
260
286
 
261
287
  | Hook file | Event | Type | Behavior |
262
288
  |-----------|-------|------|----------|
263
289
  | `kirograph-sync-if-dirty.kiro.hook` | `agentStop` | `runCommand` | Runs `kirograph sync --quiet` when the agent stops, syncing any file changes from the session. The sync command skips unchanged files via content hashing, so it's fast even when nothing changed. |
264
290
  | `kirograph-compress-hint.kiro.hook` | `preToolUse` (shell) | `askAgent` | Reminds the agent to use `kirograph_exec` for commands that benefit from token compression (git, gh, test, lint, build, docker, aws, grep). Only installed when shell compression is enabled. |
291
+ | `kirograph-mem-capture.kiro.hook` | `agentStop` | `askAgent` | Prompts the agent to store important observations (decisions, errors, patterns) in memory at the end of each session. Only installed when memory is enabled. |
265
292
 
266
293
  The sync hook replaces the previous per-file approach (mark-dirty-on-save, mark-dirty-on-create, sync-on-delete). A single `agentStop` hook handles all file changes in one pass with zero overhead during active editing.
267
294
 
@@ -275,6 +302,8 @@ A custom agent for Kiro CLI that wires up the MCP server, references the steerin
275
302
  | `userPromptSubmit` | `kirograph sync-if-dirty --quiet` (keeps graph fresh within a session) |
276
303
  | `stop` | `kirograph sync-if-dirty --quiet` (deferred flush, mirrors IDE `agentStop`) |
277
304
 
305
+ > Note: The CLI agent format only supports `command` hooks (shell commands), not `askAgent` prompts. Memory capture and compression hints are handled via the steering file instructions instead — the agent reads them from `.kiro/steering/kirograph.md` which is referenced as a resource.
306
+
278
307
  Use it with:
279
308
 
280
309
  ```bash
@@ -667,6 +696,93 @@ Find code symbols referenced by a doc section, or doc sections that reference a
667
696
  | `nodeId` | string | - | Code symbol qualified name (find doc sections that reference it) |
668
697
  | `projectPath` | string | cwd | Project root path |
669
698
 
699
+ ### `kirograph_data_list` *(requires `enableData: true`)*
700
+
701
+ List all indexed datasets with row counts, column counts, and file sizes.
702
+
703
+ | Parameter | Type | Default | Description |
704
+ |-----------|------|---------|-------------|
705
+ | `projectPath` | string | cwd | Project root path |
706
+
707
+ ### `kirograph_data_describe` *(requires `enableData: true`)*
708
+
709
+ Full schema profile for a dataset: column names, inferred types, cardinality, null percentages, min/max values, and sample values.
710
+
711
+ | Parameter | Type | Default | Description |
712
+ |-----------|------|---------|-------------|
713
+ | `dataset` | string | required | Dataset ID (from `kirograph_data_list`) |
714
+ | `column` | string | - | Deep dive on a single column |
715
+ | `projectPath` | string | cwd | Project root path |
716
+
717
+ ### `kirograph_data_query` *(requires `enableData: true`)*
718
+
719
+ Filtered row retrieval with structured operators. Multiple filters are ANDed. All queries use parameterized SQL (zero injection surface).
720
+
721
+ | Parameter | Type | Default | Description |
722
+ |-----------|------|---------|-------------|
723
+ | `dataset` | string | required | Dataset ID |
724
+ | `filters` | Filter[] | - | Array of `{column, op, value}`. Ops: `eq`, `neq`, `gt`, `gte`, `lt`, `lte`, `contains`, `in`, `is_null`, `between` |
725
+ | `columns` | string[] | all | Column projection |
726
+ | `limit` | number | 500 | Max rows (hard cap: 500) |
727
+ | `offset` | number | 0 | Pagination offset |
728
+ | `projectPath` | string | cwd | Project root path |
729
+
730
+ ### `kirograph_data_aggregate` *(requires `enableData: true`)*
731
+
732
+ Server-side GROUP BY aggregation. Computation happens in SQLite; only results enter the context window.
733
+
734
+ | Parameter | Type | Default | Description |
735
+ |-----------|------|---------|-------------|
736
+ | `dataset` | string | required | Dataset ID |
737
+ | `groupBy` | string[] | required | Columns to group by |
738
+ | `metrics` | Metric[] | required | Array of `{column, op}`. Ops: `count`, `sum`, `avg`, `min`, `max`, `count_distinct` |
739
+ | `filters` | Filter[] | - | Pre-aggregation filters |
740
+ | `projectPath` | string | cwd | Project root path |
741
+
742
+ ### `kirograph_data_search` *(requires `enableData: true`)*
743
+
744
+ Search column names and sample values by keyword within a dataset.
745
+
746
+ | Parameter | Type | Default | Description |
747
+ |-----------|------|---------|-------------|
748
+ | `dataset` | string | required | Dataset ID |
749
+ | `query` | string | required | Search keyword |
750
+ | `projectPath` | string | cwd | Project root path |
751
+
752
+ ### `kirograph_data_join` *(requires `enableData: true`)*
753
+
754
+ SQL JOIN across two indexed datasets. Combines data without loading either file into context.
755
+
756
+ | Parameter | Type | Default | Description |
757
+ |-----------|------|---------|-------------|
758
+ | `left` | string | required | Left dataset ID |
759
+ | `right` | string | required | Right dataset ID |
760
+ | `leftColumn` | string | required | Join column from left dataset |
761
+ | `rightColumn` | string | required | Join column from right dataset |
762
+ | `type` | string | `inner` | Join type: `inner`, `left`, `right` |
763
+ | `columns` | string[] | all | Column projection (prefix with dataset ID) |
764
+ | `limit` | number | 100 | Max rows (hard cap: 500) |
765
+ | `projectPath` | string | cwd | Project root path |
766
+
767
+ ### `kirograph_data_correlations` *(requires `enableData: true`)*
768
+
769
+ Pairwise Pearson correlations between numeric columns. Discovers hidden relationships without loading data.
770
+
771
+ | Parameter | Type | Default | Description |
772
+ |-----------|------|---------|-------------|
773
+ | `dataset` | string | required | Dataset ID |
774
+ | `threshold` | number | 0.3 | Min absolute correlation to include |
775
+ | `projectPath` | string | cwd | Project root path |
776
+
777
+ ### `kirograph_data_quality` *(requires `enableData: true`)*
778
+
779
+ Data quality triage: rank columns by composite risk score (null rate, cardinality anomalies, type issues).
780
+
781
+ | Parameter | Type | Default | Description |
782
+ |-----------|------|---------|-------------|
783
+ | `dataset` | string | required | Dataset ID |
784
+ | `projectPath` | string | cwd | Project root path |
785
+
670
786
  ## CLI Reference
671
787
 
672
788
  ### Setup
@@ -949,6 +1065,11 @@ The `kirograph_gain` MCP tool exposes the same stats to the agent.
949
1065
  | `kirograph_architecture` | Not feasible manually | 4× output, min 7,500 |
950
1066
  | `kirograph_mem_search` | Re-read 3-5 files to recall past decisions + grep | ~5,800 tokens |
951
1067
  | `kirograph_mem_timeline` | Ask user or re-read session history | ~2,300 tokens |
1068
+ | `kirograph_data_list` | Run ls/find on data files + inspect each | ~3,500 tokens |
1069
+ | `kirograph_data_describe` | Read the full data file to understand schema | ~45,000 tokens |
1070
+ | `kirograph_data_query` | Read the full file and scan for matching rows | ~45,000 tokens |
1071
+ | `kirograph_data_aggregate` | Read the full file + reason about aggregation | ~52,500 tokens |
1072
+ | `kirograph_data_search` | Read file headers + grep for values | ~9,100 tokens |
952
1073
 
953
1074
  Constants used: 1,500 tokens per average source file (~200 lines), 800 tokens per grep result set, 2,000 tokens per directory listing. These are conservative estimates; in practice agents often read more files, retry failed searches, and explore dead ends.
954
1075
 
@@ -1146,6 +1267,65 @@ kirograph docs reembed # re-embed with current model
1146
1267
 
1147
1268
  **How code linking works:** When `docsLinkCode: true` (default), the indexer scans section content for backtick references (`` `functionName` ``), CamelCase identifiers, and snake_case patterns, then resolves them against the code graph. Matches are stored as `doc_code_refs` using `qualified_name` (stable across reindex).
1148
1269
 
1270
+ ### Data *(requires `enableData: true`)*
1271
+
1272
+ Tabular data navigation — list, describe, query, aggregate, search, join, correlate, and inspect data quality from the CLI.
1273
+
1274
+ ```bash
1275
+ # List datasets
1276
+ kirograph data list # all indexed datasets
1277
+ kirograph data list --json # JSON output
1278
+
1279
+ # Describe schema
1280
+ kirograph data describe tests-fixtures-users # full schema + column profiles + validation rules + sample hints
1281
+ kirograph data describe tests-fixtures-users --column email # deep dive on one column
1282
+ kirograph data describe tests-fixtures-users --json
1283
+
1284
+ # Query rows
1285
+ kirograph data query orders --filter status:eq:shipped --limit 10
1286
+ kirograph data query users --filter age:gt:18 --columns name,email
1287
+ kirograph data query products --filter price:between:10:50 --json
1288
+
1289
+ # Aggregate
1290
+ kirograph data aggregate orders --group-by region --metric sum:amount
1291
+ kirograph data aggregate users --group-by role --metric count:id --metric avg:age
1292
+ kirograph data aggregate orders --group-by status --metric count_distinct:customer_id --json
1293
+
1294
+ # Search columns
1295
+ kirograph data search orders "price" # find columns matching "price"
1296
+ kirograph data search users "email" # find columns matching "email"
1297
+
1298
+ # Join two datasets
1299
+ kirograph data join users orders --left-col id --right-col user_id
1300
+ kirograph data join users orders --left-col id --right-col user_id --type left --limit 50
1301
+
1302
+ # Correlations
1303
+ kirograph data correlations sales-data # Pearson correlations between numeric columns
1304
+ kirograph data correlations sales-data --threshold 0.5 # only strong correlations
1305
+
1306
+ # Quality
1307
+ kirograph data quality orders # rank columns by risk (null rate, cardinality anomalies)
1308
+
1309
+ # History & drift
1310
+ kirograph data history orders # show schema change history
1311
+ kirograph data drift orders # compare last two indexes (added/removed/changed columns)
1312
+
1313
+ # Indexing
1314
+ kirograph data index # incremental index (skips unchanged files)
1315
+ kirograph data reindex # force re-index all data files
1316
+
1317
+ # Maintenance
1318
+ kirograph data lint # validate index integrity (row counts, stale files, missing deps)
1319
+ ```
1320
+
1321
+ **How datasets are identified:** Each data file gets a stable ID derived from its relative path: `tests/fixtures/users.csv` → `tests-fixtures-users`. IDs remain stable across re-indexing.
1322
+
1323
+ **Filter format (CLI):** `column:op:value` — e.g. `age:gt:18`, `status:eq:active`, `price:between:10:50`. Multiple `--filter` flags are ANDed.
1324
+
1325
+ **Metric format (CLI):** `op:column` — e.g. `sum:amount`, `avg:price`, `count:id`, `count_distinct:customer_id`.
1326
+
1327
+ **How code linking works:** When `dataLinkCode: true` (default), the indexer scans source files for references to data file paths (`readFileSync('data/users.csv')`, `pd.read_csv(...)`, SQL `COPY FROM`, etc.) and stores matches in `data_code_refs`. This enables test fixture awareness in `kirograph affected` and dataset schema enrichment in `kirograph_context`.
1328
+
1149
1329
  ### Graph Export
1150
1330
 
1151
1331
  Export the full graph as an interactive dashboard. three files served from a local directory, no server required, works offline.
@@ -1248,23 +1428,59 @@ KiroGraph stores its config in `.kirograph/config.json`. You can edit it directl
1248
1428
 
1249
1429
  | Field | Type | Default | Description |
1250
1430
  |-------|------|---------|-------------|
1431
+ | **Indexing** | | | |
1251
1432
  | `languages` | string[] | `[]` | Limit indexing to specific languages (empty = all) |
1252
1433
  | `include` | string[] | `[]` | Glob patterns to include (empty = include everything not excluded) |
1253
1434
  | `exclude` | string[] | see below | Glob patterns to exclude |
1254
1435
  | `maxFileSize` | number | `1048576` | Skip files larger than this (bytes) |
1255
1436
  | `extractDocstrings` | boolean | `true` | Extract JSDoc, docstrings, and comments |
1256
1437
  | `trackCallSites` | boolean | `true` | Record line/column for call edges |
1438
+ | `frameworkHints` | string[] | auto | Override framework detection (e.g. `["react", "express"]`) |
1439
+ | `fuzzyResolutionThreshold` | number | `0.5` | Name matching threshold for cross-file resolution (0.0–1.0) |
1440
+ | `syncWarningThreshold` | number | `10` | Warn in `kirograph_status` when pending files exceed this count |
1441
+ | **Semantic Search** | | | |
1257
1442
  | `enableEmbeddings` | boolean | `false` | Generate semantic embeddings (opt-in) |
1258
1443
  | `embeddingModel` | string | `nomic-ai/nomic-embed-text-v1.5` | HuggingFace `feature-extraction` model ID |
1259
1444
  | `embeddingDim` | number | `768` | Output dimension of the chosen embedding model |
1260
- | `semanticEngine` | string | `cosine` | Search engine: `cosine`, `sqlite-vec`, `orama`, `pglite`, `lancedb`, `qdrant`, or `typesense` |
1445
+ | `semanticEngine` | string | `cosine` | Engine: `cosine`, `sqlite-vec`, `orama`, `pglite`, `lancedb`, `qdrant`, `typesense` |
1261
1446
  | `useVecIndex` | boolean | `false` | Deprecated alias for `semanticEngine: "sqlite-vec"` |
1262
- | `enableArchitecture` | boolean | `false` | Enable architecture analysis (package graph + layer detection, opt-in) |
1447
+ | `typesenseDashboard` | boolean | `false` | Open Typesense dashboard after indexing |
1448
+ | `qdrantDashboard` | boolean | `false` | Open Qdrant dashboard after indexing |
1449
+ | **Architecture** | | | |
1450
+ | `enableArchitecture` | boolean | `false` | Enable architecture analysis (package graph + layer detection) |
1263
1451
  | `architectureLayers` | object | - | Custom layer definitions: `{ "layerName": ["glob/**"] }` |
1452
+ | **Memory** | | | |
1453
+ | `enableMemory` | boolean | `false` | Enable persistent cross-session memory |
1454
+ | `memorySearchAlpha` | number | `0.5` | Blend weight for hybrid search (0 = FTS only, 1 = vector only) |
1455
+ | `memoryKeepRaw` | boolean | `true` | Store original text alongside compressed version |
1456
+ | `memoryMaxObservations` | number | `10000` | Max observations before auto-pruning oldest |
1457
+ | `memorySessionTimeout` | number | `3600000` | Session timeout in ms (default 1 hour) |
1458
+ | `memoryContextLimit` | number | `3` | Max observations surfaced in `kirograph_context` |
1459
+ | `memoryContextThreshold` | number | `0.3` | Min relevance score to surface in context |
1460
+ | `memoryExcludePatterns` | string[] | `[]` | Glob patterns for files to exclude from symbol linking |
1461
+ | **Documentation** | | | |
1462
+ | `enableDocs` | boolean | `false` | Enable documentation indexing (section-level retrieval) |
1463
+ | `docsInclude` | string[] | `["**/*.md", ...]` | Glob patterns for doc files to include |
1464
+ | `docsExclude` | string[] | `["node_modules/**", ...]` | Glob patterns for doc files to exclude |
1465
+ | `docsLinkCode` | boolean | `true` | Auto-link doc sections to code symbols |
1466
+ | `docsContextLimit` | number | `0` | Max doc sections in `kirograph_context` (0 = disabled) |
1467
+ | `docsContextThreshold` | number | `0.5` | Min confidence for doc refs in context |
1468
+ | `docsMaxFileSize` | number | `1048576` | Max doc file size in bytes |
1469
+ | `docsSummarization` | string | `first-sentence` | Summary strategy: `embedding`, `first-sentence`, `off` |
1470
+ | **Data** | | | |
1471
+ | `enableData` | boolean | `false` | Enable tabular data indexing and querying |
1472
+ | `dataInclude` | string[] | `["**/*.csv", ...]` | Glob patterns for data files to include |
1473
+ | `dataExclude` | string[] | `["node_modules/**", ...]` | Glob patterns for data files to exclude |
1474
+ | `dataLinkCode` | boolean | `true` | Auto-link data files to code symbols via path detection |
1475
+ | `dataContextLimit` | number | `0` | Max datasets in `kirograph_context` (0 = disabled) |
1476
+ | `dataMaxFileSize` | number | `52428800` | Max data file size in bytes (50MB) |
1477
+ | `dataMaxRows` | number | `1000000` | Max rows to index per file |
1478
+ | `dataQueryLimit` | number | `500` | Max rows returned per query (hard cap) |
1479
+ | `dataMaxResponseTokens` | number | `8000` | Max token budget per data tool response |
1480
+ | **Agent Behavior** | | | |
1481
+ | `cavemanMode` | string | `off` | Communication style: `off`, `lite`, `full`, `ultra` |
1482
+ | `shellCompressionLevel` | string | `normal` | Shell compression: `off`, `normal`, `aggressive`, `ultra` |
1264
1483
  | `minLogLevel` | string | `warn` | Log level: `debug`, `info`, `warn`, `error` |
1265
- | `fuzzyResolutionThreshold` | number | `0.5` | Name matching threshold for cross-file resolution (0.0–1.0) |
1266
- | `cavemanMode` | string | `off` | Agent communication style: `off`, `lite`, `full`, `ultra` |
1267
- | `shellCompressionLevel` | string | `normal` | Shell command compression level: `off`, `normal`, `aggressive`, `ultra` |
1268
1484
 
1269
1485
  Default exclude patterns: `node_modules/**`, `dist/**`, `build/**`, `.git/**`, `*.min.js`, `.kirograph/**`
1270
1486
 
@@ -1668,6 +1884,7 @@ KiroGraph is inspired by [CodeGraph](https://github.com/colbymchenry/codegraph)
1668
1884
 
1669
1885
  - [cavemem](https://github.com/JuliusBrussee/cavemem) by [Julius Brussee](https://www.linkedin.com/in/julius-brussee/): the memory module's hook-based observation capture, deterministic compression, and SQLite storage pattern.
1670
1886
  - [jDocMunch-MCP](https://github.com/jgravelle/jdocmunch-mcp) by [J. Gravelle](https://www.linkedin.com/in/j-gravelle-2778223/): the documentation module's section-first retrieval approach, stable section IDs, and byte-offset addressing.
1887
+ - [jDataMunch-MCP](https://github.com/jgravelle/jdatamunch-mcp) by [J. Gravelle](https://www.linkedin.com/in/j-gravelle-2778223/): the data module's column profiling, streaming parsers, and server-side aggregation approach.
1671
1888
 
1672
1889
  ### Contributors
1673
1890
 
@@ -53,18 +53,53 @@ function register(program) {
53
53
  depth: parseInt(opts.depth),
54
54
  testPattern: opts.filter
55
55
  });
56
+ const affectedSet = new Set(affected);
57
+ try {
58
+ const { loadConfig } = await Promise.resolve().then(() => require("../../config.js"));
59
+ const config = await loadConfig(target);
60
+ if (config.enableData) {
61
+ const db = cg.getDatabase();
62
+ db.applyDataSchema();
63
+ const rawDb = db.getRawDb();
64
+ const tableExists = rawDb.get("SELECT name FROM sqlite_master WHERE type='table' AND name='data_code_refs'");
65
+ if (tableExists) {
66
+ for (const file of changedFiles) {
67
+ const rel = file.replace(/\\/g, "/").replace(/^\.\//, "");
68
+ const dataset = rawDb.get("SELECT id FROM data_datasets WHERE file_path = ?", [rel]);
69
+ if (dataset) {
70
+ const refs = rawDb.all(
71
+ "SELECT qualified_name FROM data_code_refs WHERE dataset_id = ?",
72
+ [dataset.id]
73
+ );
74
+ for (const ref of refs) {
75
+ const picomatch = require("picomatch");
76
+ const isTest = picomatch(
77
+ opts.filter ?? "{**/*.spec.*,**/*.test.*,**/*_test.*,**/*Test.*,**/*Spec.*,**/*.t.sol,**/*.bats,**/e2e/**,**/test/**,**/tests/**,**/spec/**,**/__tests__/**,**/src/test/**}"
78
+ );
79
+ const node = rawDb.get("SELECT file_path FROM nodes WHERE qualified_name = ?", [ref.qualified_name]);
80
+ if (node && isTest(node.file_path)) {
81
+ affectedSet.add(node.file_path);
82
+ }
83
+ }
84
+ }
85
+ }
86
+ }
87
+ }
88
+ } catch {
89
+ }
90
+ const finalAffected = [...affectedSet].sort();
56
91
  if (opts.json) {
57
- console.log(JSON.stringify({ changedFiles, affectedTests: affected }, null, 2));
92
+ console.log(JSON.stringify({ changedFiles, affectedTests: finalAffected }, null, 2));
58
93
  } else if (opts.quiet) {
59
- for (const f of affected) console.log(f);
94
+ for (const f of finalAffected) console.log(f);
60
95
  } else {
61
- if (affected.length === 0) {
96
+ if (finalAffected.length === 0) {
62
97
  console.log(` ${import_ui.dim}No affected test files found.${import_ui.reset}`);
63
98
  } else {
64
99
  console.log(`
65
- ${(0, import_ui.section)("Affected test files")} ${import_ui.dim}(${affected.length})${import_ui.reset}
100
+ ${(0, import_ui.section)("Affected test files")} ${import_ui.dim}(${finalAffected.length})${import_ui.reset}
66
101
  `);
67
- for (const f of affected) console.log(` ${import_ui.violet}${import_ui.bold}${f}${import_ui.reset}`);
102
+ for (const f of finalAffected) console.log(` ${import_ui.violet}${import_ui.bold}${f}${import_ui.reset}`);
68
103
  console.log();
69
104
  }
70
105
  }
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "version": 3,
3
3
  "sources": ["../../../src/bin/commands/affected.ts"],
4
- "sourcesContent": ["import { Command } from 'commander';\nimport * as path from 'path';\nimport * as fs from 'fs';\nimport { dim, reset, violet, bold, section } from '../ui';\n\nexport function register(program: Command): void {\n program\n .command('affected [files...]')\n .description('Find test files affected by changed source files')\n .option('--stdin', 'Read file list from stdin (one per line)')\n .option('-d, --depth <n>', 'Max dependency traversal depth', '5')\n .option('-f, --filter <glob>', 'Custom glob to identify test files')\n .option('-j, --json', 'Output as JSON')\n .option('-q, --quiet', 'Output file paths only')\n .option('-p, --path <path>', 'Project path')\n .action(async (files: string[], opts: {\n stdin?: boolean; depth: string; filter?: string;\n json?: boolean; quiet?: boolean; path?: string;\n }) => {\n const KiroGraph = (await Promise.resolve().then(() => require('../../index.js'))).default;\n const target = path.resolve(opts.path ?? process.cwd());\n const cg = await KiroGraph.open(target);\n\n let changedFiles = [...files];\n\n if (opts.stdin) {\n const lines = fs.readFileSync('/dev/stdin', 'utf8').split('\\n').map((l: string) => l.trim()).filter(Boolean);\n changedFiles.push(...lines);\n }\n\n if (changedFiles.length === 0) {\n console.error('No files provided. Pass files as arguments or use --stdin.');\n cg.close(); process.exit(1);\n }\n\n const affected = cg.getAffectedTests(changedFiles, {\n depth: parseInt(opts.depth),\n testPattern: opts.filter,\n });\n\n if (opts.json) {\n console.log(JSON.stringify({ changedFiles, affectedTests: affected }, null, 2));\n } else if (opts.quiet) {\n for (const f of affected) console.log(f);\n } else {\n if (affected.length === 0) {\n console.log(` ${dim}No affected test files found.${reset}`);\n } else {\n console.log(`\\n ${section('Affected test files')} ${dim}(${affected.length})${reset}\\n`);\n for (const f of affected) console.log(` ${violet}${bold}${f}${reset}`);\n console.log();\n }\n }\n cg.close();\n });\n}\n"],
5
- "mappings": ";;;;;;;;;;;;;;;;;;;;;;;;;;;;AAAA;AAAA;AAAA;AAAA;AAAA;AACA,WAAsB;AACtB,SAAoB;AACpB,gBAAkD;AAE3C,SAAS,SAAS,SAAwB;AAC/C,UACG,QAAQ,qBAAqB,EAC7B,YAAY,kDAAkD,EAC9D,OAAO,WAAW,0CAA0C,EAC5D,OAAO,mBAAmB,kCAAkC,GAAG,EAC/D,OAAO,uBAAuB,oCAAoC,EAClE,OAAO,cAAc,gBAAgB,EACrC,OAAO,eAAe,wBAAwB,EAC9C,OAAO,qBAAqB,cAAc,EAC1C,OAAO,OAAO,OAAiB,SAG1B;AACJ,UAAM,aAAa,MAAM,QAAQ,QAAQ,EAAE,KAAK,MAAM,QAAQ,gBAAgB,CAAC,GAAG;AAClF,UAAM,SAAS,KAAK,QAAQ,KAAK,QAAQ,QAAQ,IAAI,CAAC;AACtD,UAAM,KAAK,MAAM,UAAU,KAAK,MAAM;AAEtC,QAAI,eAAe,CAAC,GAAG,KAAK;AAE5B,QAAI,KAAK,OAAO;AACd,YAAM,QAAQ,GAAG,aAAa,cAAc,MAAM,EAAE,MAAM,IAAI,EAAE,IAAI,CAAC,MAAc,EAAE,KAAK,CAAC,EAAE,OAAO,OAAO;AAC3G,mBAAa,KAAK,GAAG,KAAK;AAAA,IAC5B;AAEA,QAAI,aAAa,WAAW,GAAG;AAC7B,cAAQ,MAAM,4DAA4D;AAC1E,SAAG,MAAM;AAAG,cAAQ,KAAK,CAAC;AAAA,IAC5B;AAEA,UAAM,WAAW,GAAG,iBAAiB,cAAc;AAAA,MACjD,OAAO,SAAS,KAAK,KAAK;AAAA,MAC1B,aAAa,KAAK;AAAA,IACpB,CAAC;AAED,QAAI,KAAK,MAAM;AACb,cAAQ,IAAI,KAAK,UAAU,EAAE,cAAc,eAAe,SAAS,GAAG,MAAM,CAAC,CAAC;AAAA,IAChF,WAAW,KAAK,OAAO;AACrB,iBAAW,KAAK,SAAU,SAAQ,IAAI,CAAC;AAAA,IACzC,OAAO;AACL,UAAI,SAAS,WAAW,GAAG;AACzB,gBAAQ,IAAI,KAAK,aAAG,gCAAgC,eAAK,EAAE;AAAA,MAC7D,OAAO;AACL,gBAAQ,IAAI;AAAA,QAAO,mBAAQ,qBAAqB,CAAC,KAAK,aAAG,IAAI,SAAS,MAAM,IAAI,eAAK;AAAA,CAAI;AACzF,mBAAW,KAAK,SAAU,SAAQ,IAAI,KAAK,gBAAM,GAAG,cAAI,GAAG,CAAC,GAAG,eAAK,EAAE;AACtE,gBAAQ,IAAI;AAAA,MACd;AAAA,IACF;AACA,OAAG,MAAM;AAAA,EACX,CAAC;AACL;",
4
+ "sourcesContent": ["import { Command } from 'commander';\nimport * as path from 'path';\nimport * as fs from 'fs';\nimport { dim, reset, violet, bold, section } from '../ui';\n\nexport function register(program: Command): void {\n program\n .command('affected [files...]')\n .description('Find test files affected by changed source files')\n .option('--stdin', 'Read file list from stdin (one per line)')\n .option('-d, --depth <n>', 'Max dependency traversal depth', '5')\n .option('-f, --filter <glob>', 'Custom glob to identify test files')\n .option('-j, --json', 'Output as JSON')\n .option('-q, --quiet', 'Output file paths only')\n .option('-p, --path <path>', 'Project path')\n .action(async (files: string[], opts: {\n stdin?: boolean; depth: string; filter?: string;\n json?: boolean; quiet?: boolean; path?: string;\n }) => {\n const KiroGraph = (await Promise.resolve().then(() => require('../../index.js'))).default;\n const target = path.resolve(opts.path ?? process.cwd());\n const cg = await KiroGraph.open(target);\n\n let changedFiles = [...files];\n\n if (opts.stdin) {\n const lines = fs.readFileSync('/dev/stdin', 'utf8').split('\\n').map((l: string) => l.trim()).filter(Boolean);\n changedFiles.push(...lines);\n }\n\n if (changedFiles.length === 0) {\n console.error('No files provided. Pass files as arguments or use --stdin.');\n cg.close(); process.exit(1);\n }\n\n const affected = cg.getAffectedTests(changedFiles, {\n depth: parseInt(opts.depth),\n testPattern: opts.filter,\n });\n\n // Data file awareness: if a changed file is a data file referenced by test files,\n // include those test files in the affected list.\n const affectedSet = new Set(affected);\n try {\n const { loadConfig } = await Promise.resolve().then(() => require('../../config.js'));\n const config = await loadConfig(target);\n if (config.enableData) {\n const db = cg.getDatabase();\n db.applyDataSchema();\n const rawDb = db.getRawDb();\n\n // Check if data_code_refs table exists\n const tableExists = rawDb.get(\"SELECT name FROM sqlite_master WHERE type='table' AND name='data_code_refs'\");\n if (tableExists) {\n for (const file of changedFiles) {\n const rel = file.replace(/\\\\/g, '/').replace(/^\\.\\//, '');\n // Check if this file is a data file\n const dataset = rawDb.get('SELECT id FROM data_datasets WHERE file_path = ?', [rel]);\n if (dataset) {\n // Find code files that reference this dataset\n const refs = rawDb.all(\n 'SELECT qualified_name FROM data_code_refs WHERE dataset_id = ?',\n [dataset.id],\n ) as Array<{ qualified_name: string }>;\n\n for (const ref of refs) {\n // qualified_name might be a file path or a symbol \u2014 check if it's a test file\n const picomatch = require('picomatch');\n const isTest = picomatch(\n opts.filter ?? '{**/*.spec.*,**/*.test.*,**/*_test.*,**/*Test.*,**/*Spec.*,**/*.t.sol,**/*.bats,**/e2e/**,**/test/**,**/tests/**,**/spec/**,**/__tests__/**,**/src/test/**}'\n );\n // Look up the file path for this qualified name from the nodes table\n const node = rawDb.get('SELECT file_path FROM nodes WHERE qualified_name = ?', [ref.qualified_name]);\n if (node && isTest(node.file_path)) {\n affectedSet.add(node.file_path);\n }\n }\n }\n }\n }\n }\n } catch { /* data awareness is non-critical */ }\n\n const finalAffected = [...affectedSet].sort();\n\n if (opts.json) {\n console.log(JSON.stringify({ changedFiles, affectedTests: finalAffected }, null, 2));\n } else if (opts.quiet) {\n for (const f of finalAffected) console.log(f);\n } else {\n if (finalAffected.length === 0) {\n console.log(` ${dim}No affected test files found.${reset}`);\n } else {\n console.log(`\\n ${section('Affected test files')} ${dim}(${finalAffected.length})${reset}\\n`);\n for (const f of finalAffected) console.log(` ${violet}${bold}${f}${reset}`);\n console.log();\n }\n }\n cg.close();\n });\n}\n"],
5
+ "mappings": ";;;;;;;;;;;;;;;;;;;;;;;;;;;;AAAA;AAAA;AAAA;AAAA;AAAA;AACA,WAAsB;AACtB,SAAoB;AACpB,gBAAkD;AAE3C,SAAS,SAAS,SAAwB;AAC/C,UACG,QAAQ,qBAAqB,EAC7B,YAAY,kDAAkD,EAC9D,OAAO,WAAW,0CAA0C,EAC5D,OAAO,mBAAmB,kCAAkC,GAAG,EAC/D,OAAO,uBAAuB,oCAAoC,EAClE,OAAO,cAAc,gBAAgB,EACrC,OAAO,eAAe,wBAAwB,EAC9C,OAAO,qBAAqB,cAAc,EAC1C,OAAO,OAAO,OAAiB,SAG1B;AACJ,UAAM,aAAa,MAAM,QAAQ,QAAQ,EAAE,KAAK,MAAM,QAAQ,gBAAgB,CAAC,GAAG;AAClF,UAAM,SAAS,KAAK,QAAQ,KAAK,QAAQ,QAAQ,IAAI,CAAC;AACtD,UAAM,KAAK,MAAM,UAAU,KAAK,MAAM;AAEtC,QAAI,eAAe,CAAC,GAAG,KAAK;AAE5B,QAAI,KAAK,OAAO;AACd,YAAM,QAAQ,GAAG,aAAa,cAAc,MAAM,EAAE,MAAM,IAAI,EAAE,IAAI,CAAC,MAAc,EAAE,KAAK,CAAC,EAAE,OAAO,OAAO;AAC3G,mBAAa,KAAK,GAAG,KAAK;AAAA,IAC5B;AAEA,QAAI,aAAa,WAAW,GAAG;AAC7B,cAAQ,MAAM,4DAA4D;AAC1E,SAAG,MAAM;AAAG,cAAQ,KAAK,CAAC;AAAA,IAC5B;AAEA,UAAM,WAAW,GAAG,iBAAiB,cAAc;AAAA,MACjD,OAAO,SAAS,KAAK,KAAK;AAAA,MAC1B,aAAa,KAAK;AAAA,IACpB,CAAC;AAID,UAAM,cAAc,IAAI,IAAI,QAAQ;AACpC,QAAI;AACF,YAAM,EAAE,WAAW,IAAI,MAAM,QAAQ,QAAQ,EAAE,KAAK,MAAM,QAAQ,iBAAiB,CAAC;AACpF,YAAM,SAAS,MAAM,WAAW,MAAM;AACtC,UAAI,OAAO,YAAY;AACrB,cAAM,KAAK,GAAG,YAAY;AAC1B,WAAG,gBAAgB;AACnB,cAAM,QAAQ,GAAG,SAAS;AAG1B,cAAM,cAAc,MAAM,IAAI,6EAA6E;AAC3G,YAAI,aAAa;AACf,qBAAW,QAAQ,cAAc;AAC/B,kBAAM,MAAM,KAAK,QAAQ,OAAO,GAAG,EAAE,QAAQ,SAAS,EAAE;AAExD,kBAAM,UAAU,MAAM,IAAI,oDAAoD,CAAC,GAAG,CAAC;AACnF,gBAAI,SAAS;AAEX,oBAAM,OAAO,MAAM;AAAA,gBACjB;AAAA,gBACA,CAAC,QAAQ,EAAE;AAAA,cACb;AAEA,yBAAW,OAAO,MAAM;AAEtB,sBAAM,YAAY,QAAQ,WAAW;AACrC,sBAAM,SAAS;AAAA,kBACb,KAAK,UAAU;AAAA,gBACjB;AAEA,sBAAM,OAAO,MAAM,IAAI,wDAAwD,CAAC,IAAI,cAAc,CAAC;AACnG,oBAAI,QAAQ,OAAO,KAAK,SAAS,GAAG;AAClC,8BAAY,IAAI,KAAK,SAAS;AAAA,gBAChC;AAAA,cACF;AAAA,YACF;AAAA,UACF;AAAA,QACF;AAAA,MACF;AAAA,IACF,QAAQ;AAAA,IAAuC;AAE/C,UAAM,gBAAgB,CAAC,GAAG,WAAW,EAAE,KAAK;AAE5C,QAAI,KAAK,MAAM;AACb,cAAQ,IAAI,KAAK,UAAU,EAAE,cAAc,eAAe,cAAc,GAAG,MAAM,CAAC,CAAC;AAAA,IACrF,WAAW,KAAK,OAAO;AACrB,iBAAW,KAAK,cAAe,SAAQ,IAAI,CAAC;AAAA,IAC9C,OAAO;AACL,UAAI,cAAc,WAAW,GAAG;AAC9B,gBAAQ,IAAI,KAAK,aAAG,gCAAgC,eAAK,EAAE;AAAA,MAC7D,OAAO;AACL,gBAAQ,IAAI;AAAA,QAAO,mBAAQ,qBAAqB,CAAC,KAAK,aAAG,IAAI,cAAc,MAAM,IAAI,eAAK;AAAA,CAAI;AAC9F,mBAAW,KAAK,cAAe,SAAQ,IAAI,KAAK,gBAAM,GAAG,cAAI,GAAG,CAAC,GAAG,eAAK,EAAE;AAC3E,gBAAQ,IAAI;AAAA,MACd;AAAA,IACF;AACA,OAAG,MAAM;AAAA,EACX,CAAC;AACL;",
6
6
  "names": []
7
7
  }