kirograph 0.16.1 → 0.17.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. package/README.md +225 -8
  2. package/dist/bin/commands/affected.js +40 -5
  3. package/dist/bin/commands/affected.js.map +2 -2
  4. package/dist/bin/commands/data.js +665 -0
  5. package/dist/bin/commands/data.js.map +7 -0
  6. package/dist/bin/commands/help.js +5 -0
  7. package/dist/bin/commands/help.js.map +2 -2
  8. package/dist/bin/installer/config-prompt.js +28 -1
  9. package/dist/bin/installer/config-prompt.js.map +2 -2
  10. package/dist/bin/installer/index.js +28 -1
  11. package/dist/bin/installer/index.js.map +2 -2
  12. package/dist/bin/installer/steering.js +33 -0
  13. package/dist/bin/installer/steering.js.map +2 -2
  14. package/dist/bin/installer/targets/index.js.map +1 -1
  15. package/dist/bin/installer/targets/kiro.js +2 -2
  16. package/dist/bin/installer/targets/kiro.js.map +2 -2
  17. package/dist/bin/kirograph.js +3 -1
  18. package/dist/bin/kirograph.js.map +3 -3
  19. package/dist/bin/progress.js +30 -0
  20. package/dist/bin/progress.js.map +2 -2
  21. package/dist/compression/naive-cost.js +25 -0
  22. package/dist/compression/naive-cost.js.map +2 -2
  23. package/dist/compression/tracker.js +11 -2
  24. package/dist/compression/tracker.js.map +2 -2
  25. package/dist/compression/types.js.map +1 -1
  26. package/dist/config.js +37 -1
  27. package/dist/config.js.map +2 -2
  28. package/dist/core/pipeline.js +56 -2
  29. package/dist/core/pipeline.js.map +2 -2
  30. package/dist/data/filters.js +105 -0
  31. package/dist/data/filters.js.map +7 -0
  32. package/dist/data/indexer.js +225 -0
  33. package/dist/data/indexer.js.map +7 -0
  34. package/dist/data/linker.js +150 -0
  35. package/dist/data/linker.js.map +7 -0
  36. package/dist/data/lint.js +109 -0
  37. package/dist/data/lint.js.map +7 -0
  38. package/dist/data/parsers/csv.js +132 -0
  39. package/dist/data/parsers/csv.js.map +7 -0
  40. package/dist/data/parsers/excel.js +88 -0
  41. package/dist/data/parsers/excel.js.map +7 -0
  42. package/dist/data/parsers/index.js +79 -0
  43. package/dist/data/parsers/index.js.map +7 -0
  44. package/dist/data/parsers/json-array.js +89 -0
  45. package/dist/data/parsers/json-array.js.map +7 -0
  46. package/dist/data/parsers/jsonl.js +84 -0
  47. package/dist/data/parsers/jsonl.js.map +7 -0
  48. package/dist/data/parsers/parquet.js +95 -0
  49. package/dist/data/parsers/parquet.js.map +7 -0
  50. package/dist/data/profiler.js +182 -0
  51. package/dist/data/profiler.js.map +7 -0
  52. package/dist/data/queries.js +512 -0
  53. package/dist/data/queries.js.map +7 -0
  54. package/dist/data/types.js +17 -0
  55. package/dist/data/types.js.map +7 -0
  56. package/dist/db/data-schema.sql +61 -0
  57. package/dist/db/database.js +11 -0
  58. package/dist/db/database.js.map +2 -2
  59. package/dist/mcp/tool-names.js +9 -1
  60. package/dist/mcp/tool-names.js.map +2 -2
  61. package/dist/mcp/tools.js +423 -0
  62. package/dist/mcp/tools.js.map +2 -2
  63. package/dist/types.js.map +1 -1
  64. package/package.json +4 -2
package/dist/mcp/tools.js CHANGED
@@ -480,11 +480,119 @@ const tools = [
480
480
  projectPath: { type: "string", description: "Project root path (optional)" }
481
481
  }
482
482
  }
483
+ },
484
+ // ── Data tools (require enableData=true) ────────────────────────────────────
485
+ {
486
+ name: "kirograph_data_list",
487
+ description: "List all indexed datasets with row counts, column counts, and file sizes.",
488
+ inputSchema: { type: "object", properties: { projectPath: { type: "string", description: "Project root path (optional)" } } }
489
+ },
490
+ {
491
+ name: "kirograph_data_describe",
492
+ description: "Full schema profile of a dataset: column names, types, cardinality, null%, sample values. Use to orient on a dataset without reading any rows.",
493
+ inputSchema: {
494
+ type: "object",
495
+ properties: {
496
+ dataset: { type: "string", description: "Dataset ID (from kirograph_data_list)" },
497
+ column: { type: "string", description: "Optional: deep-dive on a single column" },
498
+ projectPath: { type: "string", description: "Project root path (optional)" }
499
+ },
500
+ required: ["dataset"]
501
+ }
502
+ },
503
+ {
504
+ name: "kirograph_data_query",
505
+ description: "Filtered row retrieval with structured operators. Returns only matching rows (max 500). Use instead of reading raw data files.",
506
+ inputSchema: {
507
+ type: "object",
508
+ properties: {
509
+ dataset: { type: "string", description: "Dataset ID" },
510
+ filters: { type: "array", description: "Array of {column, op, value} filters. Ops: eq, neq, gt, gte, lt, lte, contains, in, is_null, between" },
511
+ columns: { type: "array", description: "Column projection (only return these columns)" },
512
+ limit: { type: "number", description: "Max rows (default: 100, hard cap: 500)", default: 100 },
513
+ offset: { type: "number", description: "Pagination offset", default: 0 },
514
+ projectPath: { type: "string", description: "Project root path (optional)" }
515
+ },
516
+ required: ["dataset"]
517
+ }
518
+ },
519
+ {
520
+ name: "kirograph_data_aggregate",
521
+ description: "Server-side GROUP BY aggregation. Computation runs in SQLite \u2014 only the result set enters context. Use for count, sum, avg, min, max questions.",
522
+ inputSchema: {
523
+ type: "object",
524
+ properties: {
525
+ dataset: { type: "string", description: "Dataset ID" },
526
+ groupBy: { type: "array", description: "Columns to group by" },
527
+ metrics: { type: "array", description: "Array of {column, op} metrics. Ops: count, sum, avg, min, max, count_distinct" },
528
+ filters: { type: "array", description: "Optional pre-filters (same format as kirograph_data_query)" },
529
+ projectPath: { type: "string", description: "Project root path (optional)" }
530
+ },
531
+ required: ["dataset", "groupBy", "metrics"]
532
+ }
533
+ },
534
+ {
535
+ name: "kirograph_data_search",
536
+ description: "Search column names and sample values by keyword. Tells you which column holds the answer without loading data.",
537
+ inputSchema: {
538
+ type: "object",
539
+ properties: {
540
+ dataset: { type: "string", description: "Dataset ID" },
541
+ query: { type: "string", description: "Search keyword" },
542
+ projectPath: { type: "string", description: "Project root path (optional)" }
543
+ },
544
+ required: ["dataset", "query"]
545
+ }
546
+ },
547
+ {
548
+ name: "kirograph_data_join",
549
+ description: "SQL JOIN across two indexed datasets. Combines data without loading either file into context.",
550
+ inputSchema: {
551
+ type: "object",
552
+ properties: {
553
+ left: { type: "string", description: "Left dataset ID" },
554
+ right: { type: "string", description: "Right dataset ID" },
555
+ leftColumn: { type: "string", description: "Join column from left dataset" },
556
+ rightColumn: { type: "string", description: "Join column from right dataset" },
557
+ type: { type: "string", description: "Join type: inner (default), left, right", enum: ["inner", "left", "right"], default: "inner" },
558
+ columns: { type: "array", description: "Column projection (prefix with dataset ID)" },
559
+ limit: { type: "number", description: "Max rows (default: 100, hard cap: 500)", default: 100 },
560
+ projectPath: { type: "string", description: "Project root path (optional)" }
561
+ },
562
+ required: ["left", "right", "leftColumn", "rightColumn"]
563
+ }
564
+ },
565
+ {
566
+ name: "kirograph_data_correlations",
567
+ description: "Pairwise Pearson correlations between numeric columns. Discovers hidden relationships without loading data.",
568
+ inputSchema: {
569
+ type: "object",
570
+ properties: {
571
+ dataset: { type: "string", description: "Dataset ID" },
572
+ threshold: { type: "number", description: "Min absolute correlation to include (default: 0.3)", default: 0.3 },
573
+ projectPath: { type: "string", description: "Project root path (optional)" }
574
+ },
575
+ required: ["dataset"]
576
+ }
577
+ },
578
+ {
579
+ name: "kirograph_data_quality",
580
+ description: "Data quality triage: rank columns by risk (null rate, cardinality anomalies, type issues). Identifies problematic columns without loading data.",
581
+ inputSchema: {
582
+ type: "object",
583
+ properties: {
584
+ dataset: { type: "string", description: "Dataset ID" },
585
+ projectPath: { type: "string", description: "Project root path (optional)" }
586
+ },
587
+ required: ["dataset"]
588
+ }
483
589
  }
484
590
  ];
485
591
  class ToolHandler {
486
592
  constructor(cg) {
487
593
  this.connections = /* @__PURE__ */ new Map();
594
+ /** Anti-loop: track recent data_query calls per dataset for pagination detection. */
595
+ this.queryTracker = /* @__PURE__ */ new Map();
488
596
  this.defaultCg = cg;
489
597
  }
490
598
  setDefaultKiroGraph(cg) {
@@ -500,6 +608,34 @@ class ToolHandler {
500
608
  }
501
609
  this.connections.clear();
502
610
  }
611
+ /** Anti-loop: detect pagination patterns and warn the agent. */
612
+ checkPaginationLoop(dataset, offset, response) {
613
+ const now = Date.now();
614
+ const key = dataset;
615
+ const currentOffset = offset ?? 0;
616
+ for (const [k, v] of this.queryTracker) {
617
+ if (now - v.lastCall > 6e4) this.queryTracker.delete(k);
618
+ }
619
+ const entry = this.queryTracker.get(key) ?? { offsets: [], lastCall: 0 };
620
+ entry.offsets.push(currentOffset);
621
+ entry.lastCall = now;
622
+ if (entry.offsets.length > 10) entry.offsets = entry.offsets.slice(-10);
623
+ this.queryTracker.set(key, entry);
624
+ if (entry.offsets.length > 5) {
625
+ const recent = entry.offsets.slice(-6);
626
+ let isIncrementing = true;
627
+ for (let i = 1; i < recent.length; i++) {
628
+ if (recent[i] <= recent[i - 1]) {
629
+ isIncrementing = false;
630
+ break;
631
+ }
632
+ }
633
+ if (isIncrementing) {
634
+ return response + "\n\n\u26A0 Pagination detected. Consider using kirograph_data_aggregate for summary statistics instead of paginating through all rows.";
635
+ }
636
+ }
637
+ return response;
638
+ }
503
639
  async getConnection(projectPath) {
504
640
  if (!projectPath) return this.defaultCg;
505
641
  const resolved = path.resolve(projectPath);
@@ -529,6 +665,8 @@ class ToolHandler {
529
665
  tracker.recordMemorySaving(toolName, outputTokens, naiveCost);
530
666
  } else if (toolName.startsWith("kirograph_docs_")) {
531
667
  tracker.recordDocsSaving(toolName, outputTokens, naiveCost);
668
+ } else if (toolName.startsWith("kirograph_data_")) {
669
+ tracker.recordDataSaving(toolName, outputTokens, naiveCost);
532
670
  } else {
533
671
  tracker.recordGraphSaving(toolName, outputTokens, naiveCost);
534
672
  }
@@ -606,6 +744,9 @@ class ToolHandler {
606
744
  if (stats.bySource.docs.count > 0) {
607
745
  lines.push(` Docs tools: ${stats.bySource.docs.count} calls, ~${stats.bySource.docs.saved.toLocaleString()} tokens saved (vs reading full doc files)`);
608
746
  }
747
+ if (stats.bySource.data.count > 0) {
748
+ lines.push(` Data tools: ${stats.bySource.data.count} calls, ~${stats.bySource.data.saved.toLocaleString()} tokens saved (vs loading raw data files)`);
749
+ }
609
750
  if (stats.bySource.exec.count > 0) {
610
751
  lines.push(` Compression: ${stats.bySource.exec.count} calls, ~${stats.bySource.exec.saved.toLocaleString()} tokens saved (vs raw output)`);
611
752
  }
@@ -732,6 +873,39 @@ class ToolHandler {
732
873
  }
733
874
  } catch {
734
875
  }
876
+ try {
877
+ const projectRoot3 = cg.getProjectRoot();
878
+ const config3 = await (await Promise.resolve().then(() => require("../config.js"))).loadConfig(projectRoot3);
879
+ if (config3.enableData && config3.dataContextLimit > 0) {
880
+ const db3 = cg.getDatabase();
881
+ db3.applyDataSchema();
882
+ const entryFiles = ctx.entryPoints.map((n) => n.filePath).filter(Boolean);
883
+ if (entryFiles.length > 0) {
884
+ const placeholders = entryFiles.map(() => "?").join(", ");
885
+ const dataRefs = db3.getRawDb().all(
886
+ `SELECT DISTINCT d.id, d.file_path, d.row_count, d.column_count
887
+ FROM data_code_refs r JOIN data_datasets d ON r.dataset_id = d.id
888
+ WHERE r.qualified_name IN (${placeholders})`,
889
+ entryFiles
890
+ );
891
+ if (dataRefs.length > 0) {
892
+ const { DataQueries } = await Promise.resolve().then(() => require("../data/queries.js"));
893
+ const dq = new DataQueries(db3.getRawDb());
894
+ const limit = config3.dataContextLimit;
895
+ lines.push("", "## Related Data");
896
+ for (const ref of dataRefs.slice(0, limit)) {
897
+ const info = dq.describeDataset(ref.id);
898
+ if (info) {
899
+ const colSummary = info.columns.map((c) => `${c.name}:${c.inferredType}`).join(", ");
900
+ lines.push(`- **${ref.id}** (${ref.file_path}) \u2014 ${ref.row_count} rows, ${ref.column_count} cols`);
901
+ lines.push(` Schema: ${colSummary}`);
902
+ }
903
+ }
904
+ }
905
+ }
906
+ }
907
+ } catch {
908
+ }
735
909
  return lines.join("\n");
736
910
  }
737
911
  case "kirograph_callers": {
@@ -841,6 +1015,26 @@ class ToolHandler {
841
1015
  }
842
1016
  } catch {
843
1017
  }
1018
+ let dataLine = " Data: disabled";
1019
+ try {
1020
+ const { loadConfig: loadCfg2 } = await Promise.resolve().then(() => require("../config.js"));
1021
+ const cfg2 = await loadCfg2(cg.getProjectRoot());
1022
+ if (cfg2.enableData) {
1023
+ const rawDb2 = cg.getDatabase().getRawDb();
1024
+ cg.getDatabase().applyDataSchema();
1025
+ const datasetCount = rawDb2.get("SELECT COUNT(*) as cnt FROM data_datasets")?.cnt ?? 0;
1026
+ if (datasetCount > 0) {
1027
+ const totalRows = rawDb2.get("SELECT SUM(row_count) as total FROM data_datasets")?.total ?? 0;
1028
+ const totalCols = rawDb2.get("SELECT SUM(column_count) as total FROM data_datasets")?.total ?? 0;
1029
+ const totalSize = rawDb2.get("SELECT SUM(file_size) as total FROM data_datasets")?.total ?? 0;
1030
+ const sizeMb = (totalSize / 1024 / 1024).toFixed(2);
1031
+ dataLine = ` Data: enabled \u2014 ${datasetCount} datasets, ${totalRows.toLocaleString()} rows, ${totalCols} columns (${sizeMb} MB source)`;
1032
+ } else {
1033
+ dataLine = ` Data: enabled (no datasets indexed yet \u2014 run kirograph index)`;
1034
+ }
1035
+ }
1036
+ } catch {
1037
+ }
844
1038
  const threshold = stats.syncWarningThreshold ?? 10;
845
1039
  const pendingFiles = stats.pendingFiles ?? 0;
846
1040
  const syncRunning = stats.syncRunning ?? false;
@@ -870,6 +1064,7 @@ class ToolHandler {
870
1064
  frameworkLine,
871
1065
  archLine,
872
1066
  docsLine,
1067
+ dataLine,
873
1068
  ` DB size: ${dbMb} MB`,
874
1069
  ...semanticLines,
875
1070
  ...syncLines,
@@ -1407,6 +1602,234 @@ class ToolHandler {
1407
1602
  return `[${r.refType}] ${direction} (confidence: ${r.confidence.toFixed(2)})`;
1408
1603
  }).join("\n");
1409
1604
  }
1605
+ // ── Data tools ────────────────────────────────────────────────────────────
1606
+ case "kirograph_data_list": {
1607
+ const { loadConfig } = await Promise.resolve().then(() => require("../config.js"));
1608
+ const projectRoot = cg.getProjectRoot();
1609
+ const config = await loadConfig(projectRoot);
1610
+ if (!config.enableData) return "Data indexing is not enabled. Set enableData: true in .kirograph/config.json and run kirograph index.";
1611
+ const { DataQueries } = await Promise.resolve().then(() => require("../data/queries.js"));
1612
+ const db = cg.getDatabase();
1613
+ db.applyDataSchema();
1614
+ const dq = new DataQueries(db.getRawDb());
1615
+ const datasets = dq.listDatasets();
1616
+ if (datasets.length === 0) return "No datasets indexed. Run kirograph index or kirograph data reindex.";
1617
+ return datasets.map((ds) => {
1618
+ const sizeMb = (ds.fileSize / 1024 / 1024).toFixed(2);
1619
+ return `${ds.id} (${ds.format})
1620
+ File: ${ds.filePath}
1621
+ Rows: ${ds.rowCount.toLocaleString()} | Columns: ${ds.columnCount} | Size: ${sizeMb} MB`;
1622
+ }).join("\n\n");
1623
+ }
1624
+ case "kirograph_data_describe": {
1625
+ const { loadConfig } = await Promise.resolve().then(() => require("../config.js"));
1626
+ const projectRoot = cg.getProjectRoot();
1627
+ const config = await loadConfig(projectRoot);
1628
+ if (!config.enableData) return "Data indexing is not enabled. Set enableData: true in .kirograph/config.json and run kirograph index.";
1629
+ const { DataQueries } = await Promise.resolve().then(() => require("../data/queries.js"));
1630
+ const db = cg.getDatabase();
1631
+ db.applyDataSchema();
1632
+ const dq = new DataQueries(db.getRawDb());
1633
+ if (args.column) {
1634
+ const col = dq.describeColumn(args.dataset, args.column);
1635
+ if (!col) return `Column "${args.column}" not found in dataset "${args.dataset}".`;
1636
+ return [
1637
+ `Column: ${col.name}`,
1638
+ `Type: ${col.inferredType}`,
1639
+ `Nullable: ${col.nullable} (${(col.nullPct * 100).toFixed(1)}% null)`,
1640
+ `Cardinality: ${col.cardinality}`,
1641
+ col.minValue ? `Min: ${col.minValue}` : "",
1642
+ col.maxValue ? `Max: ${col.maxValue}` : "",
1643
+ col.meanValue !== null ? `Mean: ${col.meanValue.toFixed(2)}` : "",
1644
+ `Samples: ${col.sampleValues.join(", ")}`
1645
+ ].filter(Boolean).join("\n");
1646
+ }
1647
+ const result = dq.describeDataset(args.dataset);
1648
+ if (!result) return `Dataset "${args.dataset}" not found. Use kirograph_data_list to see available datasets.`;
1649
+ const lines = [
1650
+ `Dataset: ${result.dataset.id} (${result.dataset.format})`,
1651
+ `File: ${result.dataset.filePath}`,
1652
+ `Rows: ${result.dataset.rowCount.toLocaleString()} | Columns: ${result.dataset.columnCount}`,
1653
+ "",
1654
+ "Columns:"
1655
+ ];
1656
+ for (const col of result.columns) {
1657
+ const nullInfo = col.nullable ? ` (${(col.nullPct * 100).toFixed(0)}% null)` : "";
1658
+ const samples = col.sampleValues.length > 0 ? ` \u2014 samples: ${col.sampleValues.slice(0, 3).join(", ")}` : "";
1659
+ const summary = col.summary ? ` [${col.summary}]` : "";
1660
+ lines.push(` ${col.name}: ${col.inferredType}${nullInfo} [${col.cardinality} distinct]${samples}${summary}`);
1661
+ }
1662
+ const rules = dq.validationRules(args.dataset);
1663
+ if (rules && rules.length > 0) {
1664
+ lines.push("", "Validation rules:");
1665
+ for (const r of rules.slice(0, 10)) {
1666
+ lines.push(` ${r.column}: ${r.rules.join("; ")}`);
1667
+ }
1668
+ }
1669
+ const hints = dq.sampleHints(args.dataset);
1670
+ if (hints && hints.length > 0) {
1671
+ lines.push("", "Sample data hints:");
1672
+ for (const h of hints.slice(0, 10)) {
1673
+ lines.push(` ${h.column}: ${h.hint}`);
1674
+ }
1675
+ }
1676
+ return lines.join("\n");
1677
+ }
1678
+ case "kirograph_data_query": {
1679
+ const { loadConfig } = await Promise.resolve().then(() => require("../config.js"));
1680
+ const projectRoot = cg.getProjectRoot();
1681
+ const config = await loadConfig(projectRoot);
1682
+ if (!config.enableData) return "Data indexing is not enabled. Set enableData: true in .kirograph/config.json and run kirograph index.";
1683
+ const { DataQueries } = await Promise.resolve().then(() => require("../data/queries.js"));
1684
+ const db = cg.getDatabase();
1685
+ db.applyDataSchema();
1686
+ const dq = new DataQueries(db.getRawDb());
1687
+ const result = dq.queryRows(args.dataset, {
1688
+ filters: args.filters,
1689
+ columns: args.columns,
1690
+ limit: args.limit ?? 100,
1691
+ offset: args.offset ?? 0
1692
+ });
1693
+ if (!result) return `Dataset "${args.dataset}" not found.`;
1694
+ if (result.rows.length === 0) return `No rows match the given filters (${result.totalMatching} total in dataset).`;
1695
+ const header = `${result.rows.length} rows returned (${result.totalMatching} total matching):
1696
+ `;
1697
+ const rowStrs = result.rows.slice(0, 50).map((row, i) => {
1698
+ const vals = Object.entries(row).map(([k, v]) => `${k}=${v ?? "null"}`).join(", ");
1699
+ return ` ${i + 1}. ${vals}`;
1700
+ });
1701
+ if (result.rows.length > 50) rowStrs.push(` \u2026and ${result.rows.length - 50} more rows`);
1702
+ let response = header + rowStrs.join("\n");
1703
+ const maxChars = config.dataMaxResponseTokens * 4;
1704
+ if (response.length > maxChars) {
1705
+ response = response.slice(0, maxChars) + "\n\n[truncated: response exceeded token budget]";
1706
+ }
1707
+ response = this.checkPaginationLoop(args.dataset, args.offset, response);
1708
+ return response;
1709
+ }
1710
+ case "kirograph_data_aggregate": {
1711
+ const { loadConfig } = await Promise.resolve().then(() => require("../config.js"));
1712
+ const projectRoot = cg.getProjectRoot();
1713
+ const config = await loadConfig(projectRoot);
1714
+ if (!config.enableData) return "Data indexing is not enabled. Set enableData: true in .kirograph/config.json and run kirograph index.";
1715
+ const { DataQueries } = await Promise.resolve().then(() => require("../data/queries.js"));
1716
+ const db = cg.getDatabase();
1717
+ db.applyDataSchema();
1718
+ const dq = new DataQueries(db.getRawDb());
1719
+ const result = dq.aggregate(args.dataset, {
1720
+ groupBy: args.groupBy ?? [],
1721
+ metrics: args.metrics ?? [],
1722
+ filters: args.filters
1723
+ });
1724
+ if (!result) return `Dataset "${args.dataset}" not found.`;
1725
+ if (result.rows.length === 0) return "No results (empty dataset or all rows filtered out).";
1726
+ const keys = Object.keys(result.rows[0]);
1727
+ const header = keys.join(" | ");
1728
+ const separator = keys.map(() => "---").join(" | ");
1729
+ const rows = result.rows.slice(0, 100).map((row) => keys.map((k) => row[k] ?? "null").join(" | "));
1730
+ let response = `${header}
1731
+ ${separator}
1732
+ ${rows.join("\n")}${result.rows.length > 100 ? `
1733
+ \u2026and ${result.rows.length - 100} more groups` : ""}`;
1734
+ const maxChars = config.dataMaxResponseTokens * 4;
1735
+ if (response.length > maxChars) {
1736
+ response = response.slice(0, maxChars) + "\n\n[truncated: response exceeded token budget]";
1737
+ }
1738
+ return response;
1739
+ }
1740
+ case "kirograph_data_search": {
1741
+ const { loadConfig } = await Promise.resolve().then(() => require("../config.js"));
1742
+ const projectRoot = cg.getProjectRoot();
1743
+ const config = await loadConfig(projectRoot);
1744
+ if (!config.enableData) return "Data indexing is not enabled. Set enableData: true in .kirograph/config.json and run kirograph index.";
1745
+ const { DataQueries } = await Promise.resolve().then(() => require("../data/queries.js"));
1746
+ const db = cg.getDatabase();
1747
+ db.applyDataSchema();
1748
+ const dq = new DataQueries(db.getRawDb());
1749
+ const cols = dq.searchColumns(args.dataset, args.query);
1750
+ if (cols.length === 0) return `No columns matching "${args.query}" in dataset "${args.dataset}".`;
1751
+ return cols.map((c) => {
1752
+ const samples = c.sampleValues.length > 0 ? ` \u2014 samples: ${c.sampleValues.slice(0, 3).join(", ")}` : "";
1753
+ return `${c.name}: ${c.inferredType} [${c.cardinality} distinct]${samples}`;
1754
+ }).join("\n");
1755
+ }
1756
+ case "kirograph_data_join": {
1757
+ const { loadConfig } = await Promise.resolve().then(() => require("../config.js"));
1758
+ const projectRoot = cg.getProjectRoot();
1759
+ const config = await loadConfig(projectRoot);
1760
+ if (!config.enableData) return "Data indexing is not enabled. Set enableData: true in .kirograph/config.json and run kirograph index.";
1761
+ const { DataQueries } = await Promise.resolve().then(() => require("../data/queries.js"));
1762
+ const db = cg.getDatabase();
1763
+ db.applyDataSchema();
1764
+ const dq = new DataQueries(db.getRawDb());
1765
+ try {
1766
+ const result = dq.join({
1767
+ left: args.left,
1768
+ right: args.right,
1769
+ leftColumn: args.leftColumn,
1770
+ rightColumn: args.rightColumn,
1771
+ type: args.type ?? "inner",
1772
+ columns: args.columns,
1773
+ limit: args.limit
1774
+ });
1775
+ if (!result) return `Dataset not found. Verify both dataset IDs with kirograph_data_list.`;
1776
+ const joinTypeStr = String(args.type ?? "inner").toUpperCase();
1777
+ const header = `Join: ${args.left}.${args.leftColumn} ${joinTypeStr} JOIN ${args.right}.${args.rightColumn}
1778
+ Matching rows: ${result.totalMatching} (showing ${result.rows.length})`;
1779
+ if (result.rows.length === 0) return `${header}
1780
+
1781
+ No matching rows.`;
1782
+ const lines = result.rows.map((r) => JSON.stringify(r));
1783
+ let response = `${header}
1784
+
1785
+ ${lines.join("\n")}`;
1786
+ const maxChars = config.dataMaxResponseTokens * 4;
1787
+ if (response.length > maxChars) {
1788
+ response = response.slice(0, maxChars) + "\n\n[truncated: response exceeded token budget]";
1789
+ }
1790
+ return response;
1791
+ } catch (err) {
1792
+ return `Error: ${err instanceof Error ? err.message : String(err)}`;
1793
+ }
1794
+ }
1795
+ case "kirograph_data_correlations": {
1796
+ const { loadConfig } = await Promise.resolve().then(() => require("../config.js"));
1797
+ const projectRoot = cg.getProjectRoot();
1798
+ const config = await loadConfig(projectRoot);
1799
+ if (!config.enableData) return "Data indexing is not enabled. Set enableData: true in .kirograph/config.json and run kirograph index.";
1800
+ const { DataQueries } = await Promise.resolve().then(() => require("../data/queries.js"));
1801
+ const db = cg.getDatabase();
1802
+ db.applyDataSchema();
1803
+ const dq = new DataQueries(db.getRawDb());
1804
+ const pairs = dq.correlations(args.dataset, args.threshold);
1805
+ if (pairs === null) return `Dataset "${args.dataset}" not found.`;
1806
+ if (pairs.length === 0) return `No correlations above threshold ${args.threshold ?? 0.3} found. The dataset may have fewer than 2 numeric columns or no significant correlations.`;
1807
+ const lines = pairs.map(
1808
+ (p) => `${p.column1} \u2194 ${p.column2}: ${p.correlation > 0 ? "+" : ""}${p.correlation.toFixed(4)} (${p.strength})`
1809
+ );
1810
+ return `Correlations for "${args.dataset}" (threshold: ${args.threshold ?? 0.3}):
1811
+
1812
+ ${lines.join("\n")}`;
1813
+ }
1814
+ case "kirograph_data_quality": {
1815
+ const { loadConfig } = await Promise.resolve().then(() => require("../config.js"));
1816
+ const projectRoot = cg.getProjectRoot();
1817
+ const config = await loadConfig(projectRoot);
1818
+ if (!config.enableData) return "Data indexing is not enabled. Set enableData: true in .kirograph/config.json and run kirograph index.";
1819
+ const { DataQueries } = await Promise.resolve().then(() => require("../data/queries.js"));
1820
+ const db = cg.getDatabase();
1821
+ db.applyDataSchema();
1822
+ const dq = new DataQueries(db.getRawDb());
1823
+ const quality = dq.quality(args.dataset);
1824
+ if (quality === null) return `Dataset "${args.dataset}" not found.`;
1825
+ if (quality.length === 0) return `No quality issues detected in "${args.dataset}". All columns look healthy.`;
1826
+ const lines = quality.map(
1827
+ (q) => `${q.column} (risk: ${(q.riskScore * 100).toFixed(0)}%): ${q.issues.join("; ")}`
1828
+ );
1829
+ return `Quality report for "${args.dataset}" (${quality.length} columns with issues):
1830
+
1831
+ ${lines.join("\n")}`;
1832
+ }
1410
1833
  default:
1411
1834
  return `Unknown tool: ${toolName}`;
1412
1835
  }