@ninjaxtools/slopdex 0.19.0 → 0.21.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.gitignore +2 -0
- package/README.md +154 -46
- package/binary-install.js +348 -0
- package/binary.js +124 -0
- package/install.js +4 -0
- package/npm-shrinkwrap.json +52 -0
- package/package.json +78 -61
- package/run-slopdex.js +4 -0
- package/dist/chunk-5BO2LLXC.js +0 -15
- package/dist/chunk-5BO2LLXC.js.map +0 -1
- package/dist/chunk-BXWO2KMC.js +0 -20
- package/dist/chunk-BXWO2KMC.js.map +0 -1
- package/dist/chunk-DKRB2XT5.js +0 -33
- package/dist/chunk-DKRB2XT5.js.map +0 -1
- package/dist/chunk-DTI7SXGL.js +0 -109
- package/dist/chunk-DTI7SXGL.js.map +0 -1
- package/dist/chunk-GBEJHETS.js +0 -1117
- package/dist/chunk-GBEJHETS.js.map +0 -1
- package/dist/chunk-HMKVMRLF.js +0 -2221
- package/dist/chunk-HMKVMRLF.js.map +0 -1
- package/dist/chunk-IQU3YVZ3.js +0 -85
- package/dist/chunk-IQU3YVZ3.js.map +0 -1
- package/dist/chunk-JAVPHCZH.js +0 -24
- package/dist/chunk-JAVPHCZH.js.map +0 -1
- package/dist/chunk-MKUTTXZB.js +0 -74
- package/dist/chunk-MKUTTXZB.js.map +0 -1
- package/dist/chunk-OIKBO3NJ.js +0 -133
- package/dist/chunk-OIKBO3NJ.js.map +0 -1
- package/dist/chunk-P57ZJREE.js +0 -426
- package/dist/chunk-P57ZJREE.js.map +0 -1
- package/dist/chunk-S225GYCL.js +0 -79
- package/dist/chunk-S225GYCL.js.map +0 -1
- package/dist/chunk-TSURHRFF.js +0 -240
- package/dist/chunk-TSURHRFF.js.map +0 -1
- package/dist/chunk-VH5VGRCI.js +0 -103
- package/dist/chunk-VH5VGRCI.js.map +0 -1
- package/dist/cli.d.ts +0 -1
- package/dist/cli.js +0 -1130
- package/dist/cli.js.map +0 -1
- package/dist/code-index-MWSJKV3G.js +0 -14
- package/dist/code-index-MWSJKV3G.js.map +0 -1
- package/dist/cross-search-ERBFC25T.js +0 -10
- package/dist/cross-search-ERBFC25T.js.map +0 -1
- package/dist/database-P5XWJWQL.js +0 -14
- package/dist/database-P5XWJWQL.js.map +0 -1
- package/dist/hosted-GHIUHKOU.js +0 -13
- package/dist/hosted-GHIUHKOU.js.map +0 -1
- package/dist/index.d.ts +0 -681
- package/dist/index.js +0 -102
- package/dist/index.js.map +0 -1
- package/dist/jina-43RL7C6M.js +0 -12
- package/dist/jina-43RL7C6M.js.map +0 -1
- package/dist/openai-CYUPNBGB.js +0 -12
- package/dist/openai-CYUPNBGB.js.map +0 -1
- package/dist/openai-EC6THXKV.js +0 -21
- package/dist/openai-EC6THXKV.js.map +0 -1
- package/dist/openai-TCRXM66O.js +0 -11
- package/dist/openai-TCRXM66O.js.map +0 -1
- package/docs/implementation.md +0 -162
- package/docs/reference.md +0 -187
package/dist/index.js
DELETED
|
@@ -1,102 +0,0 @@
|
|
|
1
|
-
import {
|
|
2
|
-
CohereReranker,
|
|
3
|
-
JinaReranker
|
|
4
|
-
} from "./chunk-DTI7SXGL.js";
|
|
5
|
-
import {
|
|
6
|
-
OpenAILLMReranker
|
|
7
|
-
} from "./chunk-OIKBO3NJ.js";
|
|
8
|
-
import {
|
|
9
|
-
CodeIndex
|
|
10
|
-
} from "./chunk-HMKVMRLF.js";
|
|
11
|
-
import {
|
|
12
|
-
readIndexErrors
|
|
13
|
-
} from "./chunk-GBEJHETS.js";
|
|
14
|
-
import {
|
|
15
|
-
OpenAIDescriptionProvider
|
|
16
|
-
} from "./chunk-TSURHRFF.js";
|
|
17
|
-
import {
|
|
18
|
-
analyzeCohesion,
|
|
19
|
-
cohesionLocation,
|
|
20
|
-
crossSearch
|
|
21
|
-
} from "./chunk-P57ZJREE.js";
|
|
22
|
-
import "./chunk-5BO2LLXC.js";
|
|
23
|
-
import {
|
|
24
|
-
JinaEmbeddingProvider
|
|
25
|
-
} from "./chunk-IQU3YVZ3.js";
|
|
26
|
-
import {
|
|
27
|
-
OpenAIEmbeddingProvider
|
|
28
|
-
} from "./chunk-MKUTTXZB.js";
|
|
29
|
-
import "./chunk-JAVPHCZH.js";
|
|
30
|
-
import "./chunk-BXWO2KMC.js";
|
|
31
|
-
import "./chunk-S225GYCL.js";
|
|
32
|
-
import "./chunk-VH5VGRCI.js";
|
|
33
|
-
import {
|
|
34
|
-
CodeIndexError,
|
|
35
|
-
GitDivergenceError,
|
|
36
|
-
GitUnavailableError,
|
|
37
|
-
IncompatibleIndexError
|
|
38
|
-
} from "./chunk-DKRB2XT5.js";
|
|
39
|
-
|
|
40
|
-
// src/index.ts
|
|
41
|
-
function openCodeIndex(options) {
|
|
42
|
-
return new CodeIndex(options);
|
|
43
|
-
}
|
|
44
|
-
function updateFiles(index, options) {
|
|
45
|
-
return index.updateFiles(options);
|
|
46
|
-
}
|
|
47
|
-
function updateFromGit(index, options) {
|
|
48
|
-
return index.updateFromGit(options);
|
|
49
|
-
}
|
|
50
|
-
function updateFromWorkingTree(index, options) {
|
|
51
|
-
return index.updateFromWorkingTree(options);
|
|
52
|
-
}
|
|
53
|
-
function reindexFiles(index, options) {
|
|
54
|
-
return index.reindexFiles(options);
|
|
55
|
-
}
|
|
56
|
-
function similaritySearch(index, options) {
|
|
57
|
-
return index.similaritySearch(options);
|
|
58
|
-
}
|
|
59
|
-
function useDescriptions(index, options) {
|
|
60
|
-
return index.useDescriptions(options);
|
|
61
|
-
}
|
|
62
|
-
function disableDescriptions(index) {
|
|
63
|
-
return index.disableDescriptions();
|
|
64
|
-
}
|
|
65
|
-
function searchDescription(index, options) {
|
|
66
|
-
return index.searchDescription(options);
|
|
67
|
-
}
|
|
68
|
-
function crossSearchFunctions(options) {
|
|
69
|
-
return crossSearch(options);
|
|
70
|
-
}
|
|
71
|
-
function analyzeCodeCohesion(options) {
|
|
72
|
-
return analyzeCohesion(options);
|
|
73
|
-
}
|
|
74
|
-
export {
|
|
75
|
-
CodeIndex,
|
|
76
|
-
CodeIndexError,
|
|
77
|
-
CohereReranker,
|
|
78
|
-
GitDivergenceError,
|
|
79
|
-
GitUnavailableError,
|
|
80
|
-
IncompatibleIndexError,
|
|
81
|
-
JinaEmbeddingProvider,
|
|
82
|
-
JinaReranker,
|
|
83
|
-
OpenAIDescriptionProvider,
|
|
84
|
-
OpenAIEmbeddingProvider,
|
|
85
|
-
OpenAILLMReranker,
|
|
86
|
-
analyzeCodeCohesion,
|
|
87
|
-
analyzeCohesion,
|
|
88
|
-
cohesionLocation,
|
|
89
|
-
crossSearch,
|
|
90
|
-
crossSearchFunctions,
|
|
91
|
-
disableDescriptions,
|
|
92
|
-
openCodeIndex,
|
|
93
|
-
readIndexErrors,
|
|
94
|
-
reindexFiles,
|
|
95
|
-
searchDescription,
|
|
96
|
-
similaritySearch,
|
|
97
|
-
updateFiles,
|
|
98
|
-
updateFromGit,
|
|
99
|
-
updateFromWorkingTree,
|
|
100
|
-
useDescriptions
|
|
101
|
-
};
|
|
102
|
-
//# sourceMappingURL=index.js.map
|
package/dist/index.js.map
DELETED
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"sources":["../src/index.ts"],"sourcesContent":["export { CodeIndex } from \"./code-index.js\";\nexport { readIndexErrors } from \"./storage/database.js\";\nexport { analyzeCohesion, cohesionLocation } from \"./analysis/cohesion.js\";\nexport { crossSearch } from \"./search/cross-search.js\";\nexport { OpenAIEmbeddingProvider } from \"./embeddings/openai.js\";\nexport { OpenAIDescriptionProvider } from \"./descriptions/openai.js\";\nexport type { DescriptionProviderName, OpenAIDescriptionProviderOptions } from \"./descriptions/openai.js\";\nexport type { OpenAIEmbeddingProviderOptions } from \"./embeddings/openai.js\";\nexport { JinaEmbeddingProvider } from \"./embeddings/jina.js\";\nexport type { JinaEmbeddingProviderOptions } from \"./embeddings/jina.js\";\nexport { CohereReranker, JinaReranker } from \"./rerankers/hosted.js\";\nexport type { CohereRerankerOptions, JinaRerankerOptions } from \"./rerankers/hosted.js\";\nexport { OpenAILLMReranker } from \"./rerankers/openai.js\";\nexport type { OpenAILLMRerankerOptions } from \"./rerankers/openai.js\";\nexport { CodeIndexError, GitDivergenceError, GitUnavailableError, IncompatibleIndexError } from \"./errors.js\";\nexport type * from \"./types.js\";\n\nimport { CodeIndex } from \"./code-index.js\";\nimport { analyzeCohesion } from \"./analysis/cohesion.js\";\nimport { crossSearch } from \"./search/cross-search.js\";\nimport type {\n CodeIndexOptions,\n CohesionAnalysisOptions,\n CrossSearchOptions,\n ReindexFilesOptions,\n SimilaritySearchOptions,\n UpdateFilesOptions,\n UpdateFromGitOptions,\n UpdateFromWorkingTreeOptions,\n} from \"./types.js\";\n\nexport function openCodeIndex(options: CodeIndexOptions): CodeIndex {\n return new CodeIndex(options);\n}\n\nexport function updateFiles(index: CodeIndex, options: UpdateFilesOptions) {\n return index.updateFiles(options);\n}\n\nexport function updateFromGit(index: CodeIndex, options?: UpdateFromGitOptions) {\n return index.updateFromGit(options);\n}\n\nexport function updateFromWorkingTree(index: CodeIndex, options?: UpdateFromWorkingTreeOptions) {\n return index.updateFromWorkingTree(options);\n}\n\nexport function reindexFiles(index: CodeIndex, options?: ReindexFilesOptions) {\n return index.reindexFiles(options);\n}\n\nexport function similaritySearch(index: CodeIndex, options: SimilaritySearchOptions) {\n return index.similaritySearch(options);\n}\n\nexport function useDescriptions(index: CodeIndex, options?: { signal?: AbortSignal }) {\n return index.useDescriptions(options);\n}\n\nexport function disableDescriptions(index: CodeIndex) {\n return index.disableDescriptions();\n}\n\nexport function searchDescription(index: CodeIndex, options: SimilaritySearchOptions) {\n return index.searchDescription(options);\n}\n\nexport function crossSearchFunctions(options: CrossSearchOptions) {\n return crossSearch(options);\n}\n\nexport function analyzeCodeCohesion(options: CohesionAnalysisOptions) {\n return analyzeCohesion(options);\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AA+BO,SAAS,cAAc,SAAsC;AAClE,SAAO,IAAI,UAAU,OAAO;AAC9B;AAEO,SAAS,YAAY,OAAkB,SAA6B;AACzE,SAAO,MAAM,YAAY,OAAO;AAClC;AAEO,SAAS,cAAc,OAAkB,SAAgC;AAC9E,SAAO,MAAM,cAAc,OAAO;AACpC;AAEO,SAAS,sBAAsB,OAAkB,SAAwC;AAC9F,SAAO,MAAM,sBAAsB,OAAO;AAC5C;AAEO,SAAS,aAAa,OAAkB,SAA+B;AAC5E,SAAO,MAAM,aAAa,OAAO;AACnC;AAEO,SAAS,iBAAiB,OAAkB,SAAkC;AACnF,SAAO,MAAM,iBAAiB,OAAO;AACvC;AAEO,SAAS,gBAAgB,OAAkB,SAAoC;AACpF,SAAO,MAAM,gBAAgB,OAAO;AACtC;AAEO,SAAS,oBAAoB,OAAkB;AACpD,SAAO,MAAM,oBAAoB;AACnC;AAEO,SAAS,kBAAkB,OAAkB,SAAkC;AACpF,SAAO,MAAM,kBAAkB,OAAO;AACxC;AAEO,SAAS,qBAAqB,SAA6B;AAChE,SAAO,YAAY,OAAO;AAC5B;AAEO,SAAS,oBAAoB,SAAkC;AACpE,SAAO,gBAAgB,OAAO;AAChC;","names":[]}
|
package/dist/jina-43RL7C6M.js
DELETED
|
@@ -1,12 +0,0 @@
|
|
|
1
|
-
import {
|
|
2
|
-
JinaEmbeddingProvider
|
|
3
|
-
} from "./chunk-IQU3YVZ3.js";
|
|
4
|
-
import "./chunk-JAVPHCZH.js";
|
|
5
|
-
import "./chunk-BXWO2KMC.js";
|
|
6
|
-
import "./chunk-S225GYCL.js";
|
|
7
|
-
import "./chunk-VH5VGRCI.js";
|
|
8
|
-
import "./chunk-DKRB2XT5.js";
|
|
9
|
-
export {
|
|
10
|
-
JinaEmbeddingProvider
|
|
11
|
-
};
|
|
12
|
-
//# sourceMappingURL=jina-43RL7C6M.js.map
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"sources":[],"sourcesContent":[],"mappings":"","names":[]}
|
package/dist/openai-CYUPNBGB.js
DELETED
|
@@ -1,12 +0,0 @@
|
|
|
1
|
-
import {
|
|
2
|
-
OpenAIEmbeddingProvider
|
|
3
|
-
} from "./chunk-MKUTTXZB.js";
|
|
4
|
-
import "./chunk-JAVPHCZH.js";
|
|
5
|
-
import "./chunk-BXWO2KMC.js";
|
|
6
|
-
import "./chunk-S225GYCL.js";
|
|
7
|
-
import "./chunk-VH5VGRCI.js";
|
|
8
|
-
import "./chunk-DKRB2XT5.js";
|
|
9
|
-
export {
|
|
10
|
-
OpenAIEmbeddingProvider
|
|
11
|
-
};
|
|
12
|
-
//# sourceMappingURL=openai-CYUPNBGB.js.map
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"sources":[],"sourcesContent":[],"mappings":"","names":[]}
|
package/dist/openai-EC6THXKV.js
DELETED
|
@@ -1,21 +0,0 @@
|
|
|
1
|
-
import {
|
|
2
|
-
DESCRIPTION_PROVIDER_NAMES,
|
|
3
|
-
OpenAIDescriptionProvider,
|
|
4
|
-
descriptionProviderBaseUrl,
|
|
5
|
-
isDescriptionProviderName,
|
|
6
|
-
openCodeAuthKey,
|
|
7
|
-
openCodeAuthPath
|
|
8
|
-
} from "./chunk-TSURHRFF.js";
|
|
9
|
-
import "./chunk-BXWO2KMC.js";
|
|
10
|
-
import "./chunk-S225GYCL.js";
|
|
11
|
-
import "./chunk-VH5VGRCI.js";
|
|
12
|
-
import "./chunk-DKRB2XT5.js";
|
|
13
|
-
export {
|
|
14
|
-
DESCRIPTION_PROVIDER_NAMES,
|
|
15
|
-
OpenAIDescriptionProvider,
|
|
16
|
-
descriptionProviderBaseUrl,
|
|
17
|
-
isDescriptionProviderName,
|
|
18
|
-
openCodeAuthKey,
|
|
19
|
-
openCodeAuthPath
|
|
20
|
-
};
|
|
21
|
-
//# sourceMappingURL=openai-EC6THXKV.js.map
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"sources":[],"sourcesContent":[],"mappings":"","names":[]}
|
package/dist/openai-TCRXM66O.js
DELETED
|
@@ -1,11 +0,0 @@
|
|
|
1
|
-
import {
|
|
2
|
-
OpenAILLMReranker
|
|
3
|
-
} from "./chunk-OIKBO3NJ.js";
|
|
4
|
-
import "./chunk-BXWO2KMC.js";
|
|
5
|
-
import "./chunk-S225GYCL.js";
|
|
6
|
-
import "./chunk-VH5VGRCI.js";
|
|
7
|
-
import "./chunk-DKRB2XT5.js";
|
|
8
|
-
export {
|
|
9
|
-
OpenAILLMReranker
|
|
10
|
-
};
|
|
11
|
-
//# sourceMappingURL=openai-TCRXM66O.js.map
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"sources":[],"sourcesContent":[],"mappings":"","names":[]}
|
package/docs/implementation.md
DELETED
|
@@ -1,162 +0,0 @@
|
|
|
1
|
-
# Implementation and library API
|
|
2
|
-
|
|
3
|
-
For installation, see the [operator README](../README.md). For command examples, configuration, and result interpretation, see the [command reference](reference.md). This document covers the implementation and programmatic interface.
|
|
4
|
-
|
|
5
|
-
## Code map
|
|
6
|
-
|
|
7
|
-
| Module | Responsibility |
|
|
8
|
-
| --- | --- |
|
|
9
|
-
| `src/cli.ts` | Argument parsing, configuration, automatic refresh/recovery, diagnostics, and output selection. |
|
|
10
|
-
| `src/code-index.ts` | Index lifecycle, file preparation, Git/working-tree reconciliation, embedding/description caching, and search facade. |
|
|
11
|
-
| `src/parser/` | Language dispatch, native Tree-sitter extraction, callable identity, and recoverable diagnostics. |
|
|
12
|
-
| `src/source-policy.ts`, `src/gitignore.ts` | Supported paths, built-in/config exclusions, nested ignore rules. |
|
|
13
|
-
| `src/git/repository.ts` | Git commits, trees, blobs, diffs, ancestry, and working-tree changes. |
|
|
14
|
-
| `src/embeddings/`, `src/descriptions/`, `src/rerankers/` | Provider requests and provider profiles. |
|
|
15
|
-
| `src/storage/database.ts` | SQLite schema, durable artifact caches, transactions, metadata, and vector queries. |
|
|
16
|
-
| `src/search/` | Analysis scoring selection and cross-index neighbor discovery. |
|
|
17
|
-
| `src/analysis/cohesion.ts` | Physical distance, gap scores, aggregate affinity, and groups. |
|
|
18
|
-
| `src/format.ts` | Human-readable search and cluster output, plus library cohesion-report formatting. |
|
|
19
|
-
| `src/index.ts`, `src/types.ts` | Public exports and data contracts. |
|
|
20
|
-
|
|
21
|
-
## Callable extraction
|
|
22
|
-
|
|
23
|
-
Language selection uses file extensions. Native Tree-sitter grammars ship as dependencies; extraction does not require a language server, type checker, or project compiler configuration.
|
|
24
|
-
|
|
25
|
-
| Language | Extracted callables |
|
|
26
|
-
| --- | --- |
|
|
27
|
-
| Python | Functions, async functions, class methods, constructors, generators, and bound lambdas; decorators are included in source. |
|
|
28
|
-
| JavaScript / JSX | Functions, generators, methods, constructors, named function expressions/arrows, and components returning JSX. |
|
|
29
|
-
| TypeScript / TSX | Typed functions, methods, constructors, named function expressions/arrows, and generic JSX components. |
|
|
30
|
-
| Rust | Functions, `impl` methods/associated functions, trait default methods, and `let`-bound closures. |
|
|
31
|
-
| Go | Functions, receiver methods, and function literals bound to variables or assignments. |
|
|
32
|
-
| Java | Methods, constructors (including compact record constructors), and variable-bound lambdas. |
|
|
33
|
-
| C | Function definitions, including static/inline functions and functions returning pointers. |
|
|
34
|
-
|
|
35
|
-
Qualified names include enclosing classes, functions, and explicit modules. Go methods include receiver types (`Store[T].Get`); Rust trait implementations include type and trait (`<Store<T> as Read>.read`). Records retain source, signature, line/column locations, source hash, and identity used during incremental reconciliation.
|
|
36
|
-
|
|
37
|
-
Bodyless declarations and anonymous callbacks are omitted. Extraction is syntactic: Rust and C macros are not expanded, C preprocessor branches are indexed as written, and `.h` files use the C grammar. Recoverable callables survive syntax errors; malformed regions produce file or function diagnostics.
|
|
38
|
-
|
|
39
|
-
Source policy combines extension detection, built-in excluded path segments, and Node glob matching for configured includes/excludes. The `ignore` package implements root and nested `.gitignore` semantics, including anchoring, escaping, negation, and excluded-parent behavior. Working-tree updates read current rules; committed-only updates read rules from Git blobs. Rule changes are checked before applying prepared updates.
|
|
40
|
-
|
|
41
|
-
## Index lifecycle and storage
|
|
42
|
-
|
|
43
|
-
The CLI reconciles an index before most commands. Library callers choose when to update explicitly.
|
|
44
|
-
|
|
45
|
-
Git updates resolve the target commit, validate ancestry against the checkpoint, reconcile committed blobs, and optionally overlay current working-tree files when the target equals HEAD. Overlays account for staged, unstaged, untracked, renamed, and deleted paths. The saved checkpoint remains the committed base. Historical/other-branch targets are committed-only. Explicit `updateFiles` operations do not advance the checkpoint.
|
|
46
|
-
|
|
47
|
-
Tree-sitter results and each valid provider result are committed to content-addressed cache tables immediately. Generation checks detect concurrent logical index changes; working-tree updates also verify source and ignore-rule stability. A separate SQLite transaction atomically applies related file, function, diagnostic, description-reference, and checkpoint changes. Failed or aborted initialization retains the valid database and its cache rows so the next invocation resumes without repeating completed work.
|
|
48
|
-
|
|
49
|
-
Storage uses Node's `node:sqlite` and `sqlite-vec`. Writable connections enable WAL, foreign keys, and full synchronization. The schema contains:
|
|
50
|
-
|
|
51
|
-
- `metadata`: repository root, embedding and description profiles, generation, checkpoint, schema version, and feature/scan state.
|
|
52
|
-
- `files`: paths, content hashes, blob IDs, source mode, previous paths, language, and size.
|
|
53
|
-
- `functions`: identity, names, signatures, locations, source, provenance, and embedding/description references.
|
|
54
|
-
- `embeddings`: code, description, and query vectors keyed by embedding profile, operation, and exact input.
|
|
55
|
-
- `function_vectors`: synchronized `vec0` storage for filtered code-vector nearest-neighbor queries.
|
|
56
|
-
- `similarity_cache` / `similarity_cache_state`: persisted per-function top-neighbor lists backing same-index cross-search and cohesion analysis (see below).
|
|
57
|
-
- `description_cache`: generated description text keyed independently by description profile and complete source context.
|
|
58
|
-
- `parse_cache`: successful Tree-sitter extraction results keyed by parser strategy, path, and file-content hash.
|
|
59
|
-
- `callable_provenance`: first-seen committed source identity.
|
|
60
|
-
- `indexing_errors`: diagnostics associated with files.
|
|
61
|
-
|
|
62
|
-
Current schema version is `10`. Schemas 6-9 migrate in place: 6 and 7 gain file-description state and the nearest-neighbor vector table, 8 gains the similarity-cache tables, and 9 gains the similarity-cache floor/completeness columns (backfilled from existing rows); earlier schemas require `--force-reindex`. Metadata validation rejects incompatible roots, embedding profiles, and unsupported schemas. Forced rebuilds clear logical index state while retaining content-addressed caches when possible; incompatible older databases are recreated. Enabled OpenAI description settings are preserved for the same repository where possible. Git divergence reconciliation is a separate operation controlled by `--rebuild-on-divergence`.
|
|
63
|
-
|
|
64
|
-
### Diagnostics
|
|
65
|
-
|
|
66
|
-
Diagnostics cover parse errors, parser exceptions, extraction failures, read failures, and file-size limits. Records include scope, code, message, language, path, recoverable qualified name, location, available source, and Git/working-tree provenance. Healthy callables remain searchable.
|
|
67
|
-
|
|
68
|
-
Diagnostics commit with their corresponding file update. Updates retry failed files even when their Git blobs are unchanged; successful replacement, deletion, and exclusion clear failures. Standalone readers inspect saved diagnostics without constructing an embedding provider. The CLI's exit handler reports remaining failures for source and target indexes; version exits before registering that handler.
|
|
69
|
-
|
|
70
|
-
## Embeddings and descriptions
|
|
71
|
-
|
|
72
|
-
Embedding profiles consist of provider, model, dimensions, and strategy version. Cross-index analysis requires matching profiles. Vectors are validated and normalized before storage/search.
|
|
73
|
-
|
|
74
|
-
- OpenAI defaults to `text-embedding-3-large`, 3072 dimensions, strategy `callable-v2`. Inputs are truncated to 8192 `cl100k_base` tokens.
|
|
75
|
-
- Jina defaults to `jina-embeddings-v4`, 1024 dimensions, strategy `callable-v2:code-query-passage`. Requests distinguish `code.passage` documents from `code.query` queries and enable truncation.
|
|
76
|
-
|
|
77
|
-
Embedding inputs identify language, callable kind, qualified symbol, signature, documentation, and source. For Python, a function's first-statement docstring is included in a separate `documentation` section in addition to remaining part of the callable source.
|
|
78
|
-
|
|
79
|
-
Purpose generation uses the AI SDK and strategy `callable-purpose-v2`. The default is OpenAI's Responses API with `gpt-5.6-luna`; OpenCode Zen and Go are also selectable, both defaulting to `muse-spark-1.3-contributor` unless overridden. OpenCode catalog models are routed through their published protocol using the OpenAI Responses, Anthropic Messages, Google Generative AI, or OpenAI-compatible adapter. Generation opens one conversation per file: stable instructions and complete file context form the prefix, the first request describes the file overall, then callable prompts and generated answers are appended sequentially in source order. This avoids repeating the file within a request and gives provider prompt caches an increasingly large reusable prefix. Responses requests use `store: false`. File and callable text are embedded with the configured embedding provider.
|
|
80
|
-
|
|
81
|
-
Description generation is cached under both a model-specific key (provider, model, strategy, source) and a model-shared key (strategy, source), so file-context, path, or generation-strategy changes invalidate relevant cached results while a description-provider/model change reuses already-assigned or already-generated text for unchanged sources. Description text identity is independent of the embedding profile, allowing an embedding-model change to reuse generation output while producing the required new vector. Every validated description and vector is cached before indexing continues. If generation resumes partway through a file, completed file/callable prompts and cached answers are replayed locally before the next request so the conversation prefix remains equivalent. Ordinary source updates retain the previous file description and its source hash, making staleness explicit without incurring automatic regeneration; `reindex-files` replaces stale file descriptions, optionally continuing through callable regeneration. `useDescriptions` persists the profile and enabled state; `disableDescriptions` turns automatic updates and description search/scoring off without deleting cached artifacts. Function references are attached only by the final logical transaction, so provider failure cannot expose partially updated callable records. Deleting a function removes it from description search but retains reusable cache rows.
|
|
82
|
-
|
|
83
|
-
## Similarity and analysis
|
|
84
|
-
|
|
85
|
-
`search` embeds a query and, when descriptions are complete, scores code, callable-description, and containing-file-description vectors. `search-description` scores callable and file descriptions. Filter-compatible code-only searches up to sqlite-vec's 8,192-dimension limit use the synchronized `vec0` nearest-neighbor table; larger custom profiles plus fused, regex, upper-bound, and multi-path searches retain the exact scalar scoring path. Analysis uses code-only cosine similarity unless all indexed callables and files have enabled description embeddings. Cross-index analysis requires completeness on both sides. Description-generator models may differ across indexes even though embedding profiles must match.
|
|
86
|
-
|
|
87
|
-
An optional `Reranker` performs a second-stage pass for the two natural-language query methods. After applying name and similarity filters, Cohere and Jina retrieve five times the requested result limit (all threshold-passing candidates when no limit is requested). `OpenAILLMReranker` retrieves its configured candidate count (10 by default), or the result limit when larger; without a requested limit it reranks all threshold-passing candidates up to its 100-candidate maximum. Every reranker receives the query plus candidate path, code metadata/source, and purpose description when available, then returns the requested number (or all candidates when unlimited) in relevance order. Results preserve the embedding/fused `similarity` and add `rerankScore`.
|
|
88
|
-
|
|
89
|
-
Cohere defaults to `rerank-v4.0-pro` with `COHERE_API_KEY`; Jina defaults to `jina-reranker-v3.5` with `JINA_API_KEY`. The OpenAI LLM path defaults to `gpt-5.6-luna`, sends a strict JSON schema through the Responses API with high reasoning, no reasoning summary, and `store: false`, and validates result cardinality, indexes, uniqueness, and 0-1 scores. Its prompt preserves descriptions before source and caps source-bearing candidate text at 12,000 tokens each and 80,000 tokens in aggregate. Reranking does not participate in index metadata or cross-search because it creates no persisted artifacts and cross-search is callable-to-callable analysis rather than natural-language retrieval.
|
|
90
|
-
|
|
91
|
-
When descriptions are complete:
|
|
92
|
-
|
|
93
|
-
```text
|
|
94
|
-
similarity = (codeSimilarity + descriptionSimilarity + fileDescriptionSimilarity) / 3
|
|
95
|
-
```
|
|
96
|
-
|
|
97
|
-
All component scores and the average are computed in one SQLite query. Name, line-count, and path exclusions plus similarity bounds apply before ranking/limiting. Range upper bounds are exclusive. Combined JSON includes component scores; analysis metadata records mode, weights, and description profiles. Cohesion JSON uses schema version 3 for the three-component scoring contract.
|
|
98
|
-
|
|
99
|
-
Cross-search selects sources, queries neighbors per source, and deduplicates unordered same-index pairs unless symmetric results are requested. Self-matches are excluded in same-index queries. Same-file exclusion uses canonical roots and file identity to handle aliases. With the `cohesion` option, each selected match receives its physical path distance and matches are re-ranked by descending distance, then similarity. Cluster formatting builds connected components from emitted matches and sorts by member count, then name; transitive connectivity does not imply all-to-all similarity.
|
|
100
|
-
|
|
101
|
-
### Similarity cache
|
|
102
|
-
|
|
103
|
-
Same-index cross-search and cohesion analysis read through a persisted pairwise-similarity cache instead of issuing one vector query per source function on every run. `CodeIndex.refreshSimilarityCache({ width, minSimilarity })` runs first with the caller's effective threshold as the floor (cross-search and cohesion anchor it to a shared 0.3 band unless a lower threshold is given, so sweeping thresholds never rebuilds the cache): it compares each function's current code/description/file-description embedding triple against `similarity_cache_state` and re-queries only new or changed functions (full neighbor scans for the dirty set, then an exact merge-trim pass over clean functions reusing the symmetric dirty scans), storing only pairs at or above the floor. A lower floor than previously cached re-queries every function, since the missing band cannot be rebuilt incrementally; a higher floor reuses the cache for free. The merge never narrows an entry the read-repair widened: each row keeps its lowest served floor and widest stored width, and completeness is evaluated per row, so one incomplete entry cannot mark the whole index incomplete and force every later lookup to re-scan. Each state row records the embedding triple, floor, stored row count, and a completeness flag, so a sparse row set is never mistaken for an unfinished computation, and row loss from deletions is detected by comparing the live row count against the stored count. `cachedSimilarToFunction` then serves each source's top-`width` neighbors with the same threshold/line-count/name-regex/path filters as a live query, falling back to a live vector query on any miss, staleness, higher-floor request, or filter truncation; short results from a complete entry are exact. A fallback scan doubles as a read-repair: when the entry is fresh but unusable, the scan is widened to the full cache width and written back (never narrowing the recorded floor), so later lookups hit and stale flags heal on contact; dense-at-max-width entries are left to live queries. Per-function analysis loops use `cachedSimilarityReader`, which snapshots the validity state once per run and pushes the query filters and limit into the neighbor query, instead of re-reading whole tables and materializing full widths for every source. Cross-index search still queries live vectors; read-only indexes read (but never refresh or repair) the cache. The fill reports a `similarity-cache` progress phase and the source loop a `cross-search` phase (both silent when non-interactive), surfaced through the index-level `onProgress` callback — or the per-call `onCacheProgress` override on cross-search/cohesion options — so the CLI renders the same terminal bar as vector/description generation. Unchanged pairs are never recomputed, deleted functions disappear through foreign-key cascades plus a backfill check, and cache writes never touch the index generation. `cachedSimilarToFunction` then serves each source's top-`width` neighbors with the same threshold/line-count/name-regex/path filters as a live query, falling back to a live vector query on any miss, staleness, narrow entry, or filter truncation. Cross-index search still queries live vectors; read-only indexes read (but never refresh) the cache.
|
|
104
|
-
|
|
105
|
-
### Cohesion metrics
|
|
106
|
-
|
|
107
|
-
Cohesion builds a graph from selected-source top neighbors, retaining unique unordered edges. Reciprocity is true when both endpoints selected each other, false when both were evaluated but only one selected the other, and null when an endpoint was not evaluated.
|
|
108
|
-
|
|
109
|
-
```text
|
|
110
|
-
physicalDistance = 0 # same file
|
|
111
|
-
physicalDistance = 1 + directory-tree hops # different files
|
|
112
|
-
semanticWeight = clamp((similarity - threshold) / (1 - threshold), 0, 1)
|
|
113
|
-
separationWeight = 1 - exp(-physicalDistance / 2)
|
|
114
|
-
cohesionGap = semanticWeight * separationWeight
|
|
115
|
-
```
|
|
116
|
-
|
|
117
|
-
Affinity ratios and mean distance are weighted by `semanticWeight`. Cohesion metrics use all qualifying edges before output limiting. File reports aggregate internal, same-folder, and external affinity for selected-source files. Pairs rank by gap and tie-breakers; groups are connected components of reported pairs. Source/test classification is a path heuristic, not a dependency or call-graph analysis.
|
|
118
|
-
|
|
119
|
-
## Library API
|
|
120
|
-
|
|
121
|
-
```ts
|
|
122
|
-
import { OpenAIEmbeddingProvider, OpenAILLMReranker, openCodeIndex } from "@ninjaxtools/slopdex";
|
|
123
|
-
|
|
124
|
-
const index = openCodeIndex({
|
|
125
|
-
rootDir: "/path/to/repository",
|
|
126
|
-
provider: new OpenAIEmbeddingProvider(),
|
|
127
|
-
reranker: new OpenAILLMReranker({ candidateCount: 10 }),
|
|
128
|
-
});
|
|
129
|
-
|
|
130
|
-
try {
|
|
131
|
-
await index.updateFromGit();
|
|
132
|
-
const results = await index.similaritySearch({
|
|
133
|
-
query: "validate an authenticated session",
|
|
134
|
-
limit: 10,
|
|
135
|
-
});
|
|
136
|
-
console.log(results);
|
|
137
|
-
} finally {
|
|
138
|
-
index.close();
|
|
139
|
-
}
|
|
140
|
-
```
|
|
141
|
-
|
|
142
|
-
Exports include `CodeIndex`, `crossSearch`, `analyzeCohesion`, `cohesionLocation`, embedding/description providers, `CohereReranker`, `JinaReranker`, `OpenAILLMReranker`, error types, and the contracts in `src/types.ts`. Standalone update/search helpers wrap the corresponding index methods.
|
|
143
|
-
|
|
144
|
-
- Use `updateFromWorkingTree()` when Git is unavailable. Unlike the CLI, the library does not automatically refresh before queries or fall back from Git.
|
|
145
|
-
- Set `sourceFilter.nameRegex` for source-only cross-search/cohesion filtering; combine it with `path` and a filter type (`all`, `changed-since`, or `uncommitted`). `changed-since` also accepts `uncommitted: true`.
|
|
146
|
-
- Top-level analysis `nameRegex` filters both sources and candidates. Query `SimilaritySearchOptions.nameRegex` filters result names before limiting.
|
|
147
|
-
- Call `await index.useDescriptions()`, then `await index.searchDescription({ query: "maintain the repository index" })`. Select a model via `descriptionProvider: new OpenAIDescriptionProvider({ model: "gpt-5.6-sol" })` in index options. Custom description providers implement both stateless file/callable methods and may add `startFile()` for contextual sessions.
|
|
148
|
-
- Inspect failures through `index.indexErrors()` or exported `readIndexErrors(indexPath)` without a provider. Records use `IndexingError`.
|
|
149
|
-
- `analyzeCohesion` remains a programmatic report API. The CLI exposes physical-distance review through `cross-search --cohesion` instead of a separate command.
|
|
150
|
-
|
|
151
|
-
## Build and development
|
|
152
|
-
|
|
153
|
-
```bash
|
|
154
|
-
npm install
|
|
155
|
-
npm run check
|
|
156
|
-
```
|
|
157
|
-
|
|
158
|
-
`check` runs TypeScript checking, Vitest, the tsup build, and smoke tests. The smoke script verifies built exports, CLI help/version, language parsing, ignore behavior, and saved diagnostics without network calls. `npm run dev -- <arguments>` runs the source CLI through tsx.
|
|
159
|
-
|
|
160
|
-
The build produces ESM library and CLI files with declarations and source maps in `dist/`. `tsup.config.ts` reads `package.json` and injects `__SLOPDEX_VERSION__`; source-mode version output falls back to reading package metadata. `prepack` runs build and smoke checks.
|
|
161
|
-
|
|
162
|
-
The repository skill lives at `.agents/skills/slopdex/SKILL.md` and is included in the package. `npm run install:skill:opencode` copies it into the OpenCode skill directory.
|
package/docs/reference.md
DELETED
|
@@ -1,187 +0,0 @@
|
|
|
1
|
-
# Command reference
|
|
2
|
-
|
|
3
|
-
## Commands
|
|
4
|
-
|
|
5
|
-
Usage: `slopdex <command> [arguments] [options]`.
|
|
6
|
-
|
|
7
|
-
| Command | Purpose | Output |
|
|
8
|
-
| --- | --- | --- |
|
|
9
|
-
| `models [opencode\|opencode-go]` | Fetch valid models from the current published Zen and/or Go catalogs. | Qualified `provider/model` lines; optional JSON array |
|
|
10
|
-
| `config model <model\|provider/model>` | Validate a published OpenCode model and persist its provider/model selection without opening an index. Bare IDs auto-resolve only when unambiguous. | Updated setting summary; optional JSON |
|
|
11
|
-
| `config descriptions <enable\|disable>` | Persist whether the next index-using command should enable or disable descriptions. Does not open an index. | Updated setting summary; optional JSON |
|
|
12
|
-
| `config reranker <cohere\|jina\|openai\|disable> [model]` | Enable a hosted or OpenAI LLM query reranker, optionally selecting a model, or disable it. OpenAI accepts `--reranker-candidates <number>` from 1 to 100 and defaults to 10. Does not open an index. | Updated setting summary; optional JSON |
|
|
13
|
-
| `search <query>` | Search function code by meaning. Quote multiword queries. | Summary; optional JSON array |
|
|
14
|
-
| `descriptions <enable\|disable>` | Enable or disable automatic purpose descriptions. Re-enabling with unchanged inputs reuses cached descriptions. | JSON statistics |
|
|
15
|
-
| `search-description <query>` | Search purpose descriptions after enabling them. | Summary including description text; optional JSON array |
|
|
16
|
-
| `cross-search` | Find neighbors for each selected function in this or another index. | `clusters` by default; optional `summary` or JSONL |
|
|
17
|
-
| `status` | Refresh and show index metadata, counts, and profiles. | JSON object |
|
|
18
|
-
| `index-errors` | Read saved file/function indexing failures. | Summary; optional JSON array |
|
|
19
|
-
| `update-git` | Explicitly refresh a Git snapshot, with current working-tree changes when targeting HEAD. | JSON update statistics |
|
|
20
|
-
| `update-files <path...>` | After automatic refresh, explicitly reparse selected working-tree files. Paths are repository-relative or absolute within the root. | JSON update statistics |
|
|
21
|
-
| `reindex-files [--callables]` | Regenerate descriptions for files changed since their stored file description. By default stops after each file description; `--callables` also regenerates its callable descriptions. | JSON description statistics |
|
|
22
|
-
| `delete-files <path...>` | After automatic refresh, remove paths from the index; source files are not deleted. A later refresh can restore eligible files. | JSON update statistics |
|
|
23
|
-
|
|
24
|
-
Manual maintenance examples:
|
|
25
|
-
|
|
26
|
-
```bash
|
|
27
|
-
slopdex update-git
|
|
28
|
-
slopdex update-files src/service.ts src/model.ts
|
|
29
|
-
slopdex reindex-files
|
|
30
|
-
slopdex reindex-files --callables
|
|
31
|
-
slopdex delete-files src/removed.ts
|
|
32
|
-
slopdex update-git --target HEAD --rebuild-on-divergence
|
|
33
|
-
slopdex update-git --force-reindex
|
|
34
|
-
```
|
|
35
|
-
|
|
36
|
-
### Location, providers, and diagnostics
|
|
37
|
-
|
|
38
|
-
| Argument | Meaning / default |
|
|
39
|
-
| --- | --- |
|
|
40
|
-
| `--root <path>` | Repository root; current directory by default. |
|
|
41
|
-
| `--config <path>` | Config file; `<root>/.slopdex/config.json` by default. |
|
|
42
|
-
| `--index <path>` | Index file; `<root>/.slopdex/index.sqlite` by default. Overrides `indexPath` in config. |
|
|
43
|
-
| `--provider <openai\|jina>` | Embedding provider; `openai` by default. |
|
|
44
|
-
| `--model <name>` | Embedding model; `text-embedding-3-large` for OpenAI, `jina-embeddings-v4` for Jina. |
|
|
45
|
-
| `--dimensions <number>` | Positive embedding dimension count; OpenAI `3072`, Jina `1024`. Must be supported by the model. |
|
|
46
|
-
| `--description-provider <openai\|opencode\|opencode-go>` | Description provider; OpenAI by default. OpenCode values use Zen or Go with `OPENCODE_API_KEY` or `~/.local/share/opencode/auth.json`. |
|
|
47
|
-
| `--description-model <name>` | Description model; `gpt-5.6-luna` for OpenAI and `muse-spark-1.3-contributor` for Zen/Go, then the persisted model unless overridden. Published OpenCode models use their documented protocol. |
|
|
48
|
-
| `--ignore-errors` | Silence warnings about saved indexing errors; records remain available. |
|
|
49
|
-
| `--verbose` | Write one stderr notice for every external model call instead of one per call kind/provider/model. |
|
|
50
|
-
| `-h`, `--help` | Show CLI usage without refreshing. |
|
|
51
|
-
| `--version` | Print the package version and exit. |
|
|
52
|
-
|
|
53
|
-
Explicit relative config and index paths resolve from the current directory, not `--root`.
|
|
54
|
-
|
|
55
|
-
### Search and analysis
|
|
56
|
-
|
|
57
|
-
| Argument | Applies to | Meaning / default |
|
|
58
|
-
| --- | --- | --- |
|
|
59
|
-
| `--limit <number>` | Both query searches, cross-search | Positive integer output limit; unlimited unless passed. Query matches, or cross-search output entries (clusters for `clusters`, matched sources for `summary`/JSONL). Threshold filters results; scans all sources and only truncates emitted output. |
|
|
60
|
-
| `--matches <number>` | Cross-search | Positive integer matches kept per source function; default `5`. |
|
|
61
|
-
| `--threshold <number\|min-max>` | Both query searches, cross-search | Minimum similarity, or range with inclusive minimum and exclusive maximum. Default `0.3`. |
|
|
62
|
-
| `--format <json\|summary\|clusters>` | Both query searches, cross-search, index-errors | Output format; see the commands table. `clusters` is only for ordinary cross-search. |
|
|
63
|
-
| `-e <regex>`, `--regexp <regex>`, `--regex <regex>` | Both query searches, cross-search | Equivalent case-sensitive JavaScript regex options over qualified names. Query searches filter results before limiting; cross-search filters sources only. |
|
|
64
|
-
| `--min-lines <number>` | Cross-search | Minimum source and candidate callable length; positive integer, default `2`. Use `1` to include one-line wrappers. |
|
|
65
|
-
| `--source-path <path>` | Cross-search | Select sources in a file or recursive directory, relative to the repository root (or absolute within it). |
|
|
66
|
-
| `--changed-since <commit>` | Cross-search | Select added, modified, or moved functions relative to an ancestor of the indexed Git checkpoint, including working-tree changes. Requires Git. |
|
|
67
|
-
| `--uncommitted` | Cross-search | Select functions indexed from working-tree files: staged, unstaged, or untracked changes in Git; all working-tree functions without Git. |
|
|
68
|
-
| `--cross-file-only` | Cross-search | Exclude matches from the same physical file. |
|
|
69
|
-
| `--include-symmetric-duplicates` | Cross-search | Allow both directions of same-index matches; otherwise each unordered pair is emitted once. |
|
|
70
|
-
| `--cohesion` | Cross-search | Add `physicalDistance` and re-rank each source's matches by descending distance, with similarity as the tie-breaker. Defaults to summary output; incompatible with clusters. |
|
|
71
|
-
| `--target-root <path>` | Cross-search | Second repository root; requires `--target-index`. |
|
|
72
|
-
| `--target-index <path>` | Cross-search | Second index file; requires `--target-root`. |
|
|
73
|
-
| `--target-config <path>` | Cross-search | Target config; defaults to `<target-root>/.slopdex/config.json`. Requires both target options. |
|
|
74
|
-
|
|
75
|
-
### Refresh and recovery
|
|
76
|
-
|
|
77
|
-
| Argument | Meaning |
|
|
78
|
-
| --- | --- |
|
|
79
|
-
| `--target <ref>` | Git snapshot for `update-git`; default `HEAD`. Non-HEAD targets exclude working-tree changes. Later commands normally refresh back to HEAD. |
|
|
80
|
-
| `--rebuild-on-divergence` | Allow reconciliation when the saved checkpoint is not an ancestor of the target, such as after a rebase or branch switch. |
|
|
81
|
-
| `--force-reindex` | Recreate an **incompatible** index (repository, provider, model, dimensions, strategy, or schema mismatch). A compatible index still follows normal refresh behavior. |
|
|
82
|
-
| `--no-reindex` | With Git, still reconcile the committed snapshot but skip working-tree overlays. Without Git, reuse a non-empty index; missing/empty indexes are still populated. Not a general offline switch. |
|
|
83
|
-
|
|
84
|
-
## Reading results
|
|
85
|
-
|
|
86
|
-
### Similarity and duplicate clusters
|
|
87
|
-
|
|
88
|
-
Similarity is a model-dependent score, not a probability of duplication. Higher scores mean greater semantic resemblance. Query summaries show `score path :: qualifiedName`; cross-search summaries group those lines beneath each source. Functions without matches are omitted from cross-search output.
|
|
89
|
-
|
|
90
|
-
With reranking enabled, query summaries show `rerankScore rerank (similarity similarity)`. JSON retains `similarity` and adds `rerankScore`. The similarity threshold first filters embedding candidates. With an explicit `--limit`, Cohere and Jina receive up to five times the requested limit; without `--limit` they rerank all threshold-passing candidates. The OpenAI LLM receives the configured number of top embedding results, 10 by default, or the requested result limit when it is larger; without `--limit` it reranks all threshold-passing candidates (up to its 100-candidate maximum, otherwise pass `--limit`). Candidates include function descriptions when available and function metadata/source code. Cross-search and cohesion analysis are not sent to rerankers.
|
|
91
|
-
|
|
92
|
-
```text
|
|
93
|
-
Cluster 1 (3 functions, similarity 0.9124-0.9568)
|
|
94
|
-
src/auth/session.ts:18:1 :: validateSession
|
|
95
|
-
src/http/middleware.ts:42:1 :: authenticate
|
|
96
|
-
src/users/user-service.ts:27:3 :: UserService.authenticate
|
|
97
|
-
```
|
|
98
|
-
|
|
99
|
-
- A cluster groups functions connected by matches. Its range covers observed links; not every pair necessarily matches directly.
|
|
100
|
-
- Clusters sort by member count, then name. Cluster number is not severity.
|
|
101
|
-
- Locations identify where to inspect behavior, callers, and architectural roles. Wrappers, adapters, tests, and separate interface implementations can legitimately resemble one another.
|
|
102
|
-
|
|
103
|
-
### Physical cohesion
|
|
104
|
-
|
|
105
|
-
```text
|
|
106
|
-
src/auth/session.ts :: validateSession
|
|
107
|
-
0.9400 packages/http/middleware.ts :: authenticate [distance 4]
|
|
108
|
-
0.9300 src/auth/token.ts :: validateToken [distance 1]
|
|
109
|
-
```
|
|
110
|
-
|
|
111
|
-
Run cross-search with `--cohesion` to put physically distant matches first. Distance is `0` within one file, `1` between files in one folder, and `1` plus directory-tree hops across folders. The option only changes the order of each source's selected semantic matches; it does not change similarity scores or establish that distant code belongs together.
|
|
112
|
-
|
|
113
|
-
For automation, pass `--format json`: query searches and diagnostics return JSON arrays; cross-search returns **JSONL**, one row per matched source. With `--cohesion`, each match includes `physicalDistance`. Results go to stdout; notices and warnings go to stderr.
|
|
114
|
-
|
|
115
|
-
## System behavior
|
|
116
|
-
|
|
117
|
-
### Descriptions and scoring
|
|
118
|
-
|
|
119
|
-
Descriptions are optional and disabled initially. `descriptions enable` persists the selected provider and model and keeps callable descriptions current on later updates. Use `slopdex descriptions enable --description-provider opencode-go` for OpenCode Go, or combine `--description-provider` and `--description-model` to change both. `slopdex descriptions disable` stops automatic updates and description-based searching/scoring while retaining cached descriptions. `status` exposes callable/file description counts, stale file-description count, enabled state, and profile.
|
|
120
|
-
|
|
121
|
-
Descriptions are generated in source order through one conversation per file. Instructions and complete file source form a stable prefix; Slopdex asks for the overall file description first, then each callable request and answer extends that conversation. This allows supported providers to reuse their prompt cache instead of receiving a separate duplicated file context for every callable.
|
|
122
|
-
|
|
123
|
-
Ordinary updates preserve an existing file description even when its source changes, while still refreshing callable descriptions. `slopdex reindex-files` explicitly regenerates stale file descriptions and their embeddings; add `--callables` to continue through and replace every callable description in those files.
|
|
124
|
-
|
|
125
|
-
Tree-sitter extraction, generated descriptions, and document/query vectors are content-addressed in the same SQLite database. Each validated result is committed immediately, independently of the final logical index update. If indexing is interrupted or a later provider call fails, rerunning reuses every completed result whose profile, operation, input, and source context hash still match.
|
|
126
|
-
|
|
127
|
-
When Slopdex makes external vector, description, or reranking model calls, stderr identifies the call kind, provider, and model. By default each combination is reported once per process regardless of request count. Pass `--verbose`, or set `"verbose": true` in config, to report every request. Cache hits do not produce notices because they do not call a model.
|
|
128
|
-
|
|
129
|
-
With complete descriptions, `search` and cross-search average **one-third code similarity + one-third callable-description similarity + one-third file-description similarity**. `search-description` averages callable and file descriptions without code. Cross-repository analysis needs complete descriptions on both sides; otherwise the entire analysis uses code-only scores. Stale file descriptions remain searchable until explicitly reindexed. Thresholds and limits apply to the selected score.
|
|
130
|
-
|
|
131
|
-
Text output labels combined scores. JSON exposes `codeSimilarity`, `descriptionSimilarity`, `fileDescriptionSimilarity`, and cross-search scoring mode/weights. Compare runs only with matching scoring mode, weights, embedding and description-generator profiles, threshold, and source/candidate filters.
|
|
132
|
-
|
|
133
|
-
### Exclusions
|
|
134
|
-
|
|
135
|
-
Root and nested `.gitignore` rules apply even to tracked files and without Git. Working-tree refreshes use current rules; committed-only snapshots use the target commit's rules. Refresh removes newly excluded files and discovers newly eligible ones. Explicit `update-files` rejects ignored files.
|
|
136
|
-
|
|
137
|
-
Built-in exclusions: `.git`, `.slopdex`, `node_modules`, `dist`, `build`, `coverage`, `vendor`, `generated`, `.venv`, `venv`, `__pycache__`, `.tox`, `.mypy_cache`, `.pytest_cache`, and `target`. Config `include`/`exclude` globs narrow coverage; they cannot override built-in exclusions. Ignore exceptions cannot re-include files beneath an excluded parent directory. Files over 1 MiB are skipped unless `maxFileSize` is raised.
|
|
138
|
-
|
|
139
|
-
### Failures and recovery
|
|
140
|
-
|
|
141
|
-
Parse, extraction, read, and file-size failures are saved while healthy callables remain searchable. Inspect them with `slopdex index-errors --format summary`. JSON includes paths, locations, recoverable names, messages, available source, and snapshot provenance. `status` reports `indexingErrorCount` and `failedFileCount`; `functionCount` counts searchable callables.
|
|
142
|
-
|
|
143
|
-
Saved failures trigger stderr warnings, including on cached runs, help, and cross-search targets. `--ignore-errors` silences warnings without clearing records. Updates retry failed files; successful indexing, deletion, or exclusion clears their diagnostics. Version output bypasses diagnostics.
|
|
144
|
-
|
|
145
|
-
Use the recovery flag named in the error: `--rebuild-on-divergence` for Git history changes, `--force-reindex` for incompatible indexes. Schema versions 6 and 7 migrate in place; earlier schemas require `--force-reindex`, with compatible rebuilds preserving reusable artifact caches. For provider/authentication failures, fix the reported configuration and rerun. For source-change-during-indexing errors, rerun after edits settle. Exit status is `0` on success, `2` for argument/domain errors, and `1` for other failures (or invocation without a command).
|
|
146
|
-
|
|
147
|
-
## Configuration
|
|
148
|
-
|
|
149
|
-
Optional file: `<root>/.slopdex/config.json`. Example using Jina embeddings, OpenAI LLM reranking, and OpenCode Go descriptions (requires `JINA_API_KEY`, `OPENAI_API_KEY` for query searches, plus `OPENCODE_API_KEY` or the `opencode-go` entry in `~/.local/share/opencode/auth.json` when descriptions are enabled):
|
|
150
|
-
|
|
151
|
-
```json
|
|
152
|
-
{
|
|
153
|
-
"provider": "jina",
|
|
154
|
-
"model": "jina-embeddings-v4",
|
|
155
|
-
"dimensions": 1024,
|
|
156
|
-
"rerankingEnabled": true,
|
|
157
|
-
"rerankerProvider": "openai",
|
|
158
|
-
"rerankerModel": "gpt-5.6-luna",
|
|
159
|
-
"rerankerCandidates": 10,
|
|
160
|
-
"descriptionProvider": "opencode-go",
|
|
161
|
-
"verbose": true,
|
|
162
|
-
"exclude": ["**/fixtures/**"]
|
|
163
|
-
}
|
|
164
|
-
```
|
|
165
|
-
|
|
166
|
-
| Property | Purpose / default |
|
|
167
|
-
| --- | --- |
|
|
168
|
-
| `provider`, `model`, `dimensions` | Embedding settings; defaults are listed in the CLI table. |
|
|
169
|
-
| `descriptionProvider` | Description provider: `openai`, `opencode` (Zen), or `opencode-go`; defaults to `openai`. |
|
|
170
|
-
| `descriptionModel` | Description model; provider default unless explicitly set. |
|
|
171
|
-
| `descriptionsEnabled` | When true or false, the next index-using command applies that enabled state during its normal refresh. Unset leaves persisted index state unchanged. |
|
|
172
|
-
| `rerankingEnabled` | Enables second-stage ranking for `search` and `search-description`; disabled/unset by default. Prefer changing it through `config reranker`. |
|
|
173
|
-
| `rerankerProvider` | Reranker: `cohere`, `jina`, or `openai`. OpenAI uses an LLM rather than a dedicated reranking endpoint. |
|
|
174
|
-
| `rerankerModel` | Provider model; defaults to Cohere `rerank-v4.0-pro`, Jina `jina-reranker-v3.5`, or OpenAI `gpt-5.6-luna`. |
|
|
175
|
-
| `rerankerCandidates` | Embedding-ranked candidates sent to the OpenAI LLM; integer from `1` to `100`, default `10`. The requested result limit takes precedence when larger, up to 100. |
|
|
176
|
-
| `indexPath` | Index location; `<root>/.slopdex/index.sqlite`. |
|
|
177
|
-
| `include` | Repository-relative glob array; empty/unset includes all supported eligible files. |
|
|
178
|
-
| `exclude` | Additional repository-relative exclusion globs. |
|
|
179
|
-
| `maxFileSize` | Maximum source-file size in bytes; positive integer, default `1048576`. |
|
|
180
|
-
| `embeddingBatchSize` | Embedding inputs per batch; positive integer, default `32`. |
|
|
181
|
-
| `verbose` | When true, report every external model request on stderr; false/unset reports each call kind/provider/model once per process. |
|
|
182
|
-
|
|
183
|
-
Keep keys in the environment (`OPENAI_API_KEY`, `JINA_API_KEY`, `COHERE_API_KEY`, `OPENCODE_API_KEY`); OpenCode providers also fall back to `~/.local/share/opencode/auth.json`. Reranker settings do not change the stored index and do not require a rebuild. Changing the embedding profile requires rebuilding with `--force-reindex`.
|
|
184
|
-
|
|
185
|
-
## Development
|
|
186
|
-
|
|
187
|
-
- [Implementation and library API](implementation.md)
|