karajan-code 3.14.0 → 3.15.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. package/README.md +10 -0
  2. package/package.json +1 -1
  3. package/packages/hu-board/src/routes/api.js +7 -3
  4. package/scripts/verify-pack.mjs +5 -1
  5. package/src/agents/claude-agent.js +9 -4
  6. package/src/agents/model-registry.js +23 -0
  7. package/src/config/defaults.js +4 -0
  8. package/src/git/hu-automation.js +6 -5
  9. package/src/hu/acceptance-runner.js +8 -4
  10. package/src/hu/worktree-bootstrap.js +59 -0
  11. package/src/orchestrator/drivers/iteration-phases/guards.js +1 -1
  12. package/src/orchestrator/drivers/iteration-phases/quality-gates.js +1 -1
  13. package/src/orchestrator/drivers/run-hu-batch.js +82 -52
  14. package/src/orchestrator/hu-sub-pipeline.js +26 -2
  15. package/src/orchestrator/stages/coder-stage.js +10 -2
  16. package/src/orchestrator/stages/reviewer-stage.js +3 -3
  17. package/src/plan/plan-executor.js +4 -0
  18. package/src/plan/preflight-hu.js +3 -0
  19. package/src/rag/embedder.js +2 -81
  20. package/src/rag/embedders/_cloud-base.js +2 -24
  21. package/src/rag/embedders/cohere.js +2 -28
  22. package/src/rag/embedders/factory.js +2 -29
  23. package/src/rag/embedders/mistral.js +2 -26
  24. package/src/rag/embedders/onnx.js +2 -63
  25. package/src/rag/embedders/openai.js +2 -22
  26. package/src/rag/embedders/voyage.js +2 -18
  27. package/src/rag/rerank.js +4 -74
  28. package/src/rag/retriever.js +2 -127
  29. package/src/rag/where-parser.js +4 -54
  30. package/src/review/diff-generator.js +2 -2
  31. package/src/roles/agent-role.js +4 -0
  32. package/src/roles/coder-role.js +8 -2
  33. package/src/spec-review/run-spec-review.js +5 -1
  34. package/src/utils/cli-ask-question.js +9 -0
  35. package/src/utils/git.js +12 -11
@@ -68,6 +68,12 @@ export async function runCoderStage({ coderRoleInstance, coderRole, config, logg
68
68
  // the plan, the rest are scoped to this HU. All four pass through
69
69
  // untouched when the caller omits them (legacy single-task runs).
70
70
  adrs, specSection, reviewerFindings, huId,
71
+ // PAR-E2 (KJC-TSK-0629): the stage's config wins over the role's own —
72
+ // worktree lanes pass a laneConfig whose projectDir is the worktree.
73
+ projectDir: config?.projectDir || null,
74
+ // PAR-H (KJC-TSK-0631): lane env (KJ_LANE_SLOT / KJ_PORT_OFFSET)
75
+ // reaches the coder subprocess so services it starts don't collide.
76
+ env: config?.lane_env || null,
71
77
  onOutput: coderStall.onOutput,
72
78
  // Lets Brain Recovery persist a standby snapshot if the coder's
73
79
  // provider hits a quota cap mid-run (KJC hibernation wiring).
@@ -429,8 +435,10 @@ export async function runTddCheckStage({ config, logger, emitter, eventBase, ses
429
435
  logger.setContext({ iteration, stage: "tdd" });
430
436
  let tddDiff, untrackedFiles;
431
437
  try {
432
- tddDiff = await generateDiff({ baseRef: session.session_start_sha });
433
- untrackedFiles = await getUntrackedFiles();
438
+ // PAR-E2 (KJC-TSK-0629): diff where the coder actually worked — a
439
+ // worktree lane's changes are invisible from the main tree.
440
+ tddDiff = await generateDiff({ baseRef: session.session_start_sha, projectDir: config?.projectDir || null });
441
+ untrackedFiles = await getUntrackedFiles(config?.projectDir || null);
434
442
  } catch (err) {
435
443
  logger.warn(`TDD diff generation failed: ${err.message}`);
436
444
  return { action: "continue", stageResult: { ok: false, summary: `TDD check failed: ${err.message}` } };
@@ -194,14 +194,14 @@ async function handleReviewerRejection({ review, repeatDetector, config, logger,
194
194
  });
195
195
  }
196
196
 
197
- export async function fetchReviewDiff(session, logger) {
197
+ export async function fetchReviewDiff(session, logger, projectDir = null) {
198
198
  let diff;
199
199
  if (session.ci_pr_number) {
200
200
  const { getPrDiff } = await import("../../ci/pr-diff.js");
201
201
  diff = await getPrDiff(session.ci_pr_number);
202
202
  logger.info(`Reviewer reading PR diff #${session.ci_pr_number}`);
203
203
  } else {
204
- diff = await generateDiff({ baseRef: session.session_start_sha, stageNewFiles: true });
204
+ diff = await generateDiff({ baseRef: session.session_start_sha, stageNewFiles: true, projectDir });
205
205
  }
206
206
  // Inbound boundary: mask hardcoded secrets before the diff reaches the
207
207
  // (possibly cloud) reviewer model. Runs BEFORE the injection guard below so
@@ -221,7 +221,7 @@ export async function runReviewerStage({ reviewerRole, config, logger, emitter,
221
221
 
222
222
  let diff;
223
223
  try {
224
- diff = await fetchReviewDiff(session, logger);
224
+ diff = await fetchReviewDiff(session, logger, config?.projectDir || null);
225
225
  } catch (err) {
226
226
  logger.warn(`Review diff generation failed: ${err.message}`);
227
227
  return { approved: false, blocking_issues: [{ description: `Diff generation failed: ${err.message}` }], non_blocking_suggestions: [], summary: `Reviewer failed: cannot generate diff — ${err.message}`, confidence: 0 };
@@ -24,6 +24,10 @@ export function planToHuBatch(plan) {
24
24
  task_type: hu.task_type || "sw",
25
25
  status: hu.status === "certified" ? "certified" : hu.status,
26
26
  blocked_by: hu.blocked_by || [],
27
+ // KJC-BUG-0110: scope must survive as its own field — the parallel
28
+ // scheduler treats a scopeless story as exclusive, so dropping it
29
+ // silently degrades --parallel N to sequential.
30
+ scope: hu.scope || null,
27
31
  certified: { text: hu.scope || hu.title },
28
32
  acceptance_criteria: hu.acceptance_criteria || [],
29
33
  acceptance_tests: hu.acceptance_tests || [],
@@ -88,6 +88,9 @@ export async function prependPreflightHu(plan, projectDir) {
88
88
  title: "[PREFLIGHT-000] Verificar entorno de desarrollo listo",
89
89
  description: "[PREFLIGHT-000] Block functional HUs until the project's environment is reproducible: deps installed, build/test/lint passing, git tree clean, cloud auth ready when applicable.",
90
90
  task_type: "infra",
91
+ // KJC-BUG-0108: without an explicit status, `kj plan ready` rejects the
92
+ // whole plan with "invalid status undefined".
93
+ status: "pending",
91
94
  blocked_by: [],
92
95
  reuse: [],
93
96
  acceptance_tests: composePreflightTests(stack, projectDir),
@@ -1,81 +1,2 @@
1
- // KJC-PCS-0049 Step 2 — Ollama embedder adapter for the RAG pipeline.
2
- // Talks to a local Ollama server (POST /api/embeddings) and returns
3
- // Float32Array vectors. Cero deps externas; usa fetch global.
4
- //
5
- // Defaults:
6
- // url = process.env.KJ_OLLAMA_URL or "http://localhost:11434"
7
- // model = process.env.KJ_OLLAMA_EMBED_MODEL or "nomic-embed-text"
8
- // dim = 768 (matches nomic-embed-text)
9
- // timeoutMs = 30000
10
-
11
- const DEFAULT_URL = "http://localhost:11434";
12
- const DEFAULT_MODEL = "nomic-embed-text";
13
- const DEFAULT_DIM = 768;
14
- const DEFAULT_TIMEOUT_MS = 30000;
15
-
16
- export class OllamaEmbedderError extends Error {
17
- constructor(message, { cause, status } = {}) {
18
- super(message);
19
- this.name = "OllamaEmbedderError";
20
- if (cause) this.cause = cause;
21
- if (status != null) this.status = status;
22
- }
23
- }
24
-
25
- export class OllamaEmbedder {
26
- constructor({
27
- url = process.env.KJ_OLLAMA_URL || DEFAULT_URL,
28
- model = process.env.KJ_OLLAMA_EMBED_MODEL || DEFAULT_MODEL,
29
- dim = DEFAULT_DIM,
30
- timeoutMs = DEFAULT_TIMEOUT_MS,
31
- fetchFn = globalThis.fetch,
32
- } = {}) {
33
- this.url = url.replace(/\/$/, "");
34
- this.model = model;
35
- this.dim = dim;
36
- this.timeoutMs = timeoutMs;
37
- this.fetch = fetchFn;
38
- }
39
-
40
- /** Single-text embedding. Returns Float32Array(dim). */
41
- async embed(text) {
42
- if (typeof text !== "string" || text.length === 0) {
43
- throw new OllamaEmbedderError("embed: text must be a non-empty string");
44
- }
45
- const ctrl = new AbortController();
46
- const timer = setTimeout(() => ctrl.abort(), this.timeoutMs);
47
- let res;
48
- try {
49
- res = await this.fetch(`${this.url}/api/embeddings`, {
50
- method: "POST",
51
- headers: { "Content-Type": "application/json" },
52
- body: JSON.stringify({ model: this.model, prompt: text }),
53
- signal: ctrl.signal,
54
- });
55
- } catch (err) {
56
- throw new OllamaEmbedderError(`Ollama embed request failed (${this.url}): ${err.message}`, { cause: err });
57
- } finally {
58
- clearTimeout(timer);
59
- }
60
- if (!res.ok) {
61
- throw new OllamaEmbedderError(`Ollama embed HTTP ${res.status} from ${this.url}`, { status: res.status });
62
- }
63
- const body = await res.json();
64
- const arr = body?.embedding;
65
- if (!Array.isArray(arr) || arr.length === 0) {
66
- throw new OllamaEmbedderError("Ollama response missing 'embedding' array");
67
- }
68
- if (arr.length !== this.dim) {
69
- throw new OllamaEmbedderError(`Ollama dim mismatch: got ${arr.length}, expected ${this.dim} for model ${this.model}`);
70
- }
71
- return Float32Array.from(arr);
72
- }
73
-
74
- /** Batch embedding (sequential — Ollama /api/embeddings is single-text). */
75
- async embedBatch(texts) {
76
- if (!Array.isArray(texts)) throw new OllamaEmbedderError("embedBatch: texts must be an array");
77
- const out = [];
78
- for (const t of texts) out.push(await this.embed(t));
79
- return out;
80
- }
81
- }
1
+ // Shim: embedder now lives in karajan-core/rag (KJC-TSK-0632 PR3).
2
+ export * from "karajan-core/rag/embedder";
@@ -1,24 +1,2 @@
1
- // KJC-TSK-0442 — shared POST helper for cloud embedders (OpenAI, Voyage).
2
- // Same envelope: Bearer auth, JSON body shaped by caller, JSON response
3
- // extracted by caller, dim validated against the adapter's expected size.
4
- export async function cloudEmbed(adapter, text, ErrCls, { provider, body, extract }) {
5
- if (typeof text !== "string" || text.length === 0) throw new ErrCls("embed: text must be a non-empty string");
6
- const ctrl = new AbortController();
7
- const timer = setTimeout(() => ctrl.abort(), adapter.timeoutMs);
8
- let res;
9
- try {
10
- res = await adapter.fetch(adapter.url, {
11
- method: "POST",
12
- headers: { "Content-Type": "application/json", Authorization: `Bearer ${adapter.apiKey}` },
13
- body: JSON.stringify(body),
14
- signal: ctrl.signal,
15
- });
16
- } catch (err) { throw new ErrCls(`${provider} embed request failed: ${err.message}`, { cause: err }); }
17
- finally { clearTimeout(timer); }
18
- if (!res.ok) throw new ErrCls(`${provider} embed HTTP ${res.status}`, { status: res.status });
19
- const json = await res.json();
20
- const arr = extract(json);
21
- if (!Array.isArray(arr) || arr.length === 0) throw new ErrCls(`${provider} response missing embedding`);
22
- if (arr.length !== adapter.dim) throw new ErrCls(`${provider} dim mismatch: got ${arr.length}, expected ${adapter.dim} for model ${adapter.model}`);
23
- return Float32Array.from(arr);
24
- }
1
+ // Shim: moved to karajan-core/rag/embedders (KJC-TSK-0632 PR3).
2
+ export * from "karajan-core/rag/embedders/_cloud-base";
@@ -1,28 +1,2 @@
1
- // KJC-TSK-0446 — Cohere embed-v3 adapter. Cohere returns
2
- // { embeddings: { float: [[...]] } } when `embedding_types: ["float"]`
3
- // is requested, or { embeddings: [[...]] } in legacy v1. We extract the
4
- // modern shape with a v1 fallback.
5
- import { cloudEmbed } from "./_cloud-base.js";
6
-
7
- const DEFAULTS = { url: "https://api.cohere.com/v2/embed", model: "embed-multilingual-v3.0", dim: 1024, timeoutMs: 30000 };
8
-
9
- export class CohereEmbedderError extends Error {
10
- constructor(message, { cause, status } = {}) { super(message); this.name = "CohereEmbedderError"; if (cause) this.cause = cause; if (status != null) this.status = status; }
11
- }
12
-
13
- export class CohereEmbedder {
14
- constructor({ url = DEFAULTS.url, apiKey = process.env.KJ_COHERE_KEY, model = process.env.KJ_COHERE_EMBED_MODEL || DEFAULTS.model, dim = DEFAULTS.dim, timeoutMs = DEFAULTS.timeoutMs, fetchFn = globalThis.fetch } = {}) {
15
- // KJC architecture invariant: scoped env var (KJ_COHERE_KEY), not the
16
- // generic COHERE_API_KEY — same rule as OpenAI/Voyage adapters.
17
- if (!apiKey) throw new CohereEmbedderError("Cohere embedder requires an api_key (config.rag.embedder.api_key or KJ_COHERE_KEY env)");
18
- Object.assign(this, { url, apiKey, model, dim, timeoutMs, fetch: fetchFn });
19
- }
20
- async embed(text) {
21
- return cloudEmbed(this, text, CohereEmbedderError, {
22
- provider: "Cohere",
23
- body: { model: this.model, texts: [text], input_type: "search_document", embedding_types: ["float"] },
24
- extract: (b) => b?.embeddings?.float?.[0] || b?.embeddings?.[0],
25
- });
26
- }
27
- async embedBatch(texts) { if (!Array.isArray(texts)) throw new CohereEmbedderError("embedBatch: texts must be an array"); const out = []; for (const t of texts) out.push(await this.embed(t)); return out; }
28
- }
1
+ // Shim: moved to karajan-core/rag/embedders (KJC-TSK-0632 PR3).
2
+ export * from "karajan-core/rag/embedders/cohere";
@@ -1,29 +1,2 @@
1
- // KJC-TSK-0442 / KJC-TSK-0446 — Embedder factory. Selects provider from
2
- // config.rag.embedder.provider (ollama | openai | voyage | cohere | mistral;
3
- // default ollama). Each provider has its own dim default; the factory wires
4
- // it so the caller does not need to know.
5
- import { OllamaEmbedder } from "../embedder.js";
6
- import { OpenAIEmbedder } from "./openai.js";
7
- import { VoyageEmbedder } from "./voyage.js";
8
- import { CohereEmbedder } from "./cohere.js";
9
- import { MistralEmbedder } from "./mistral.js";
10
- import { ONNXEmbedder } from "./onnx.js";
11
-
12
- const PROVIDERS = {
13
- ollama: { cls: OllamaEmbedder, dim: 768 },
14
- openai: { cls: OpenAIEmbedder, dim: 1536 },
15
- voyage: { cls: VoyageEmbedder, dim: 1024 },
16
- cohere: { cls: CohereEmbedder, dim: 1024 },
17
- mistral: { cls: MistralEmbedder, dim: 1024 },
18
- onnx: { cls: ONNXEmbedder, dim: 384 },
19
- };
20
-
21
- export function makeEmbedder(config = {}) {
22
- const cfg = config?.rag?.embedder || {};
23
- const provider = cfg.provider || "ollama";
24
- const spec = PROVIDERS[provider];
25
- if (!spec) throw new Error(`Unknown embedder provider: ${provider}. Supported: ${Object.keys(PROVIDERS).join(", ")}`);
26
- const dim = cfg.dim || spec.dim;
27
- return new spec.cls({ url: cfg.url, apiKey: cfg.api_key, model: cfg.model, dim, timeoutMs: cfg.timeout_ms });
28
- }
29
-
1
+ // Shim: moved to karajan-core/rag/embedders (KJC-TSK-0632 PR3).
2
+ export * from "karajan-core/rag/embedders/factory";
@@ -1,26 +1,2 @@
1
- // KJC-TSK-0446 — Mistral AI embeddings adapter. EU-hosted, useful for
2
- // users with GDPR constraints that prefer not to send chunks to US
3
- // endpoints. Single model today (`mistral-embed`, 1024 dim).
4
- import { cloudEmbed } from "./_cloud-base.js";
5
-
6
- const DEFAULTS = { url: "https://api.mistral.ai/v1/embeddings", model: "mistral-embed", dim: 1024, timeoutMs: 30000 };
7
-
8
- export class MistralEmbedderError extends Error {
9
- constructor(message, { cause, status } = {}) { super(message); this.name = "MistralEmbedderError"; if (cause) this.cause = cause; if (status != null) this.status = status; }
10
- }
11
-
12
- export class MistralEmbedder {
13
- constructor({ url = DEFAULTS.url, apiKey = process.env.KJ_MISTRAL_KEY, model = process.env.KJ_MISTRAL_EMBED_MODEL || DEFAULTS.model, dim = DEFAULTS.dim, timeoutMs = DEFAULTS.timeoutMs, fetchFn = globalThis.fetch } = {}) {
14
- // KJC architecture invariant: scoped env var (KJ_MISTRAL_KEY).
15
- if (!apiKey) throw new MistralEmbedderError("Mistral embedder requires an api_key (config.rag.embedder.api_key or KJ_MISTRAL_KEY env)");
16
- Object.assign(this, { url, apiKey, model, dim, timeoutMs, fetch: fetchFn });
17
- }
18
- async embed(text) {
19
- return cloudEmbed(this, text, MistralEmbedderError, {
20
- provider: "Mistral",
21
- body: { model: this.model, input: [text] },
22
- extract: (b) => b?.data?.[0]?.embedding,
23
- });
24
- }
25
- async embedBatch(texts) { if (!Array.isArray(texts)) throw new MistralEmbedderError("embedBatch: texts must be an array"); const out = []; for (const t of texts) out.push(await this.embed(t)); return out; }
26
- }
1
+ // Shim: moved to karajan-core/rag/embedders (KJC-TSK-0632 PR3).
2
+ export * from "karajan-core/rag/embedders/mistral";
@@ -1,63 +1,2 @@
1
- // KJC-TSK-0447 — ONNX local embedder via @huggingface/transformers
2
- // (formerly @xenova/transformers). Runs sentence-transformer models
3
- // directly in Node — no API key, no Docker, no Ollama.
4
- //
5
- // First call downloads the model weights to ~/.cache/huggingface/ (~80 MB
6
- // for the default MiniLM). Subsequent calls reuse the cache, so the steady
7
- // state is fully offline.
8
- //
9
- // Default model: `Xenova/all-MiniLM-L6-v2` (384 dim, ~80 MB, fast).
10
- // Higher-quality alternative: `Xenova/jina-embeddings-v2-base-en`
11
- // (768 dim, ~260 MB, slower but better for retrieval).
12
-
13
- const DEFAULTS = { model: "Xenova/all-MiniLM-L6-v2", dim: 384, pooling: "mean", normalize: true };
14
-
15
- export class ONNXEmbedderError extends Error {
16
- constructor(message, { cause } = {}) { super(message); this.name = "ONNXEmbedderError"; if (cause) this.cause = cause; }
17
- }
18
-
19
- async function loadTransformers() {
20
- // Prefer the modern @huggingface/transformers (post-2024 rebrand);
21
- // fall back to @xenova/transformers for users still on the legacy package.
22
- // Both are optional peer deps — Karajan does not ship them by default to
23
- // keep the install size small (~500 MB with WASM + weights). The user
24
- // installs whichever they prefer when they opt into provider: onnx.
25
- /* eslint-disable import-x/no-unresolved */
26
- try { return await import("@huggingface/transformers"); }
27
- catch (e1) {
28
- try { return await import("@xenova/transformers"); }
29
- catch (e2) {
30
- throw new ONNXEmbedderError(
31
- "ONNX embedder requires @huggingface/transformers (or legacy @xenova/transformers). Install with `npm install @huggingface/transformers`.",
32
- { cause: e2 },
33
- );
34
- }
35
- }
36
- /* eslint-enable import-x/no-unresolved */
37
- }
38
-
39
- export class ONNXEmbedder {
40
- constructor({ model = process.env.KJ_ONNX_EMBED_MODEL || DEFAULTS.model, dim = DEFAULTS.dim, pooling = DEFAULTS.pooling, normalize = DEFAULTS.normalize } = {}) {
41
- Object.assign(this, { model, dim, pooling, normalize, _pipe: null });
42
- }
43
- async _ensurePipeline() {
44
- if (this._pipe) return this._pipe;
45
- const transformers = await loadTransformers();
46
- this._pipe = await transformers.pipeline("feature-extraction", this.model);
47
- return this._pipe;
48
- }
49
- async embed(text) {
50
- if (typeof text !== "string" || text.length === 0) throw new ONNXEmbedderError("embed: text must be a non-empty string");
51
- const pipe = await this._ensurePipeline();
52
- const out = await pipe(text, { pooling: this.pooling, normalize: this.normalize });
53
- const arr = Array.from(out.data || []);
54
- if (arr.length !== this.dim) throw new ONNXEmbedderError(`ONNX dim mismatch: got ${arr.length}, expected ${this.dim} for model ${this.model}`);
55
- return Float32Array.from(arr);
56
- }
57
- async embedBatch(texts) {
58
- if (!Array.isArray(texts)) throw new ONNXEmbedderError("embedBatch: texts must be an array");
59
- const out = [];
60
- for (const t of texts) out.push(await this.embed(t));
61
- return out;
62
- }
63
- }
1
+ // Shim: moved to karajan-core/rag/embedders (KJC-TSK-0632 PR3).
2
+ export * from "karajan-core/rag/embedders/onnx";
@@ -1,22 +1,2 @@
1
- // KJC-TSK-0442 — OpenAI text-embeddings adapter.
2
- import { cloudEmbed } from "./_cloud-base.js";
3
-
4
- const DEFAULTS = { url: "https://api.openai.com/v1/embeddings", model: "text-embedding-3-small", dim: 1536, timeoutMs: 30000 };
5
-
6
- export class OpenAIEmbedderError extends Error {
7
- constructor(message, { cause, status } = {}) { super(message); this.name = "OpenAIEmbedderError"; if (cause) this.cause = cause; if (status != null) this.status = status; }
8
- }
9
-
10
- export class OpenAIEmbedder {
11
- constructor({ url = DEFAULTS.url, apiKey = process.env.KJ_OPENAI_KEY, model = process.env.KJ_OPENAI_EMBED_MODEL || DEFAULTS.model, dim = DEFAULTS.dim, timeoutMs = DEFAULTS.timeoutMs, fetchFn = globalThis.fetch } = {}) {
12
- // KJC architecture invariant: Karajan does not read provider API keys
13
- // (ANTHROPIC_API_KEY, OPENAI_API_KEY, etc.) — those belong to the CLI
14
- // agents Karajan spawns. RAG embedders are the one exception: they
15
- // call the OpenAI endpoint directly, but use a Karajan-scoped env var
16
- // (KJ_OPENAI_KEY) so the invariant stays clean.
17
- if (!apiKey) throw new OpenAIEmbedderError("OpenAI embedder requires an api_key (config.rag.embedder.api_key or KJ_OPENAI_KEY env)");
18
- Object.assign(this, { url, apiKey, model, dim, timeoutMs, fetch: fetchFn });
19
- }
20
- async embed(text) { return cloudEmbed(this, text, OpenAIEmbedderError, { provider: "OpenAI", body: { model: this.model, input: text }, extract: (b) => b?.data?.[0]?.embedding }); }
21
- async embedBatch(texts) { if (!Array.isArray(texts)) throw new OpenAIEmbedderError("embedBatch: texts must be an array"); const out = []; for (const t of texts) out.push(await this.embed(t)); return out; }
22
- }
1
+ // Shim: moved to karajan-core/rag/embedders (KJC-TSK-0632 PR3).
2
+ export * from "karajan-core/rag/embedders/openai";
@@ -1,18 +1,2 @@
1
- // KJC-TSK-0442 — Voyage AI embeddings adapter.
2
- import { cloudEmbed } from "./_cloud-base.js";
3
-
4
- const DEFAULTS = { url: "https://api.voyageai.com/v1/embeddings", model: "voyage-code-3", dim: 1024, timeoutMs: 30000 };
5
-
6
- export class VoyageEmbedderError extends Error {
7
- constructor(message, { cause, status } = {}) { super(message); this.name = "VoyageEmbedderError"; if (cause) this.cause = cause; if (status != null) this.status = status; }
8
- }
9
-
10
- export class VoyageEmbedder {
11
- constructor({ url = DEFAULTS.url, apiKey = process.env.KJ_VOYAGE_KEY, model = process.env.KJ_VOYAGE_EMBED_MODEL || DEFAULTS.model, dim = DEFAULTS.dim, timeoutMs = DEFAULTS.timeoutMs, fetchFn = globalThis.fetch } = {}) {
12
- // Same invariant as OpenAIEmbedder: Karajan-scoped env var.
13
- if (!apiKey) throw new VoyageEmbedderError("Voyage embedder requires an api_key (config.rag.embedder.api_key or KJ_VOYAGE_KEY env)");
14
- Object.assign(this, { url, apiKey, model, dim, timeoutMs, fetch: fetchFn });
15
- }
16
- async embed(text) { return cloudEmbed(this, text, VoyageEmbedderError, { provider: "Voyage", body: { model: this.model, input: [text], input_type: "document" }, extract: (b) => b?.data?.[0]?.embedding }); }
17
- async embedBatch(texts) { if (!Array.isArray(texts)) throw new VoyageEmbedderError("embedBatch: texts must be an array"); const out = []; for (const t of texts) out.push(await this.embed(t)); return out; }
18
- }
1
+ // Shim: moved to karajan-core/rag/embedders (KJC-TSK-0632 PR3).
2
+ export * from "karajan-core/rag/embedders/voyage";
package/src/rag/rerank.js CHANGED
@@ -1,74 +1,4 @@
1
- // KJC-TSK-0449 — Cross-encoder rerank stage. Re-scores the top-N hits
2
- // from the hybrid retriever using a (query, passage) cross-encoder.
3
- //
4
- // Cross-encoders are slower than the bi-encoder embedders (they jointly
5
- // encode query + passage instead of caching the passage), but they are
6
- // substantially more precise for the final top-K ranking. Karajan calls
7
- // the reranker only on the post-fusion candidates (≤topK*2), so the
8
- // added latency is bounded.
9
- //
10
- // Model is loaded dynamically via @huggingface/transformers (same dep as
11
- // the ONNX embedder, KJC-TSK-0447). Default: `Xenova/ms-marco-MiniLM-L-6-v2`,
12
- // the de-facto standard sentence-transformers reranker.
13
-
14
- const DEFAULT_MODEL = "Xenova/ms-marco-MiniLM-L-6-v2";
15
-
16
- export class RerankError extends Error {
17
- constructor(message, { cause } = {}) { super(message); this.name = "RerankError"; if (cause) this.cause = cause; }
18
- }
19
-
20
- async function loadTransformers() {
21
- /* eslint-disable import-x/no-unresolved */
22
- try { return await import("@huggingface/transformers"); }
23
- catch (e1) {
24
- try { return await import("@xenova/transformers"); }
25
- catch (e2) {
26
- throw new RerankError(
27
- "rerank requires @huggingface/transformers (or legacy @xenova/transformers). Install with `npm install @huggingface/transformers`.",
28
- { cause: e2 },
29
- );
30
- }
31
- }
32
- /* eslint-enable import-x/no-unresolved */
33
- }
34
-
35
- /**
36
- * Lazy singleton — first call downloads weights (~80 MB), every subsequent
37
- * call reuses the same pipeline instance.
38
- */
39
- let _pipelinePromise = null;
40
- function getPipeline(modelName) {
41
- if (!_pipelinePromise) _pipelinePromise = (async () => {
42
- const t = await loadTransformers();
43
- // text-classification pipeline returns the cross-encoder relevance score
44
- // for a `[query, passage]` text pair when applied to a CE model.
45
- return t.pipeline("text-classification", modelName);
46
- })();
47
- return _pipelinePromise;
48
- }
49
-
50
- /** Reset for tests. */
51
- export function _resetPipeline() { _pipelinePromise = null; }
52
-
53
- /**
54
- * Rerank `hits` against `queryText` using the cross-encoder. Returns a NEW
55
- * array sorted by descending rerank score (best first), `score` mutated
56
- * so the existing downstream code (kindBoost, slicing) keeps working.
57
- *
58
- * Each input hit is treated as a [query, hit.text] pair. Truncates the
59
- * passage to `maxChars` to avoid blowing past the CE model's context
60
- * window — most CE models handle 512 tokens, ~2000 chars is safe.
61
- *
62
- * If `pipelineFn` is injected (tests), it bypasses model loading.
63
- */
64
- export async function rerank(queryText, hits, { model = process.env.KJ_RERANK_MODEL || DEFAULT_MODEL, maxChars = 2000, pipelineFn = null } = {}) {
65
- if (!queryText || typeof queryText !== "string") throw new RerankError("rerank: queryText must be a non-empty string");
66
- if (!Array.isArray(hits)) throw new RerankError("rerank: hits must be an array");
67
- if (hits.length === 0) return hits;
68
- const pipe = pipelineFn || await getPipeline(model);
69
- const pairs = hits.map((h) => ({ text: queryText, text_pair: String(h.text || "").slice(0, maxChars) }));
70
- const results = await pipe(pairs);
71
- return hits
72
- .map((h, i) => ({ ...h, _rerank: Number(results[i]?.score) || 0, score: -Number(results[i]?.score) || 0 }))
73
- .sort((a, b) => a.score - b.score);
74
- }
1
+ // Shim: rerank now lives in karajan-core/rag so the hu-board workspace
2
+ // can consume it without a relative dep on the CLI src tree.
3
+ // KJC-TSK-0632 PR2.
4
+ export { RerankError, _resetPipeline, rerank } from "karajan-core/rag/rerank";
@@ -1,127 +1,2 @@
1
- // KJC-PCS-0049 Step 5 — Retriever for the RAG pipeline.
2
- import { searchSimilar, searchBM25, getEmbeddingsByIds } from "./vec-store.js";
3
- import { parseWhere, buildWhereSql } from "./where-parser.js";
4
- import { rerank } from "./rerank.js";
5
-
6
- const DEFAULT_KIND_BOOST = { plan: 0.05, onboarding: 0.03, code: 0 };
7
-
8
- // KJC-TSK-0440 — asymmetric source/test boost.
9
- const SOURCE_BOOST_NON_TEST = 0.05;
10
- const TEST_TERMS_RE = /\b(test|tests|spec|specs|expect|describe|it\(|jest|vitest|mocha)\b/i;
11
- const TEST_PATH_RE = /[\\/](tests?|specs?|__tests__)[\\/]|\.test\.[jt]sx?$|\.spec\.[jt]sx?$/i;
12
-
13
- function shouldBoostSources(queryText) { return !TEST_TERMS_RE.test(queryText); }
14
- function isTestPath(source) { return TEST_PATH_RE.test(source || ""); }
15
-
16
- // KJC-TSK-0443 — fuse semantic + keyword hits into a unified candidate set.
17
- // For 'semantic'/'keyword' modes we surface that side. For 'hybrid' (default)
18
- // we min-max normalise both scores to [0,1] (lower=better) and linear-combine
19
- // via alpha * semantic + (1-alpha) * keyword. Result written back to
20
- // `distance` so the kind+source boost pipeline keeps working unchanged.
21
- function fuseHits(semantic, keyword, alpha, mode) {
22
- if (mode === "semantic") return semantic;
23
- if (mode === "keyword") return keyword.map((h) => ({ ...h, distance: h.bm25 }));
24
- const byId = new Map();
25
- for (const h of semantic) byId.set(h.id, { ...h, _sem: h.distance });
26
- for (const h of keyword) {
27
- const prev = byId.get(h.id);
28
- if (prev) prev._kw = h.bm25;
29
- else byId.set(h.id, { ...h, _kw: h.bm25, distance: h.bm25 });
30
- }
31
- const list = [...byId.values()];
32
- const norm = (vals) => {
33
- const xs = vals.filter((v) => Number.isFinite(v));
34
- if (xs.length === 0) return () => 0.5;
35
- const min = Math.min(...xs); const max = Math.max(...xs); const span = max - min || 1;
36
- return (v) => Number.isFinite(v) ? (v - min) / span : 1;
37
- };
38
- const nSem = norm(list.map((h) => h._sem));
39
- const nKw = norm(list.map((h) => h._kw));
40
- for (const h of list) h.distance = alpha * nSem(h._sem) + (1 - alpha) * nKw(h._kw);
41
- return list;
42
- }
43
-
44
- // KJC-TSK-0484 PR-B — Maximal Marginal Relevance. Given an ordered list of
45
- // candidates (best-first by `score`), pick `topK` that balance relevance
46
- // against intra-result diversity. `lambda=1` collapses to plain relevance;
47
- // `lambda=0` maximises diversity. Cosine sim between candidate embeddings
48
- // drives the redundancy penalty. Candidates without an embedding are kept
49
- // at the end (no penalty applies).
50
- function cosineSim(a, b) {
51
- if (!a || !b || a.length !== b.length) return 0;
52
- let dot = 0; let na = 0; let nb = 0;
53
- for (let i = 0; i < a.length; i += 1) { dot += a[i] * b[i]; na += a[i] * a[i]; nb += b[i] * b[i]; }
54
- const denom = Math.sqrt(na) * Math.sqrt(nb);
55
- return denom === 0 ? 0 : dot / denom;
56
- }
57
-
58
- export function mmrRerank(candidates, embeddings, { topK, lambda = 0.7 }) {
59
- const remaining = candidates.slice();
60
- const picked = [];
61
- const relevance = (c) => 1 / (1 + Math.max(0, c.score ?? c.distance ?? 0));
62
- while (picked.length < topK && remaining.length > 0) {
63
- let bestIdx = 0; let bestVal = -Infinity;
64
- for (let i = 0; i < remaining.length; i += 1) {
65
- const c = remaining[i];
66
- const rel = relevance(c);
67
- let maxSim = 0;
68
- const ce = embeddings.get(c.id);
69
- if (ce && picked.length > 0) {
70
- for (const p of picked) {
71
- const pe = embeddings.get(p.id);
72
- if (pe) maxSim = Math.max(maxSim, cosineSim(ce, pe));
73
- }
74
- }
75
- const score = lambda * rel - (1 - lambda) * maxSim;
76
- if (score > bestVal) { bestVal = score; bestIdx = i; }
77
- }
78
- picked.push(remaining.splice(bestIdx, 1)[0]);
79
- }
80
- return picked;
81
- }
82
-
83
- export async function query(db, embedder, text, { topK = 5, scope = "all", kindBoost = DEFAULT_KIND_BOOST, project = null, mode = "hybrid", alpha = 0.6, where = null, rerankOpts = null, diversify = false, mmrLambda = 0.7 } = {}) {
84
- if (!text || typeof text !== "string") throw new Error("query: text must be a non-empty string");
85
- const fetchK = Math.min(50, topK * 2);
86
- const scopeKind = scope === "plans" ? "plan" : scope === "code" ? "code" : scope === "onboarding" ? "onboarding" : null;
87
- // KJC-TSK-0448 — metadata filter. `where` is a string like "symbol=Foo AND
88
- // hu_id=HU-003"; we parse once and reuse the SQL fragment across both
89
- // semantic and keyword searches.
90
- const parsed = parseWhere(where);
91
- if (!parsed.ok) throw new Error(`query: ${parsed.error}`);
92
- const { sql: whereSql, params: whereParams } = buildWhereSql(parsed.clauses);
93
- const opts = { kind: scopeKind, project, whereSql, whereParams };
94
- const wantSemantic = mode !== "keyword";
95
- const wantKeyword = mode !== "semantic";
96
- const semanticHits = wantSemantic ? searchSimilar(db, await embedder.embed(text), fetchK, opts) : [];
97
- const keywordHits = wantKeyword ? searchBM25(db, text, fetchK, opts) : [];
98
- const raw = fuseHits(semanticHits, keywordHits, alpha, mode);
99
- if (raw.length === 0) return [];
100
- const boostSources = shouldBoostSources(text);
101
- const scored = raw.map((r) => {
102
- const kindB = kindBoost[r.kind] || 0;
103
- const sourceB = (boostSources && r.kind === "code" && !isTestPath(r.source)) ? SOURCE_BOOST_NON_TEST : 0;
104
- return { ...r, score: r.distance - kindB - sourceB };
105
- }).sort((a, b) => a.score - b.score);
106
- // KJC-TSK-0484 PR-B — MMR diversification over topK*2 candidates. Fetches
107
- // embeddings only when requested; cosine between candidate vectors drives
108
- // the redundancy penalty so near-duplicates that survived dedup (e.g. a
109
- // boilerplate chunk repeated under slightly different wording) get pushed
110
- // down. Skipped entirely when `diversify=false` (default).
111
- let reranked;
112
- if (diversify) {
113
- const pool = scored.slice(0, Math.min(scored.length, topK * 2));
114
- const embeddings = getEmbeddingsByIds(db, pool.map((c) => c.id));
115
- reranked = mmrRerank(pool, embeddings, { topK, lambda: mmrLambda });
116
- } else {
117
- reranked = scored.slice(0, topK);
118
- }
119
- // KJC-TSK-0449 — optional cross-encoder rerank. When `rerankOpts` is set,
120
- // we re-score the topK survivors with a (query, passage) cross-encoder;
121
- // the kind+source boost has already been applied, so the rerank acts as
122
- // a finer-grained quality lever on top.
123
- if (rerankOpts) return await rerank(text, reranked, rerankOpts);
124
- return reranked;
125
- }
126
-
127
- export { fuseHits, cosineSim };
1
+ // Shim: retriever now lives in karajan-core/rag (KJC-TSK-0632 PR3).
2
+ export * from "karajan-core/rag/retriever";