karajan-code 3.14.1 → 3.15.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -383,6 +383,16 @@ Each AI role is executed by the agent you choose:
383
383
  >
384
384
  > Full per-stage reference: [Pipeline roles](https://karajan-code.web.app/docs/handbook/pipeline-roles/) (handbook).
385
385
 
386
+ ## Step mode and parallel lanes (v3.14+)
387
+
388
+ Two ways to control how a plan executes:
389
+
390
+ **`kj run --step`** — supervise the orchestra iteration by iteration. After every iteration the pipeline pauses with a compact report (what happened, the reviewer's must-fix list, what the next iteration will do, spend vs cap) and asks: press Enter to continue, type `stop` to halt (resumable with `kj resume`), or **type instructions** — free text is injected into the feedback the coder reads next iteration, without clobbering the reviewer's own findings. Also offered as a question in the `kj init` wizard (`session.iteration_gate`).
391
+
392
+ **`kj run --plan <id> --parallel <n>`** — run a plan's independent HUs concurrently, each in its own **git worktree** under `.kj/worktrees/<huId>` on branch `kj-hu-<huId>`. The scheduler walks the `blocked_by` graph and only pairs HUs with disjoint `scope` paths (scopeless HUs run alone); the whole lane — coder, acceptance tests, diffs, sonar, final commit — runs inside its worktree while the main working tree stays parked. Governance is built in: default is `1` (fully sequential), a plan-level budget ceiling (`n × max_budget_usd`) stops the batch loudly when exhausted, and SonarQube serializes across lanes. Each fresh worktree is bootstrapped automatically (submodules + `npm ci`, or your `session.worktree_setup` command) and receives `KJ_LANE_SLOT` / `KJ_PORT_OFFSET` env vars so services started by tests don't collide on ports.
393
+
394
+ Full guide: [`docs/parallel-hus.md`](docs/parallel-hus.md).
395
+
386
396
  ## 5 AI agents supported
387
397
 
388
398
  | Agent | CLI | Install |
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "karajan-code",
3
- "version": "3.14.1",
3
+ "version": "3.15.0",
4
4
  "description": "Local multi-agent coding orchestrator with TDD, SonarQube, and code review pipeline",
5
5
  "type": "module",
6
6
  "license": "AGPL-3.0",
@@ -76,7 +76,9 @@ router.get('/version', (_req, res) => {
76
76
  */
77
77
  router.get('/standby', async (_req, res) => {
78
78
  try {
79
- const { listPendingStandby } = await import('../../../../src/brain/standby-store.js');
79
+ // KJC-TSK-0632: resolve from karajan-core directly — the CLI's
80
+ // src/brain/standby-store.js is just a re-export shim of this.
81
+ const { listPendingStandby } = await import('karajan-core/standby-store');
80
82
  const sessions = listPendingStandby();
81
83
  res.set('Cache-Control', 'no-store');
82
84
  res.json({ sessions });
@@ -1362,8 +1364,10 @@ router.post('/rag/query', async (req, res) => {
1362
1364
  }
1363
1365
  try {
1364
1366
  const { openVecStore, countChunks } = await import('karajan-core/vec-store');
1365
- const { makeEmbedder } = await import('../../../../src/rag/embedders/factory.js');
1366
- const { query } = await import('../../../../src/rag/retriever.js');
1367
+ // KJC-TSK-0632: resolved from karajan-core — the board carries zero
1368
+ // relative imports into the CLI src tree (see no-cli-imports test).
1369
+ const { makeEmbedder } = await import('karajan-core/rag/embedders/factory');
1370
+ const { query } = await import('karajan-core/rag/retriever');
1367
1371
  const db = openVecStore({ dim: 768 });
1368
1372
  try {
1369
1373
  if (countChunks(db) === 0) return res.json({ hits: [], empty: true, topK, scope });
@@ -166,7 +166,11 @@ try {
166
166
  console.log(`verify-pack: installing the tarball with pnpm into ${pnpmTmp}…`);
167
167
  // pnpm exits non-zero on ERR_PNPM_IGNORED_BUILDS (it skips native build
168
168
  // scripts by default) — expected here, so don't treat the exit as failure.
169
- spawnSync("pnpm", ["add", tgzPath, "--store-dir", path.join(pnpmTmp, ".store")], {
169
+ // minimum-release-age=0: modern pnpm quarantines freshly published
170
+ // versions (supply-chain protection), so right after publishing
171
+ // karajan-core it silently resolves an OLD one and this smoke fails on
172
+ // missing subpaths. The gate verifies packaging, not release-age policy.
173
+ spawnSync("pnpm", ["add", tgzPath, "--store-dir", path.join(pnpmTmp, ".store"), "--config.minimum-release-age=0"], {
170
174
  encoding: "utf8",
171
175
  env: childEnv,
172
176
  cwd: pnpmTmp,
@@ -300,7 +300,10 @@ export function createStreamJsonFilter(onOutput) {
300
300
  */
301
301
  function cleanExecaOpts(extra = {}) {
302
302
  const { CLAUDECODE: _CLAUDECODE, ...env } = process.env;
303
- return { env, stdin: "ignore", ...extra };
303
+ // PAR-H (KJC-TSK-0631): extra.env ADDS to the inherited env (lane slot
304
+ // vars) instead of replacing it wholesale.
305
+ const { env: extraEnv, ...rest } = extra;
306
+ return { env: extraEnv ? { ...env, ...extraEnv } : env, stdin: "ignore", ...rest };
304
307
  }
305
308
 
306
309
  /**
@@ -370,7 +373,8 @@ export class ClaudeAgent extends BaseAgent {
370
373
  const res = await this.runCommand(resolveBin("claude"), args, cleanExecaOpts({
371
374
  onOutput: streamFilter,
372
375
  silenceTimeoutMs: task.silenceTimeoutMs,
373
- timeout: task.timeoutMs
376
+ timeout: task.timeoutMs,
377
+ env: task.env
374
378
  }));
375
379
  const raw = pickOutput(res);
376
380
  const output = extractTextFromStreamJson(raw);
@@ -380,7 +384,7 @@ export class ClaudeAgent extends BaseAgent {
380
384
 
381
385
  // Without streaming, use json output to get structured response via stderr
382
386
  args.push("--output-format", "json");
383
- const res = await this.runCommand(resolveBin("claude"), args, cleanExecaOpts());
387
+ const res = await this.runCommand(resolveBin("claude"), args, cleanExecaOpts({ env: task.env }));
384
388
  const raw = pickOutput(res);
385
389
  const output = extractTextFromStreamJson(raw);
386
390
  const usage = extractUsageFromStreamJson(raw);
@@ -393,7 +397,8 @@ export class ClaudeAgent extends BaseAgent {
393
397
  const res = await this.runCommand(resolveBin("claude"), args, cleanExecaOpts({
394
398
  onOutput: task.onOutput,
395
399
  silenceTimeoutMs: task.silenceTimeoutMs,
396
- timeout: task.timeoutMs
400
+ timeout: task.timeoutMs,
401
+ env: task.env
397
402
  }));
398
403
  const raw = pickOutput(res);
399
404
  const usage = extractUsageFromStreamJson(raw);
@@ -108,6 +108,29 @@ registerModelAlias("gemini", "gemini-2.5-pro");
108
108
  registerModel("aider", { provider: "aider", pricing: { input_per_million: 3, output_per_million: 15 } });
109
109
  registerModel("opencode", { provider: "opencode", pricing: { input_per_million: 0, output_per_million: 0 } });
110
110
 
111
+ /**
112
+ * Moonshot Kimi Family (KJC-TSK-0633) — consumed through OpenCode as an
113
+ * OpenAI-compatible provider (docs/providers-via-opencode.md).
114
+ * Pricing: https://platform.moonshot.ai/docs/pricing — verify before
115
+ * trusting for billing decisions; ids/prices move fast.
116
+ */
117
+ registerModel("kimi-k2", { provider: "moonshot", pricing: { input_per_million: 0.6, output_per_million: 2.5 } });
118
+ registerModel("kimi-k2-thinking", { provider: "moonshot", pricing: { input_per_million: 0.6, output_per_million: 2.5 } });
119
+ registerModelAlias("kimi", "kimi-k2");
120
+ // Prefixed ids as the documented opencode.json snippet produces them.
121
+ registerModelAlias("kimi/kimi-k2", "kimi-k2");
122
+ registerModelAlias("kimi/kimi-k2-thinking", "kimi-k2-thinking");
123
+
124
+ /**
125
+ * DeepSeek Family (KJC-TSK-0633) — same OpenCode route.
126
+ * Pricing: https://api-docs.deepseek.com/quick_start/pricing
127
+ */
128
+ registerModel("deepseek-chat", { provider: "deepseek", pricing: { input_per_million: 0.28, output_per_million: 0.42 } });
129
+ registerModel("deepseek-reasoner", { provider: "deepseek", pricing: { input_per_million: 0.28, output_per_million: 0.42 } });
130
+ registerModelAlias("deepseek", "deepseek-chat");
131
+ registerModelAlias("deepseek/deepseek-chat", "deepseek-chat");
132
+ registerModelAlias("deepseek/deepseek-reasoner", "deepseek-reasoner");
133
+
111
134
  // Common CLI Aliases (with provider overrides)
112
135
  registerModelAlias("aider/claude-3-7-sonnet", "claude-sonnet-4.6", { provider: "aider" });
113
136
  registerModelAlias("aider/gpt-4o", "gpt-5.4-standard", { provider: "aider" });
@@ -182,6 +182,10 @@ const DEFAULTS = {
182
182
  // Concurrent HU lanes per plan run (KJC-TSK-0626). 1 = sequential.
183
183
  // Raising it multiplies token burn rate — the plan budget scales with it.
184
184
  max_parallel_hus: 1,
185
+ // Command run inside each fresh lane worktree before the coder starts
186
+ // (KJC-TSK-0630). null = auto-detect: `npm ci` when package-lock.json
187
+ // exists, nothing otherwise. Submodules are always initialized first.
188
+ worktree_setup: null,
185
189
  max_iteration_minutes: 30,
186
190
  max_total_minutes: 120,
187
191
  max_planner_minutes: 60,
@@ -12,11 +12,15 @@ import { runCommand } from "../utils/process.js";
12
12
  * @param {number} [timeoutMs=30000] - Timeout per test
13
13
  * @returns {Promise<{cmd: string, passed: boolean, output: string, exitCode: number}>}
14
14
  */
15
- async function runSingleTest(cmd, cwd, timeoutMs = 30000) {
15
+ async function runSingleTest(cmd, cwd, timeoutMs = 30000, env = null) {
16
16
  try {
17
17
  const result = await runCommand("bash", ["-c", cmd], {
18
18
  timeout: timeoutMs,
19
- cwd
19
+ cwd,
20
+ // PAR-H (KJC-TSK-0631): lane env (KJ_LANE_SLOT / KJ_PORT_OFFSET) so
21
+ // tests that start services can offset their ports. execa merges
22
+ // this on top of process.env.
23
+ ...(env ? { env } : {})
20
24
  });
21
25
  const output = (result.stdout || "") + (result.stderr || "");
22
26
  return {
@@ -51,7 +55,7 @@ async function runSingleTest(cmd, cwd, timeoutMs = 30000) {
51
55
  * @param {string} cwd - Working directory
52
56
  * @returns {Promise<{allPassed: boolean, results: object[], summary: string, diagnostics: string|null, pending: number}>}
53
57
  */
54
- export async function runAcceptanceTests(tests, cwd) {
58
+ export async function runAcceptanceTests(tests, cwd, { env = null } = {}) {
55
59
  if (!tests || tests.length === 0) {
56
60
  return { allPassed: false, results: [], summary: "No acceptance tests defined", diagnostics: null, pending: 0 };
57
61
  }
@@ -86,7 +90,7 @@ export async function runAcceptanceTests(tests, cwd) {
86
90
  });
87
91
  continue;
88
92
  }
89
- const result = await runSingleTest(cmd, cwd);
93
+ const result = await runSingleTest(cmd, cwd, 30000, env);
90
94
  results.push({ ...result, type: "shell" });
91
95
  }
92
96
 
@@ -0,0 +1,59 @@
1
+ /**
2
+ * PAR-G (KJC-TSK-0630): make a freshly created lane worktree operative.
3
+ *
4
+ * `git worktree add` produces a clean checkout: no node_modules, no
5
+ * initialized submodules — and a container/bind-mounted process cannot
6
+ * init them itself because the worktree's real .git lives in the parent
7
+ * repo (gotcha reported by Jorge del Casar's worktree-docker-envs skill).
8
+ * Without this step, --parallel lanes die on the first `npm test` in any
9
+ * real project.
10
+ *
11
+ * Best-effort by contract: a failed or slow bootstrap warns and the lane
12
+ * continues — the acceptance tests deliver the real verdict.
13
+ */
14
+ import { existsSync } from "node:fs";
15
+ import { join } from "node:path";
16
+ import { runCommand } from "../utils/process.js";
17
+
18
+ const STEP_TIMEOUT_MS = 5 * 60 * 1000;
19
+
20
+ /**
21
+ * @param {object} params
22
+ * @param {string} params.worktreePath Absolute path of the lane worktree.
23
+ * @param {string|null} [params.setupCommand] session.worktree_setup — wins over auto-detect.
24
+ * @param {object|null} [params.logger]
25
+ * @param {number} [params.timeoutMs]
26
+ * @param {Function} [params.run] Injectable runner (tests).
27
+ * @returns {Promise<{ok: boolean, steps: string[], warnings: string[]}>}
28
+ */
29
+ export async function bootstrapWorktree({ worktreePath, setupCommand = null, logger = null, timeoutMs = STEP_TIMEOUT_MS, run = runCommand }) {
30
+ const warnings = [];
31
+ const steps = [];
32
+
33
+ if (existsSync(join(worktreePath, ".gitmodules"))) {
34
+ steps.push({ name: "submodules", cmd: "git", args: ["submodule", "update", "--init", "--recursive"] });
35
+ }
36
+
37
+ const explicit = typeof setupCommand === "string" && setupCommand.trim() ? setupCommand.trim() : null;
38
+ if (explicit) {
39
+ steps.push({ name: "setup", cmd: "sh", args: ["-c", explicit] });
40
+ } else if (existsSync(join(worktreePath, "package-lock.json"))) {
41
+ steps.push({ name: "deps", cmd: "npm", args: ["ci", "--no-audit", "--no-fund"] });
42
+ }
43
+
44
+ for (const step of steps) {
45
+ try {
46
+ const res = await run(step.cmd, step.args, { cwd: worktreePath, timeout: timeoutMs });
47
+ if (res.exitCode !== 0) {
48
+ warnings.push(`${step.name} failed (exit ${res.exitCode}): ${String(res.stderr || res.stdout || "").slice(0, 200)}`);
49
+ } else {
50
+ logger?.info?.(`worktree bootstrap: ${step.name} ok`);
51
+ }
52
+ } catch (err) {
53
+ warnings.push(`${step.name} threw: ${err.message}`);
54
+ }
55
+ }
56
+
57
+ for (const w of warnings) logger?.warn?.(`worktree bootstrap: ${w} — lane continues`);
58
+ return { ok: warnings.length === 0, steps: steps.map((s) => s.name), warnings };
59
+ }
@@ -104,7 +104,13 @@ export async function runHuBatch({ ctx, task, askQuestion, emitter, logger }) {
104
104
  // left it clamped to hu_max_iterations for the rest of the run.
105
105
  const worktreePath = laneOpts?.worktreePath || null;
106
106
  const projectDir = worktreePath || ctx.config.projectDir || process.cwd();
107
- const laneConfig = { ...ctx.config, projectDir, max_iterations: huMaxIterations };
107
+ // PAR-H (KJC-TSK-0631): the lane's slot travels as env vars so any
108
+ // service the coder or the acceptance tests start can offset its
109
+ // ports (offset = slot × 100; the project applies its own base).
110
+ const laneEnv = Number.isInteger(laneOpts?.laneSlot)
111
+ ? { KJ_LANE_SLOT: String(laneOpts.laneSlot), KJ_PORT_OFFSET: String(laneOpts.laneSlot * 100) }
112
+ : null;
113
+ const laneConfig = { ...ctx.config, projectDir, max_iterations: huMaxIterations, lane_env: laneEnv };
108
114
  if (!huPolicies.tdd) laneConfig.development = { ...laneConfig.development, methodology: "standard", require_test_changes: false };
109
115
  if (!huPolicies.sonar) laneConfig.sonarqube = { ...laneConfig.sonarqube, enabled: false };
110
116
  const laneFlags = { ...ctx.pipelineFlags };
@@ -233,7 +239,7 @@ export async function runHuBatch({ ctx, task, askQuestion, emitter, logger }) {
233
239
  detail: { huId: story.id, testCount: story.acceptance_tests.length }
234
240
  }));
235
241
 
236
- const testResult = await runAcceptanceTests(story.acceptance_tests, projectDir);
242
+ const testResult = await runAcceptanceTests(story.acceptance_tests, projectDir, { env: laneEnv });
237
243
  emitProgress(emitter, makeEvent("hu:acceptance-end", { ...ctx.eventBase, stage: "acceptance" }, {
238
244
  status: testResult.allPassed ? "ok" : "fail",
239
245
  message: testResult.summary,
@@ -6,7 +6,11 @@ import { topologicalSort } from "../hu/graph.js";
6
6
  import { updateStoryStatus, loadHuBatch, saveHuBatch, HU_STATUS } from "../hu/store.js";
7
7
  import { emitProgress, makeEvent } from "../utils/events.js";
8
8
  import { refineHuWithContext } from "../hu/lazy-planner.js";
9
+ import { join } from "node:path";
10
+ import { acquireSlot, releaseSlot } from "karajan-core/slot-registry";
11
+ import { getKarajanHome } from "karajan-core/paths";
9
12
  import { findParallelGroups, createWorktree, mergeWorktree, removeWorktree } from "../hu/parallel-executor.js";
13
+ import { bootstrapWorktree } from "../hu/worktree-bootstrap.js";
10
14
  import { partitionConflictFree } from "./hu-scheduler.js";
11
15
  import { createParallelLimiter, planBudgetUsd } from "./parallel-limiter.js";
12
16
 
@@ -200,7 +204,7 @@ function buildHuOutcome({ story: _story, iterResult, status, startedAt, huBudget
200
204
  * @param {object} params
201
205
  * @returns {Promise<{huId: string, approved: boolean, result?: object, error?: string, blockedDependents?: string[]}>}
202
206
  */
203
- async function runSingleHu({ storyId, batch, batchSessionId, runIterationFn, emitter, eventBase, logger, config, results, worktreePath, onStatusChange = null, onOutcome = null, budgetTracker = null }) {
207
+ async function runSingleHu({ storyId, batch, batchSessionId, runIterationFn, emitter, eventBase, logger, config, results, worktreePath, laneSlot = null, onStatusChange = null, onOutcome = null, budgetTracker = null }) {
204
208
  // PR1 (live HU status): every saveHuBatch should also notify the
205
209
  // plan JSON so the board reflects state in real time. Defined as a
206
210
  // local helper so we don't repeat the try/null-check boilerplate.
@@ -287,7 +291,7 @@ async function runSingleHu({ storyId, batch, batchSessionId, runIterationFn, emi
287
291
  try {
288
292
  // PAR-E2 (KJC-TSK-0629): lanes handed a worktree must aim every git and
289
293
  // filesystem touchpoint at it — laneOpts carries that path to the runner.
290
- const iterResult = await runIterationFn(huTask, story, { worktreePath: worktreePath || null });
294
+ const iterResult = await runIterationFn(huTask, story, { worktreePath: worktreePath || null, laneSlot });
291
295
  const approved = Boolean(iterResult?.approved);
292
296
 
293
297
  // --- Transition to reviewing (post-coder, pre-reviewer evaluation) ---
@@ -527,12 +531,27 @@ export async function runHuSubPipeline({ huReviewerResult, runIterationFn, emitt
527
531
  // Multiple HUs: create worktrees, run in parallel
528
532
  const projectDir = config?.projectDir || process.cwd();
529
533
  const worktrees = new Map();
534
+ // PAR-H (KJC-TSK-0631): every lane gets a stable numeric slot so
535
+ // services it starts can offset their ports (KJ_LANE_SLOT /
536
+ // KJ_PORT_OFFSET reach the coder and the acceptance tests).
537
+ const slotRegistryPath = join(getKarajanHome(), "worktree-slots.json");
538
+ const laneSlots = new Map();
530
539
 
531
540
  // Create worktrees for each HU in the batch
532
541
  for (const id of runnableIds) {
533
542
  try {
534
543
  const wtPath = await createWorktree(projectDir, id);
535
544
  worktrees.set(id, wtPath);
545
+ // PAR-G (KJC-TSK-0630): a fresh worktree has no node_modules and
546
+ // no initialized submodules — make the lane operative before the
547
+ // coder lands. Best-effort: warnings never block the lane.
548
+ await bootstrapWorktree({ worktreePath: wtPath, setupCommand: config?.session?.worktree_setup || null, logger });
549
+ try {
550
+ const { slot } = await acquireSlot({ registryPath: slotRegistryPath, id: `${projectDir}::${id}` });
551
+ laneSlots.set(id, slot);
552
+ } catch (err) {
553
+ logger.warn(`Failed to acquire lane slot for HU ${id}: ${err.message} — lane runs without port offset`);
554
+ }
536
555
  } catch (err) {
537
556
  logger.warn(`Failed to create worktree for HU ${id}: ${err.message} — will run sequentially`);
538
557
  }
@@ -547,6 +566,7 @@ export async function runHuSubPipeline({ huReviewerResult, runIterationFn, emitt
547
566
  storyId, batch, batchSessionId, runIterationFn,
548
567
  emitter, eventBase, logger, config, results,
549
568
  worktreePath: worktrees.get(storyId),
569
+ laneSlot: laneSlots.get(storyId) ?? null,
550
570
  onStatusChange, onOutcome, budgetTracker
551
571
  });
552
572
  } finally {
@@ -562,6 +582,7 @@ export async function runHuSubPipeline({ huReviewerResult, runIterationFn, emitt
562
582
  if (res.approved && worktrees.has(res.huId)) {
563
583
  try {
564
584
  await mergeWorktree(projectDir, res.huId);
585
+ await releaseSlot({ registryPath: slotRegistryPath, id: `${projectDir}::${res.huId}` }).catch(() => {});
565
586
  } catch (err) {
566
587
  logger.warn(`Failed to merge worktree for HU ${res.huId}: ${err.message}`);
567
588
  }
@@ -581,6 +602,7 @@ export async function runHuSubPipeline({ huReviewerResult, runIterationFn, emitt
581
602
  // Clean up failed worktree
582
603
  if (worktrees.has(res.huId)) {
583
604
  try { await removeWorktree(projectDir, res.huId); } catch { /* ignore */ }
605
+ await releaseSlot({ registryPath: slotRegistryPath, id: `${projectDir}::${res.huId}` }).catch(() => {});
584
606
  }
585
607
  }
586
608
  }
@@ -71,6 +71,9 @@ export async function runCoderStage({ coderRoleInstance, coderRole, config, logg
71
71
  // PAR-E2 (KJC-TSK-0629): the stage's config wins over the role's own —
72
72
  // worktree lanes pass a laneConfig whose projectDir is the worktree.
73
73
  projectDir: config?.projectDir || null,
74
+ // PAR-H (KJC-TSK-0631): lane env (KJ_LANE_SLOT / KJ_PORT_OFFSET)
75
+ // reaches the coder subprocess so services it starts don't collide.
76
+ env: config?.lane_env || null,
74
77
  onOutput: coderStall.onOutput,
75
78
  // Lets Brain Recovery persist a standby snapshot if the coder's
76
79
  // provider hits a quota cap mid-run (KJC hibernation wiring).
@@ -1,81 +1,2 @@
1
- // KJC-PCS-0049 Step 2 — Ollama embedder adapter for the RAG pipeline.
2
- // Talks to a local Ollama server (POST /api/embeddings) and returns
3
- // Float32Array vectors. Cero deps externas; usa fetch global.
4
- //
5
- // Defaults:
6
- // url = process.env.KJ_OLLAMA_URL or "http://localhost:11434"
7
- // model = process.env.KJ_OLLAMA_EMBED_MODEL or "nomic-embed-text"
8
- // dim = 768 (matches nomic-embed-text)
9
- // timeoutMs = 30000
10
-
11
- const DEFAULT_URL = "http://localhost:11434";
12
- const DEFAULT_MODEL = "nomic-embed-text";
13
- const DEFAULT_DIM = 768;
14
- const DEFAULT_TIMEOUT_MS = 30000;
15
-
16
- export class OllamaEmbedderError extends Error {
17
- constructor(message, { cause, status } = {}) {
18
- super(message);
19
- this.name = "OllamaEmbedderError";
20
- if (cause) this.cause = cause;
21
- if (status != null) this.status = status;
22
- }
23
- }
24
-
25
- export class OllamaEmbedder {
26
- constructor({
27
- url = process.env.KJ_OLLAMA_URL || DEFAULT_URL,
28
- model = process.env.KJ_OLLAMA_EMBED_MODEL || DEFAULT_MODEL,
29
- dim = DEFAULT_DIM,
30
- timeoutMs = DEFAULT_TIMEOUT_MS,
31
- fetchFn = globalThis.fetch,
32
- } = {}) {
33
- this.url = url.replace(/\/$/, "");
34
- this.model = model;
35
- this.dim = dim;
36
- this.timeoutMs = timeoutMs;
37
- this.fetch = fetchFn;
38
- }
39
-
40
- /** Single-text embedding. Returns Float32Array(dim). */
41
- async embed(text) {
42
- if (typeof text !== "string" || text.length === 0) {
43
- throw new OllamaEmbedderError("embed: text must be a non-empty string");
44
- }
45
- const ctrl = new AbortController();
46
- const timer = setTimeout(() => ctrl.abort(), this.timeoutMs);
47
- let res;
48
- try {
49
- res = await this.fetch(`${this.url}/api/embeddings`, {
50
- method: "POST",
51
- headers: { "Content-Type": "application/json" },
52
- body: JSON.stringify({ model: this.model, prompt: text }),
53
- signal: ctrl.signal,
54
- });
55
- } catch (err) {
56
- throw new OllamaEmbedderError(`Ollama embed request failed (${this.url}): ${err.message}`, { cause: err });
57
- } finally {
58
- clearTimeout(timer);
59
- }
60
- if (!res.ok) {
61
- throw new OllamaEmbedderError(`Ollama embed HTTP ${res.status} from ${this.url}`, { status: res.status });
62
- }
63
- const body = await res.json();
64
- const arr = body?.embedding;
65
- if (!Array.isArray(arr) || arr.length === 0) {
66
- throw new OllamaEmbedderError("Ollama response missing 'embedding' array");
67
- }
68
- if (arr.length !== this.dim) {
69
- throw new OllamaEmbedderError(`Ollama dim mismatch: got ${arr.length}, expected ${this.dim} for model ${this.model}`);
70
- }
71
- return Float32Array.from(arr);
72
- }
73
-
74
- /** Batch embedding (sequential — Ollama /api/embeddings is single-text). */
75
- async embedBatch(texts) {
76
- if (!Array.isArray(texts)) throw new OllamaEmbedderError("embedBatch: texts must be an array");
77
- const out = [];
78
- for (const t of texts) out.push(await this.embed(t));
79
- return out;
80
- }
81
- }
1
+ // Shim: embedder now lives in karajan-core/rag (KJC-TSK-0632 PR3).
2
+ export * from "karajan-core/rag/embedder";
@@ -1,24 +1,2 @@
1
- // KJC-TSK-0442 — shared POST helper for cloud embedders (OpenAI, Voyage).
2
- // Same envelope: Bearer auth, JSON body shaped by caller, JSON response
3
- // extracted by caller, dim validated against the adapter's expected size.
4
- export async function cloudEmbed(adapter, text, ErrCls, { provider, body, extract }) {
5
- if (typeof text !== "string" || text.length === 0) throw new ErrCls("embed: text must be a non-empty string");
6
- const ctrl = new AbortController();
7
- const timer = setTimeout(() => ctrl.abort(), adapter.timeoutMs);
8
- let res;
9
- try {
10
- res = await adapter.fetch(adapter.url, {
11
- method: "POST",
12
- headers: { "Content-Type": "application/json", Authorization: `Bearer ${adapter.apiKey}` },
13
- body: JSON.stringify(body),
14
- signal: ctrl.signal,
15
- });
16
- } catch (err) { throw new ErrCls(`${provider} embed request failed: ${err.message}`, { cause: err }); }
17
- finally { clearTimeout(timer); }
18
- if (!res.ok) throw new ErrCls(`${provider} embed HTTP ${res.status}`, { status: res.status });
19
- const json = await res.json();
20
- const arr = extract(json);
21
- if (!Array.isArray(arr) || arr.length === 0) throw new ErrCls(`${provider} response missing embedding`);
22
- if (arr.length !== adapter.dim) throw new ErrCls(`${provider} dim mismatch: got ${arr.length}, expected ${adapter.dim} for model ${adapter.model}`);
23
- return Float32Array.from(arr);
24
- }
1
+ // Shim: moved to karajan-core/rag/embedders (KJC-TSK-0632 PR3).
2
+ export * from "karajan-core/rag/embedders/_cloud-base";
@@ -1,28 +1,2 @@
1
- // KJC-TSK-0446 — Cohere embed-v3 adapter. Cohere returns
2
- // { embeddings: { float: [[...]] } } when `embedding_types: ["float"]`
3
- // is requested, or { embeddings: [[...]] } in legacy v1. We extract the
4
- // modern shape with a v1 fallback.
5
- import { cloudEmbed } from "./_cloud-base.js";
6
-
7
- const DEFAULTS = { url: "https://api.cohere.com/v2/embed", model: "embed-multilingual-v3.0", dim: 1024, timeoutMs: 30000 };
8
-
9
- export class CohereEmbedderError extends Error {
10
- constructor(message, { cause, status } = {}) { super(message); this.name = "CohereEmbedderError"; if (cause) this.cause = cause; if (status != null) this.status = status; }
11
- }
12
-
13
- export class CohereEmbedder {
14
- constructor({ url = DEFAULTS.url, apiKey = process.env.KJ_COHERE_KEY, model = process.env.KJ_COHERE_EMBED_MODEL || DEFAULTS.model, dim = DEFAULTS.dim, timeoutMs = DEFAULTS.timeoutMs, fetchFn = globalThis.fetch } = {}) {
15
- // KJC architecture invariant: scoped env var (KJ_COHERE_KEY), not the
16
- // generic COHERE_API_KEY — same rule as OpenAI/Voyage adapters.
17
- if (!apiKey) throw new CohereEmbedderError("Cohere embedder requires an api_key (config.rag.embedder.api_key or KJ_COHERE_KEY env)");
18
- Object.assign(this, { url, apiKey, model, dim, timeoutMs, fetch: fetchFn });
19
- }
20
- async embed(text) {
21
- return cloudEmbed(this, text, CohereEmbedderError, {
22
- provider: "Cohere",
23
- body: { model: this.model, texts: [text], input_type: "search_document", embedding_types: ["float"] },
24
- extract: (b) => b?.embeddings?.float?.[0] || b?.embeddings?.[0],
25
- });
26
- }
27
- async embedBatch(texts) { if (!Array.isArray(texts)) throw new CohereEmbedderError("embedBatch: texts must be an array"); const out = []; for (const t of texts) out.push(await this.embed(t)); return out; }
28
- }
1
+ // Shim: moved to karajan-core/rag/embedders (KJC-TSK-0632 PR3).
2
+ export * from "karajan-core/rag/embedders/cohere";
@@ -1,29 +1,2 @@
1
- // KJC-TSK-0442 / KJC-TSK-0446 — Embedder factory. Selects provider from
2
- // config.rag.embedder.provider (ollama | openai | voyage | cohere | mistral;
3
- // default ollama). Each provider has its own dim default; the factory wires
4
- // it so the caller does not need to know.
5
- import { OllamaEmbedder } from "../embedder.js";
6
- import { OpenAIEmbedder } from "./openai.js";
7
- import { VoyageEmbedder } from "./voyage.js";
8
- import { CohereEmbedder } from "./cohere.js";
9
- import { MistralEmbedder } from "./mistral.js";
10
- import { ONNXEmbedder } from "./onnx.js";
11
-
12
- const PROVIDERS = {
13
- ollama: { cls: OllamaEmbedder, dim: 768 },
14
- openai: { cls: OpenAIEmbedder, dim: 1536 },
15
- voyage: { cls: VoyageEmbedder, dim: 1024 },
16
- cohere: { cls: CohereEmbedder, dim: 1024 },
17
- mistral: { cls: MistralEmbedder, dim: 1024 },
18
- onnx: { cls: ONNXEmbedder, dim: 384 },
19
- };
20
-
21
- export function makeEmbedder(config = {}) {
22
- const cfg = config?.rag?.embedder || {};
23
- const provider = cfg.provider || "ollama";
24
- const spec = PROVIDERS[provider];
25
- if (!spec) throw new Error(`Unknown embedder provider: ${provider}. Supported: ${Object.keys(PROVIDERS).join(", ")}`);
26
- const dim = cfg.dim || spec.dim;
27
- return new spec.cls({ url: cfg.url, apiKey: cfg.api_key, model: cfg.model, dim, timeoutMs: cfg.timeout_ms });
28
- }
29
-
1
+ // Shim: moved to karajan-core/rag/embedders (KJC-TSK-0632 PR3).
2
+ export * from "karajan-core/rag/embedders/factory";
@@ -1,26 +1,2 @@
1
- // KJC-TSK-0446 — Mistral AI embeddings adapter. EU-hosted, useful for
2
- // users with GDPR constraints that prefer not to send chunks to US
3
- // endpoints. Single model today (`mistral-embed`, 1024 dim).
4
- import { cloudEmbed } from "./_cloud-base.js";
5
-
6
- const DEFAULTS = { url: "https://api.mistral.ai/v1/embeddings", model: "mistral-embed", dim: 1024, timeoutMs: 30000 };
7
-
8
- export class MistralEmbedderError extends Error {
9
- constructor(message, { cause, status } = {}) { super(message); this.name = "MistralEmbedderError"; if (cause) this.cause = cause; if (status != null) this.status = status; }
10
- }
11
-
12
- export class MistralEmbedder {
13
- constructor({ url = DEFAULTS.url, apiKey = process.env.KJ_MISTRAL_KEY, model = process.env.KJ_MISTRAL_EMBED_MODEL || DEFAULTS.model, dim = DEFAULTS.dim, timeoutMs = DEFAULTS.timeoutMs, fetchFn = globalThis.fetch } = {}) {
14
- // KJC architecture invariant: scoped env var (KJ_MISTRAL_KEY).
15
- if (!apiKey) throw new MistralEmbedderError("Mistral embedder requires an api_key (config.rag.embedder.api_key or KJ_MISTRAL_KEY env)");
16
- Object.assign(this, { url, apiKey, model, dim, timeoutMs, fetch: fetchFn });
17
- }
18
- async embed(text) {
19
- return cloudEmbed(this, text, MistralEmbedderError, {
20
- provider: "Mistral",
21
- body: { model: this.model, input: [text] },
22
- extract: (b) => b?.data?.[0]?.embedding,
23
- });
24
- }
25
- async embedBatch(texts) { if (!Array.isArray(texts)) throw new MistralEmbedderError("embedBatch: texts must be an array"); const out = []; for (const t of texts) out.push(await this.embed(t)); return out; }
26
- }
1
+ // Shim: moved to karajan-core/rag/embedders (KJC-TSK-0632 PR3).
2
+ export * from "karajan-core/rag/embedders/mistral";
@@ -1,63 +1,2 @@
1
- // KJC-TSK-0447 — ONNX local embedder via @huggingface/transformers
2
- // (formerly @xenova/transformers). Runs sentence-transformer models
3
- // directly in Node — no API key, no Docker, no Ollama.
4
- //
5
- // First call downloads the model weights to ~/.cache/huggingface/ (~80 MB
6
- // for the default MiniLM). Subsequent calls reuse the cache, so the steady
7
- // state is fully offline.
8
- //
9
- // Default model: `Xenova/all-MiniLM-L6-v2` (384 dim, ~80 MB, fast).
10
- // Higher-quality alternative: `Xenova/jina-embeddings-v2-base-en`
11
- // (768 dim, ~260 MB, slower but better for retrieval).
12
-
13
- const DEFAULTS = { model: "Xenova/all-MiniLM-L6-v2", dim: 384, pooling: "mean", normalize: true };
14
-
15
- export class ONNXEmbedderError extends Error {
16
- constructor(message, { cause } = {}) { super(message); this.name = "ONNXEmbedderError"; if (cause) this.cause = cause; }
17
- }
18
-
19
- async function loadTransformers() {
20
- // Prefer the modern @huggingface/transformers (post-2024 rebrand);
21
- // fall back to @xenova/transformers for users still on the legacy package.
22
- // Both are optional peer deps — Karajan does not ship them by default to
23
- // keep the install size small (~500 MB with WASM + weights). The user
24
- // installs whichever they prefer when they opt into provider: onnx.
25
- /* eslint-disable import-x/no-unresolved */
26
- try { return await import("@huggingface/transformers"); }
27
- catch (e1) {
28
- try { return await import("@xenova/transformers"); }
29
- catch (e2) {
30
- throw new ONNXEmbedderError(
31
- "ONNX embedder requires @huggingface/transformers (or legacy @xenova/transformers). Install with `npm install @huggingface/transformers`.",
32
- { cause: e2 },
33
- );
34
- }
35
- }
36
- /* eslint-enable import-x/no-unresolved */
37
- }
38
-
39
- export class ONNXEmbedder {
40
- constructor({ model = process.env.KJ_ONNX_EMBED_MODEL || DEFAULTS.model, dim = DEFAULTS.dim, pooling = DEFAULTS.pooling, normalize = DEFAULTS.normalize } = {}) {
41
- Object.assign(this, { model, dim, pooling, normalize, _pipe: null });
42
- }
43
- async _ensurePipeline() {
44
- if (this._pipe) return this._pipe;
45
- const transformers = await loadTransformers();
46
- this._pipe = await transformers.pipeline("feature-extraction", this.model);
47
- return this._pipe;
48
- }
49
- async embed(text) {
50
- if (typeof text !== "string" || text.length === 0) throw new ONNXEmbedderError("embed: text must be a non-empty string");
51
- const pipe = await this._ensurePipeline();
52
- const out = await pipe(text, { pooling: this.pooling, normalize: this.normalize });
53
- const arr = Array.from(out.data || []);
54
- if (arr.length !== this.dim) throw new ONNXEmbedderError(`ONNX dim mismatch: got ${arr.length}, expected ${this.dim} for model ${this.model}`);
55
- return Float32Array.from(arr);
56
- }
57
- async embedBatch(texts) {
58
- if (!Array.isArray(texts)) throw new ONNXEmbedderError("embedBatch: texts must be an array");
59
- const out = [];
60
- for (const t of texts) out.push(await this.embed(t));
61
- return out;
62
- }
63
- }
1
+ // Shim: moved to karajan-core/rag/embedders (KJC-TSK-0632 PR3).
2
+ export * from "karajan-core/rag/embedders/onnx";
@@ -1,22 +1,2 @@
1
- // KJC-TSK-0442 — OpenAI text-embeddings adapter.
2
- import { cloudEmbed } from "./_cloud-base.js";
3
-
4
- const DEFAULTS = { url: "https://api.openai.com/v1/embeddings", model: "text-embedding-3-small", dim: 1536, timeoutMs: 30000 };
5
-
6
- export class OpenAIEmbedderError extends Error {
7
- constructor(message, { cause, status } = {}) { super(message); this.name = "OpenAIEmbedderError"; if (cause) this.cause = cause; if (status != null) this.status = status; }
8
- }
9
-
10
- export class OpenAIEmbedder {
11
- constructor({ url = DEFAULTS.url, apiKey = process.env.KJ_OPENAI_KEY, model = process.env.KJ_OPENAI_EMBED_MODEL || DEFAULTS.model, dim = DEFAULTS.dim, timeoutMs = DEFAULTS.timeoutMs, fetchFn = globalThis.fetch } = {}) {
12
- // KJC architecture invariant: Karajan does not read provider API keys
13
- // (ANTHROPIC_API_KEY, OPENAI_API_KEY, etc.) — those belong to the CLI
14
- // agents Karajan spawns. RAG embedders are the one exception: they
15
- // call the OpenAI endpoint directly, but use a Karajan-scoped env var
16
- // (KJ_OPENAI_KEY) so the invariant stays clean.
17
- if (!apiKey) throw new OpenAIEmbedderError("OpenAI embedder requires an api_key (config.rag.embedder.api_key or KJ_OPENAI_KEY env)");
18
- Object.assign(this, { url, apiKey, model, dim, timeoutMs, fetch: fetchFn });
19
- }
20
- async embed(text) { return cloudEmbed(this, text, OpenAIEmbedderError, { provider: "OpenAI", body: { model: this.model, input: text }, extract: (b) => b?.data?.[0]?.embedding }); }
21
- async embedBatch(texts) { if (!Array.isArray(texts)) throw new OpenAIEmbedderError("embedBatch: texts must be an array"); const out = []; for (const t of texts) out.push(await this.embed(t)); return out; }
22
- }
1
+ // Shim: moved to karajan-core/rag/embedders (KJC-TSK-0632 PR3).
2
+ export * from "karajan-core/rag/embedders/openai";
@@ -1,18 +1,2 @@
1
- // KJC-TSK-0442 — Voyage AI embeddings adapter.
2
- import { cloudEmbed } from "./_cloud-base.js";
3
-
4
- const DEFAULTS = { url: "https://api.voyageai.com/v1/embeddings", model: "voyage-code-3", dim: 1024, timeoutMs: 30000 };
5
-
6
- export class VoyageEmbedderError extends Error {
7
- constructor(message, { cause, status } = {}) { super(message); this.name = "VoyageEmbedderError"; if (cause) this.cause = cause; if (status != null) this.status = status; }
8
- }
9
-
10
- export class VoyageEmbedder {
11
- constructor({ url = DEFAULTS.url, apiKey = process.env.KJ_VOYAGE_KEY, model = process.env.KJ_VOYAGE_EMBED_MODEL || DEFAULTS.model, dim = DEFAULTS.dim, timeoutMs = DEFAULTS.timeoutMs, fetchFn = globalThis.fetch } = {}) {
12
- // Same invariant as OpenAIEmbedder: Karajan-scoped env var.
13
- if (!apiKey) throw new VoyageEmbedderError("Voyage embedder requires an api_key (config.rag.embedder.api_key or KJ_VOYAGE_KEY env)");
14
- Object.assign(this, { url, apiKey, model, dim, timeoutMs, fetch: fetchFn });
15
- }
16
- async embed(text) { return cloudEmbed(this, text, VoyageEmbedderError, { provider: "Voyage", body: { model: this.model, input: [text], input_type: "document" }, extract: (b) => b?.data?.[0]?.embedding }); }
17
- async embedBatch(texts) { if (!Array.isArray(texts)) throw new VoyageEmbedderError("embedBatch: texts must be an array"); const out = []; for (const t of texts) out.push(await this.embed(t)); return out; }
18
- }
1
+ // Shim: moved to karajan-core/rag/embedders (KJC-TSK-0632 PR3).
2
+ export * from "karajan-core/rag/embedders/voyage";
package/src/rag/rerank.js CHANGED
@@ -1,74 +1,4 @@
1
- // KJC-TSK-0449 — Cross-encoder rerank stage. Re-scores the top-N hits
2
- // from the hybrid retriever using a (query, passage) cross-encoder.
3
- //
4
- // Cross-encoders are slower than the bi-encoder embedders (they jointly
5
- // encode query + passage instead of caching the passage), but they are
6
- // substantially more precise for the final top-K ranking. Karajan calls
7
- // the reranker only on the post-fusion candidates (≤topK*2), so the
8
- // added latency is bounded.
9
- //
10
- // Model is loaded dynamically via @huggingface/transformers (same dep as
11
- // the ONNX embedder, KJC-TSK-0447). Default: `Xenova/ms-marco-MiniLM-L-6-v2`,
12
- // the de-facto standard sentence-transformers reranker.
13
-
14
- const DEFAULT_MODEL = "Xenova/ms-marco-MiniLM-L-6-v2";
15
-
16
- export class RerankError extends Error {
17
- constructor(message, { cause } = {}) { super(message); this.name = "RerankError"; if (cause) this.cause = cause; }
18
- }
19
-
20
- async function loadTransformers() {
21
- /* eslint-disable import-x/no-unresolved */
22
- try { return await import("@huggingface/transformers"); }
23
- catch (e1) {
24
- try { return await import("@xenova/transformers"); }
25
- catch (e2) {
26
- throw new RerankError(
27
- "rerank requires @huggingface/transformers (or legacy @xenova/transformers). Install with `npm install @huggingface/transformers`.",
28
- { cause: e2 },
29
- );
30
- }
31
- }
32
- /* eslint-enable import-x/no-unresolved */
33
- }
34
-
35
- /**
36
- * Lazy singleton — first call downloads weights (~80 MB), every subsequent
37
- * call reuses the same pipeline instance.
38
- */
39
- let _pipelinePromise = null;
40
- function getPipeline(modelName) {
41
- if (!_pipelinePromise) _pipelinePromise = (async () => {
42
- const t = await loadTransformers();
43
- // text-classification pipeline returns the cross-encoder relevance score
44
- // for a `[query, passage]` text pair when applied to a CE model.
45
- return t.pipeline("text-classification", modelName);
46
- })();
47
- return _pipelinePromise;
48
- }
49
-
50
- /** Reset for tests. */
51
- export function _resetPipeline() { _pipelinePromise = null; }
52
-
53
- /**
54
- * Rerank `hits` against `queryText` using the cross-encoder. Returns a NEW
55
- * array sorted by descending rerank score (best first), `score` mutated
56
- * so the existing downstream code (kindBoost, slicing) keeps working.
57
- *
58
- * Each input hit is treated as a [query, hit.text] pair. Truncates the
59
- * passage to `maxChars` to avoid blowing past the CE model's context
60
- * window — most CE models handle 512 tokens, ~2000 chars is safe.
61
- *
62
- * If `pipelineFn` is injected (tests), it bypasses model loading.
63
- */
64
- export async function rerank(queryText, hits, { model = process.env.KJ_RERANK_MODEL || DEFAULT_MODEL, maxChars = 2000, pipelineFn = null } = {}) {
65
- if (!queryText || typeof queryText !== "string") throw new RerankError("rerank: queryText must be a non-empty string");
66
- if (!Array.isArray(hits)) throw new RerankError("rerank: hits must be an array");
67
- if (hits.length === 0) return hits;
68
- const pipe = pipelineFn || await getPipeline(model);
69
- const pairs = hits.map((h) => ({ text: queryText, text_pair: String(h.text || "").slice(0, maxChars) }));
70
- const results = await pipe(pairs);
71
- return hits
72
- .map((h, i) => ({ ...h, _rerank: Number(results[i]?.score) || 0, score: -Number(results[i]?.score) || 0 }))
73
- .sort((a, b) => a.score - b.score);
74
- }
1
+ // Shim: rerank now lives in karajan-core/rag so the hu-board workspace
2
+ // can consume it without a relative dep on the CLI src tree.
3
+ // KJC-TSK-0632 PR2.
4
+ export { RerankError, _resetPipeline, rerank } from "karajan-core/rag/rerank";
@@ -1,127 +1,2 @@
1
- // KJC-PCS-0049 Step 5 — Retriever for the RAG pipeline.
2
- import { searchSimilar, searchBM25, getEmbeddingsByIds } from "./vec-store.js";
3
- import { parseWhere, buildWhereSql } from "./where-parser.js";
4
- import { rerank } from "./rerank.js";
5
-
6
- const DEFAULT_KIND_BOOST = { plan: 0.05, onboarding: 0.03, code: 0 };
7
-
8
- // KJC-TSK-0440 — asymmetric source/test boost.
9
- const SOURCE_BOOST_NON_TEST = 0.05;
10
- const TEST_TERMS_RE = /\b(test|tests|spec|specs|expect|describe|it\(|jest|vitest|mocha)\b/i;
11
- const TEST_PATH_RE = /[\\/](tests?|specs?|__tests__)[\\/]|\.test\.[jt]sx?$|\.spec\.[jt]sx?$/i;
12
-
13
- function shouldBoostSources(queryText) { return !TEST_TERMS_RE.test(queryText); }
14
- function isTestPath(source) { return TEST_PATH_RE.test(source || ""); }
15
-
16
- // KJC-TSK-0443 — fuse semantic + keyword hits into a unified candidate set.
17
- // For 'semantic'/'keyword' modes we surface that side. For 'hybrid' (default)
18
- // we min-max normalise both scores to [0,1] (lower=better) and linear-combine
19
- // via alpha * semantic + (1-alpha) * keyword. Result written back to
20
- // `distance` so the kind+source boost pipeline keeps working unchanged.
21
- function fuseHits(semantic, keyword, alpha, mode) {
22
- if (mode === "semantic") return semantic;
23
- if (mode === "keyword") return keyword.map((h) => ({ ...h, distance: h.bm25 }));
24
- const byId = new Map();
25
- for (const h of semantic) byId.set(h.id, { ...h, _sem: h.distance });
26
- for (const h of keyword) {
27
- const prev = byId.get(h.id);
28
- if (prev) prev._kw = h.bm25;
29
- else byId.set(h.id, { ...h, _kw: h.bm25, distance: h.bm25 });
30
- }
31
- const list = [...byId.values()];
32
- const norm = (vals) => {
33
- const xs = vals.filter((v) => Number.isFinite(v));
34
- if (xs.length === 0) return () => 0.5;
35
- const min = Math.min(...xs); const max = Math.max(...xs); const span = max - min || 1;
36
- return (v) => Number.isFinite(v) ? (v - min) / span : 1;
37
- };
38
- const nSem = norm(list.map((h) => h._sem));
39
- const nKw = norm(list.map((h) => h._kw));
40
- for (const h of list) h.distance = alpha * nSem(h._sem) + (1 - alpha) * nKw(h._kw);
41
- return list;
42
- }
43
-
44
- // KJC-TSK-0484 PR-B — Maximal Marginal Relevance. Given an ordered list of
45
- // candidates (best-first by `score`), pick `topK` that balance relevance
46
- // against intra-result diversity. `lambda=1` collapses to plain relevance;
47
- // `lambda=0` maximises diversity. Cosine sim between candidate embeddings
48
- // drives the redundancy penalty. Candidates without an embedding are kept
49
- // at the end (no penalty applies).
50
- function cosineSim(a, b) {
51
- if (!a || !b || a.length !== b.length) return 0;
52
- let dot = 0; let na = 0; let nb = 0;
53
- for (let i = 0; i < a.length; i += 1) { dot += a[i] * b[i]; na += a[i] * a[i]; nb += b[i] * b[i]; }
54
- const denom = Math.sqrt(na) * Math.sqrt(nb);
55
- return denom === 0 ? 0 : dot / denom;
56
- }
57
-
58
- export function mmrRerank(candidates, embeddings, { topK, lambda = 0.7 }) {
59
- const remaining = candidates.slice();
60
- const picked = [];
61
- const relevance = (c) => 1 / (1 + Math.max(0, c.score ?? c.distance ?? 0));
62
- while (picked.length < topK && remaining.length > 0) {
63
- let bestIdx = 0; let bestVal = -Infinity;
64
- for (let i = 0; i < remaining.length; i += 1) {
65
- const c = remaining[i];
66
- const rel = relevance(c);
67
- let maxSim = 0;
68
- const ce = embeddings.get(c.id);
69
- if (ce && picked.length > 0) {
70
- for (const p of picked) {
71
- const pe = embeddings.get(p.id);
72
- if (pe) maxSim = Math.max(maxSim, cosineSim(ce, pe));
73
- }
74
- }
75
- const score = lambda * rel - (1 - lambda) * maxSim;
76
- if (score > bestVal) { bestVal = score; bestIdx = i; }
77
- }
78
- picked.push(remaining.splice(bestIdx, 1)[0]);
79
- }
80
- return picked;
81
- }
82
-
83
- export async function query(db, embedder, text, { topK = 5, scope = "all", kindBoost = DEFAULT_KIND_BOOST, project = null, mode = "hybrid", alpha = 0.6, where = null, rerankOpts = null, diversify = false, mmrLambda = 0.7 } = {}) {
84
- if (!text || typeof text !== "string") throw new Error("query: text must be a non-empty string");
85
- const fetchK = Math.min(50, topK * 2);
86
- const scopeKind = scope === "plans" ? "plan" : scope === "code" ? "code" : scope === "onboarding" ? "onboarding" : null;
87
- // KJC-TSK-0448 — metadata filter. `where` is a string like "symbol=Foo AND
88
- // hu_id=HU-003"; we parse once and reuse the SQL fragment across both
89
- // semantic and keyword searches.
90
- const parsed = parseWhere(where);
91
- if (!parsed.ok) throw new Error(`query: ${parsed.error}`);
92
- const { sql: whereSql, params: whereParams } = buildWhereSql(parsed.clauses);
93
- const opts = { kind: scopeKind, project, whereSql, whereParams };
94
- const wantSemantic = mode !== "keyword";
95
- const wantKeyword = mode !== "semantic";
96
- const semanticHits = wantSemantic ? searchSimilar(db, await embedder.embed(text), fetchK, opts) : [];
97
- const keywordHits = wantKeyword ? searchBM25(db, text, fetchK, opts) : [];
98
- const raw = fuseHits(semanticHits, keywordHits, alpha, mode);
99
- if (raw.length === 0) return [];
100
- const boostSources = shouldBoostSources(text);
101
- const scored = raw.map((r) => {
102
- const kindB = kindBoost[r.kind] || 0;
103
- const sourceB = (boostSources && r.kind === "code" && !isTestPath(r.source)) ? SOURCE_BOOST_NON_TEST : 0;
104
- return { ...r, score: r.distance - kindB - sourceB };
105
- }).sort((a, b) => a.score - b.score);
106
- // KJC-TSK-0484 PR-B — MMR diversification over topK*2 candidates. Fetches
107
- // embeddings only when requested; cosine between candidate vectors drives
108
- // the redundancy penalty so near-duplicates that survived dedup (e.g. a
109
- // boilerplate chunk repeated under slightly different wording) get pushed
110
- // down. Skipped entirely when `diversify=false` (default).
111
- let reranked;
112
- if (diversify) {
113
- const pool = scored.slice(0, Math.min(scored.length, topK * 2));
114
- const embeddings = getEmbeddingsByIds(db, pool.map((c) => c.id));
115
- reranked = mmrRerank(pool, embeddings, { topK, lambda: mmrLambda });
116
- } else {
117
- reranked = scored.slice(0, topK);
118
- }
119
- // KJC-TSK-0449 — optional cross-encoder rerank. When `rerankOpts` is set,
120
- // we re-score the topK survivors with a (query, passage) cross-encoder;
121
- // the kind+source boost has already been applied, so the rerank acts as
122
- // a finer-grained quality lever on top.
123
- if (rerankOpts) return await rerank(text, reranked, rerankOpts);
124
- return reranked;
125
- }
126
-
127
- export { fuseHits, cosineSim };
1
+ // Shim: retriever now lives in karajan-core/rag (KJC-TSK-0632 PR3).
2
+ export * from "karajan-core/rag/retriever";
@@ -1,54 +1,4 @@
1
- // KJC-TSK-0448 — minimal --where filter parser for metadata-aware RAG
2
- // queries. Grammar (intentionally small):
3
- //
4
- // clause := pair ( "AND" pair )*
5
- // pair := IDENT "=" VALUE
6
- // IDENT := [A-Za-z_][A-Za-z0-9_.-]*
7
- // VALUE := unquoted token | "quoted string"
8
- //
9
- // Examples:
10
- // symbol=loadConfig
11
- // hu_id=HU-003 AND kind=plan
12
- // "headingPath.0"=Architecture
13
- //
14
- // Returns { ok: true, clauses: [["symbol","loadConfig"], ...] } or
15
- // { ok: false, error: "...explanation..." }. The retriever turns each
16
- // clause into a `json_extract(c.metadata, '$.<key>') = ?` SQL fragment;
17
- // the `kind` key is special-cased to filter the column directly.
18
-
19
- const IDENT_RE = /^[A-Za-z_][A-Za-z0-9_.-]*$/;
20
-
21
- export function parseWhere(input) {
22
- if (input == null) return { ok: true, clauses: [] };
23
- if (typeof input !== "string") return { ok: false, error: "where: input must be a string" };
24
- const s = input.trim();
25
- if (!s) return { ok: true, clauses: [] };
26
- const parts = s.split(/\s+AND\s+/i);
27
- const clauses = [];
28
- for (const p of parts) {
29
- const eq = p.indexOf("=");
30
- if (eq <= 0) return { ok: false, error: `where: bad pair "${p}" — expected key=value` };
31
- const key = p.slice(0, eq).trim().replace(/^"|"$/g, "");
32
- let val = p.slice(eq + 1).trim();
33
- if (!IDENT_RE.test(key)) return { ok: false, error: `where: invalid key "${key}"` };
34
- if ((val.startsWith('"') && val.endsWith('"')) || (val.startsWith("'") && val.endsWith("'"))) val = val.slice(1, -1);
35
- if (val.length === 0) return { ok: false, error: `where: empty value for key "${key}"` };
36
- clauses.push([key, val]);
37
- }
38
- return { ok: true, clauses };
39
- }
40
-
41
- /**
42
- * Build a `(sqlFragment, params)` pair to append to a chunks-table query.
43
- * `kind` filters the column directly; everything else goes through json_extract.
44
- */
45
- export function buildWhereSql(clauses) {
46
- if (!clauses?.length) return { sql: "", params: [] };
47
- const fragments = [];
48
- const params = [];
49
- for (const [k, v] of clauses) {
50
- if (k === "kind") { fragments.push("c.kind = ?"); params.push(v); }
51
- else { fragments.push(`json_extract(c.metadata, '$.${k}') = ?`); params.push(v); }
52
- }
53
- return { sql: ` AND ${fragments.join(" AND ")}`, params };
54
- }
1
+ // Shim: where-parser now lives in karajan-core/rag so the hu-board
2
+ // workspace can consume it without a relative dep on the CLI src tree.
3
+ // KJC-TSK-0632 PR2.
4
+ export { parseWhere, buildWhereSql } from "karajan-core/rag/where-parser";
@@ -129,6 +129,10 @@ export class AgentRole extends BaseRole {
129
129
 
130
130
  const runArgs = { prompt, role: this.name };
131
131
  if (onOutput) runArgs.onOutput = onOutput;
132
+ // PAR-H (KJC-TSK-0631): per-call env additions for the agent
133
+ // subprocess (lane slot vars). Agents that ignore task.env simply
134
+ // run without them.
135
+ if (extracted.env) runArgs.env = extracted.env;
132
136
  // Φ1-D (KJC-PCS-0057): roles that build their prompt as a
133
137
  // stable/volatile layout forward both buckets so cache-aware agents
134
138
  // (ClaudeAgent) can ship the stable block as system prompt. Agents
@@ -56,6 +56,8 @@ export class CoderRole extends AgentRole {
56
56
  // PAR-E2 (KJC-TSK-0629): per-call project root. Worktree lanes pass
57
57
  // their lane dir here; the shared role instance keeps its own config.
58
58
  projectDir: input?.projectDir || null,
59
+ // PAR-H (KJC-TSK-0631): per-call subprocess env (lane slot vars).
60
+ env: input?.env || null,
59
61
  onOutput: input?.onOutput || null
60
62
  };
61
63
  }