karajan-code 3.14.1 → 3.15.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +10 -0
- package/package.json +1 -1
- package/packages/hu-board/src/routes/api.js +7 -3
- package/scripts/verify-pack.mjs +5 -1
- package/src/agents/claude-agent.js +9 -4
- package/src/agents/model-registry.js +23 -0
- package/src/config/defaults.js +4 -0
- package/src/hu/acceptance-runner.js +8 -4
- package/src/hu/worktree-bootstrap.js +59 -0
- package/src/orchestrator/drivers/run-hu-batch.js +8 -2
- package/src/orchestrator/hu-sub-pipeline.js +24 -2
- package/src/orchestrator/stages/coder-stage.js +3 -0
- package/src/rag/embedder.js +2 -81
- package/src/rag/embedders/_cloud-base.js +2 -24
- package/src/rag/embedders/cohere.js +2 -28
- package/src/rag/embedders/factory.js +2 -29
- package/src/rag/embedders/mistral.js +2 -26
- package/src/rag/embedders/onnx.js +2 -63
- package/src/rag/embedders/openai.js +2 -22
- package/src/rag/embedders/voyage.js +2 -18
- package/src/rag/rerank.js +4 -74
- package/src/rag/retriever.js +2 -127
- package/src/rag/where-parser.js +4 -54
- package/src/roles/agent-role.js +4 -0
- package/src/roles/coder-role.js +2 -0
package/README.md
CHANGED
|
@@ -383,6 +383,16 @@ Each AI role is executed by the agent you choose:
|
|
|
383
383
|
>
|
|
384
384
|
> Full per-stage reference: [Pipeline roles](https://karajan-code.web.app/docs/handbook/pipeline-roles/) (handbook).
|
|
385
385
|
|
|
386
|
+
## Step mode and parallel lanes (v3.14+)
|
|
387
|
+
|
|
388
|
+
Two ways to control how a plan executes:
|
|
389
|
+
|
|
390
|
+
**`kj run --step`** — supervise the orchestra iteration by iteration. After every iteration the pipeline pauses with a compact report (what happened, the reviewer's must-fix list, what the next iteration will do, spend vs cap) and asks: press Enter to continue, type `stop` to halt (resumable with `kj resume`), or **type instructions** — free text is injected into the feedback the coder reads next iteration, without clobbering the reviewer's own findings. Also offered as a question in the `kj init` wizard (`session.iteration_gate`).
|
|
391
|
+
|
|
392
|
+
**`kj run --plan <id> --parallel <n>`** — run a plan's independent HUs concurrently, each in its own **git worktree** under `.kj/worktrees/<huId>` on branch `kj-hu-<huId>`. The scheduler walks the `blocked_by` graph and only pairs HUs with disjoint `scope` paths (scopeless HUs run alone); the whole lane — coder, acceptance tests, diffs, sonar, final commit — runs inside its worktree while the main working tree stays parked. Governance is built in: default is `1` (fully sequential), a plan-level budget ceiling (`n × max_budget_usd`) stops the batch loudly when exhausted, and SonarQube serializes across lanes. Each fresh worktree is bootstrapped automatically (submodules + `npm ci`, or your `session.worktree_setup` command) and receives `KJ_LANE_SLOT` / `KJ_PORT_OFFSET` env vars so services started by tests don't collide on ports.
|
|
393
|
+
|
|
394
|
+
Full guide: [`docs/parallel-hus.md`](docs/parallel-hus.md).
|
|
395
|
+
|
|
386
396
|
## 5 AI agents supported
|
|
387
397
|
|
|
388
398
|
| Agent | CLI | Install |
|
package/package.json
CHANGED
|
@@ -76,7 +76,9 @@ router.get('/version', (_req, res) => {
|
|
|
76
76
|
*/
|
|
77
77
|
router.get('/standby', async (_req, res) => {
|
|
78
78
|
try {
|
|
79
|
-
|
|
79
|
+
// KJC-TSK-0632: resolve from karajan-core directly — the CLI's
|
|
80
|
+
// src/brain/standby-store.js is just a re-export shim of this.
|
|
81
|
+
const { listPendingStandby } = await import('karajan-core/standby-store');
|
|
80
82
|
const sessions = listPendingStandby();
|
|
81
83
|
res.set('Cache-Control', 'no-store');
|
|
82
84
|
res.json({ sessions });
|
|
@@ -1362,8 +1364,10 @@ router.post('/rag/query', async (req, res) => {
|
|
|
1362
1364
|
}
|
|
1363
1365
|
try {
|
|
1364
1366
|
const { openVecStore, countChunks } = await import('karajan-core/vec-store');
|
|
1365
|
-
|
|
1366
|
-
|
|
1367
|
+
// KJC-TSK-0632: resolved from karajan-core — the board carries zero
|
|
1368
|
+
// relative imports into the CLI src tree (see no-cli-imports test).
|
|
1369
|
+
const { makeEmbedder } = await import('karajan-core/rag/embedders/factory');
|
|
1370
|
+
const { query } = await import('karajan-core/rag/retriever');
|
|
1367
1371
|
const db = openVecStore({ dim: 768 });
|
|
1368
1372
|
try {
|
|
1369
1373
|
if (countChunks(db) === 0) return res.json({ hits: [], empty: true, topK, scope });
|
package/scripts/verify-pack.mjs
CHANGED
|
@@ -166,7 +166,11 @@ try {
|
|
|
166
166
|
console.log(`verify-pack: installing the tarball with pnpm into ${pnpmTmp}…`);
|
|
167
167
|
// pnpm exits non-zero on ERR_PNPM_IGNORED_BUILDS (it skips native build
|
|
168
168
|
// scripts by default) — expected here, so don't treat the exit as failure.
|
|
169
|
-
|
|
169
|
+
// minimum-release-age=0: modern pnpm quarantines freshly published
|
|
170
|
+
// versions (supply-chain protection), so right after publishing
|
|
171
|
+
// karajan-core it silently resolves an OLD one and this smoke fails on
|
|
172
|
+
// missing subpaths. The gate verifies packaging, not release-age policy.
|
|
173
|
+
spawnSync("pnpm", ["add", tgzPath, "--store-dir", path.join(pnpmTmp, ".store"), "--config.minimum-release-age=0"], {
|
|
170
174
|
encoding: "utf8",
|
|
171
175
|
env: childEnv,
|
|
172
176
|
cwd: pnpmTmp,
|
|
@@ -300,7 +300,10 @@ export function createStreamJsonFilter(onOutput) {
|
|
|
300
300
|
*/
|
|
301
301
|
function cleanExecaOpts(extra = {}) {
|
|
302
302
|
const { CLAUDECODE: _CLAUDECODE, ...env } = process.env;
|
|
303
|
-
|
|
303
|
+
// PAR-H (KJC-TSK-0631): extra.env ADDS to the inherited env (lane slot
|
|
304
|
+
// vars) instead of replacing it wholesale.
|
|
305
|
+
const { env: extraEnv, ...rest } = extra;
|
|
306
|
+
return { env: extraEnv ? { ...env, ...extraEnv } : env, stdin: "ignore", ...rest };
|
|
304
307
|
}
|
|
305
308
|
|
|
306
309
|
/**
|
|
@@ -370,7 +373,8 @@ export class ClaudeAgent extends BaseAgent {
|
|
|
370
373
|
const res = await this.runCommand(resolveBin("claude"), args, cleanExecaOpts({
|
|
371
374
|
onOutput: streamFilter,
|
|
372
375
|
silenceTimeoutMs: task.silenceTimeoutMs,
|
|
373
|
-
timeout: task.timeoutMs
|
|
376
|
+
timeout: task.timeoutMs,
|
|
377
|
+
env: task.env
|
|
374
378
|
}));
|
|
375
379
|
const raw = pickOutput(res);
|
|
376
380
|
const output = extractTextFromStreamJson(raw);
|
|
@@ -380,7 +384,7 @@ export class ClaudeAgent extends BaseAgent {
|
|
|
380
384
|
|
|
381
385
|
// Without streaming, use json output to get structured response via stderr
|
|
382
386
|
args.push("--output-format", "json");
|
|
383
|
-
const res = await this.runCommand(resolveBin("claude"), args, cleanExecaOpts());
|
|
387
|
+
const res = await this.runCommand(resolveBin("claude"), args, cleanExecaOpts({ env: task.env }));
|
|
384
388
|
const raw = pickOutput(res);
|
|
385
389
|
const output = extractTextFromStreamJson(raw);
|
|
386
390
|
const usage = extractUsageFromStreamJson(raw);
|
|
@@ -393,7 +397,8 @@ export class ClaudeAgent extends BaseAgent {
|
|
|
393
397
|
const res = await this.runCommand(resolveBin("claude"), args, cleanExecaOpts({
|
|
394
398
|
onOutput: task.onOutput,
|
|
395
399
|
silenceTimeoutMs: task.silenceTimeoutMs,
|
|
396
|
-
timeout: task.timeoutMs
|
|
400
|
+
timeout: task.timeoutMs,
|
|
401
|
+
env: task.env
|
|
397
402
|
}));
|
|
398
403
|
const raw = pickOutput(res);
|
|
399
404
|
const usage = extractUsageFromStreamJson(raw);
|
|
@@ -108,6 +108,29 @@ registerModelAlias("gemini", "gemini-2.5-pro");
|
|
|
108
108
|
registerModel("aider", { provider: "aider", pricing: { input_per_million: 3, output_per_million: 15 } });
|
|
109
109
|
registerModel("opencode", { provider: "opencode", pricing: { input_per_million: 0, output_per_million: 0 } });
|
|
110
110
|
|
|
111
|
+
/**
|
|
112
|
+
* Moonshot Kimi Family (KJC-TSK-0633) — consumed through OpenCode as an
|
|
113
|
+
* OpenAI-compatible provider (docs/providers-via-opencode.md).
|
|
114
|
+
* Pricing: https://platform.moonshot.ai/docs/pricing — verify before
|
|
115
|
+
* trusting for billing decisions; ids/prices move fast.
|
|
116
|
+
*/
|
|
117
|
+
registerModel("kimi-k2", { provider: "moonshot", pricing: { input_per_million: 0.6, output_per_million: 2.5 } });
|
|
118
|
+
registerModel("kimi-k2-thinking", { provider: "moonshot", pricing: { input_per_million: 0.6, output_per_million: 2.5 } });
|
|
119
|
+
registerModelAlias("kimi", "kimi-k2");
|
|
120
|
+
// Prefixed ids as the documented opencode.json snippet produces them.
|
|
121
|
+
registerModelAlias("kimi/kimi-k2", "kimi-k2");
|
|
122
|
+
registerModelAlias("kimi/kimi-k2-thinking", "kimi-k2-thinking");
|
|
123
|
+
|
|
124
|
+
/**
|
|
125
|
+
* DeepSeek Family (KJC-TSK-0633) — same OpenCode route.
|
|
126
|
+
* Pricing: https://api-docs.deepseek.com/quick_start/pricing
|
|
127
|
+
*/
|
|
128
|
+
registerModel("deepseek-chat", { provider: "deepseek", pricing: { input_per_million: 0.28, output_per_million: 0.42 } });
|
|
129
|
+
registerModel("deepseek-reasoner", { provider: "deepseek", pricing: { input_per_million: 0.28, output_per_million: 0.42 } });
|
|
130
|
+
registerModelAlias("deepseek", "deepseek-chat");
|
|
131
|
+
registerModelAlias("deepseek/deepseek-chat", "deepseek-chat");
|
|
132
|
+
registerModelAlias("deepseek/deepseek-reasoner", "deepseek-reasoner");
|
|
133
|
+
|
|
111
134
|
// Common CLI Aliases (with provider overrides)
|
|
112
135
|
registerModelAlias("aider/claude-3-7-sonnet", "claude-sonnet-4.6", { provider: "aider" });
|
|
113
136
|
registerModelAlias("aider/gpt-4o", "gpt-5.4-standard", { provider: "aider" });
|
package/src/config/defaults.js
CHANGED
|
@@ -182,6 +182,10 @@ const DEFAULTS = {
|
|
|
182
182
|
// Concurrent HU lanes per plan run (KJC-TSK-0626). 1 = sequential.
|
|
183
183
|
// Raising it multiplies token burn rate — the plan budget scales with it.
|
|
184
184
|
max_parallel_hus: 1,
|
|
185
|
+
// Command run inside each fresh lane worktree before the coder starts
|
|
186
|
+
// (KJC-TSK-0630). null = auto-detect: `npm ci` when package-lock.json
|
|
187
|
+
// exists, nothing otherwise. Submodules are always initialized first.
|
|
188
|
+
worktree_setup: null,
|
|
185
189
|
max_iteration_minutes: 30,
|
|
186
190
|
max_total_minutes: 120,
|
|
187
191
|
max_planner_minutes: 60,
|
|
@@ -12,11 +12,15 @@ import { runCommand } from "../utils/process.js";
|
|
|
12
12
|
* @param {number} [timeoutMs=30000] - Timeout per test
|
|
13
13
|
* @returns {Promise<{cmd: string, passed: boolean, output: string, exitCode: number}>}
|
|
14
14
|
*/
|
|
15
|
-
async function runSingleTest(cmd, cwd, timeoutMs = 30000) {
|
|
15
|
+
async function runSingleTest(cmd, cwd, timeoutMs = 30000, env = null) {
|
|
16
16
|
try {
|
|
17
17
|
const result = await runCommand("bash", ["-c", cmd], {
|
|
18
18
|
timeout: timeoutMs,
|
|
19
|
-
cwd
|
|
19
|
+
cwd,
|
|
20
|
+
// PAR-H (KJC-TSK-0631): lane env (KJ_LANE_SLOT / KJ_PORT_OFFSET) so
|
|
21
|
+
// tests that start services can offset their ports. execa merges
|
|
22
|
+
// this on top of process.env.
|
|
23
|
+
...(env ? { env } : {})
|
|
20
24
|
});
|
|
21
25
|
const output = (result.stdout || "") + (result.stderr || "");
|
|
22
26
|
return {
|
|
@@ -51,7 +55,7 @@ async function runSingleTest(cmd, cwd, timeoutMs = 30000) {
|
|
|
51
55
|
* @param {string} cwd - Working directory
|
|
52
56
|
* @returns {Promise<{allPassed: boolean, results: object[], summary: string, diagnostics: string|null, pending: number}>}
|
|
53
57
|
*/
|
|
54
|
-
export async function runAcceptanceTests(tests, cwd) {
|
|
58
|
+
export async function runAcceptanceTests(tests, cwd, { env = null } = {}) {
|
|
55
59
|
if (!tests || tests.length === 0) {
|
|
56
60
|
return { allPassed: false, results: [], summary: "No acceptance tests defined", diagnostics: null, pending: 0 };
|
|
57
61
|
}
|
|
@@ -86,7 +90,7 @@ export async function runAcceptanceTests(tests, cwd) {
|
|
|
86
90
|
});
|
|
87
91
|
continue;
|
|
88
92
|
}
|
|
89
|
-
const result = await runSingleTest(cmd, cwd);
|
|
93
|
+
const result = await runSingleTest(cmd, cwd, 30000, env);
|
|
90
94
|
results.push({ ...result, type: "shell" });
|
|
91
95
|
}
|
|
92
96
|
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* PAR-G (KJC-TSK-0630): make a freshly created lane worktree operative.
|
|
3
|
+
*
|
|
4
|
+
* `git worktree add` produces a clean checkout: no node_modules, no
|
|
5
|
+
* initialized submodules — and a container/bind-mounted process cannot
|
|
6
|
+
* init them itself because the worktree's real .git lives in the parent
|
|
7
|
+
* repo (gotcha reported by Jorge del Casar's worktree-docker-envs skill).
|
|
8
|
+
* Without this step, --parallel lanes die on the first `npm test` in any
|
|
9
|
+
* real project.
|
|
10
|
+
*
|
|
11
|
+
* Best-effort by contract: a failed or slow bootstrap warns and the lane
|
|
12
|
+
* continues — the acceptance tests deliver the real verdict.
|
|
13
|
+
*/
|
|
14
|
+
import { existsSync } from "node:fs";
|
|
15
|
+
import { join } from "node:path";
|
|
16
|
+
import { runCommand } from "../utils/process.js";
|
|
17
|
+
|
|
18
|
+
const STEP_TIMEOUT_MS = 5 * 60 * 1000;
|
|
19
|
+
|
|
20
|
+
/**
|
|
21
|
+
* @param {object} params
|
|
22
|
+
* @param {string} params.worktreePath Absolute path of the lane worktree.
|
|
23
|
+
* @param {string|null} [params.setupCommand] session.worktree_setup — wins over auto-detect.
|
|
24
|
+
* @param {object|null} [params.logger]
|
|
25
|
+
* @param {number} [params.timeoutMs]
|
|
26
|
+
* @param {Function} [params.run] Injectable runner (tests).
|
|
27
|
+
* @returns {Promise<{ok: boolean, steps: string[], warnings: string[]}>}
|
|
28
|
+
*/
|
|
29
|
+
export async function bootstrapWorktree({ worktreePath, setupCommand = null, logger = null, timeoutMs = STEP_TIMEOUT_MS, run = runCommand }) {
|
|
30
|
+
const warnings = [];
|
|
31
|
+
const steps = [];
|
|
32
|
+
|
|
33
|
+
if (existsSync(join(worktreePath, ".gitmodules"))) {
|
|
34
|
+
steps.push({ name: "submodules", cmd: "git", args: ["submodule", "update", "--init", "--recursive"] });
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
const explicit = typeof setupCommand === "string" && setupCommand.trim() ? setupCommand.trim() : null;
|
|
38
|
+
if (explicit) {
|
|
39
|
+
steps.push({ name: "setup", cmd: "sh", args: ["-c", explicit] });
|
|
40
|
+
} else if (existsSync(join(worktreePath, "package-lock.json"))) {
|
|
41
|
+
steps.push({ name: "deps", cmd: "npm", args: ["ci", "--no-audit", "--no-fund"] });
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
for (const step of steps) {
|
|
45
|
+
try {
|
|
46
|
+
const res = await run(step.cmd, step.args, { cwd: worktreePath, timeout: timeoutMs });
|
|
47
|
+
if (res.exitCode !== 0) {
|
|
48
|
+
warnings.push(`${step.name} failed (exit ${res.exitCode}): ${String(res.stderr || res.stdout || "").slice(0, 200)}`);
|
|
49
|
+
} else {
|
|
50
|
+
logger?.info?.(`worktree bootstrap: ${step.name} ok`);
|
|
51
|
+
}
|
|
52
|
+
} catch (err) {
|
|
53
|
+
warnings.push(`${step.name} threw: ${err.message}`);
|
|
54
|
+
}
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
for (const w of warnings) logger?.warn?.(`worktree bootstrap: ${w} — lane continues`);
|
|
58
|
+
return { ok: warnings.length === 0, steps: steps.map((s) => s.name), warnings };
|
|
59
|
+
}
|
|
@@ -104,7 +104,13 @@ export async function runHuBatch({ ctx, task, askQuestion, emitter, logger }) {
|
|
|
104
104
|
// left it clamped to hu_max_iterations for the rest of the run.
|
|
105
105
|
const worktreePath = laneOpts?.worktreePath || null;
|
|
106
106
|
const projectDir = worktreePath || ctx.config.projectDir || process.cwd();
|
|
107
|
-
|
|
107
|
+
// PAR-H (KJC-TSK-0631): the lane's slot travels as env vars so any
|
|
108
|
+
// service the coder or the acceptance tests start can offset its
|
|
109
|
+
// ports (offset = slot × 100; the project applies its own base).
|
|
110
|
+
const laneEnv = Number.isInteger(laneOpts?.laneSlot)
|
|
111
|
+
? { KJ_LANE_SLOT: String(laneOpts.laneSlot), KJ_PORT_OFFSET: String(laneOpts.laneSlot * 100) }
|
|
112
|
+
: null;
|
|
113
|
+
const laneConfig = { ...ctx.config, projectDir, max_iterations: huMaxIterations, lane_env: laneEnv };
|
|
108
114
|
if (!huPolicies.tdd) laneConfig.development = { ...laneConfig.development, methodology: "standard", require_test_changes: false };
|
|
109
115
|
if (!huPolicies.sonar) laneConfig.sonarqube = { ...laneConfig.sonarqube, enabled: false };
|
|
110
116
|
const laneFlags = { ...ctx.pipelineFlags };
|
|
@@ -233,7 +239,7 @@ export async function runHuBatch({ ctx, task, askQuestion, emitter, logger }) {
|
|
|
233
239
|
detail: { huId: story.id, testCount: story.acceptance_tests.length }
|
|
234
240
|
}));
|
|
235
241
|
|
|
236
|
-
const testResult = await runAcceptanceTests(story.acceptance_tests, projectDir);
|
|
242
|
+
const testResult = await runAcceptanceTests(story.acceptance_tests, projectDir, { env: laneEnv });
|
|
237
243
|
emitProgress(emitter, makeEvent("hu:acceptance-end", { ...ctx.eventBase, stage: "acceptance" }, {
|
|
238
244
|
status: testResult.allPassed ? "ok" : "fail",
|
|
239
245
|
message: testResult.summary,
|
|
@@ -6,7 +6,11 @@ import { topologicalSort } from "../hu/graph.js";
|
|
|
6
6
|
import { updateStoryStatus, loadHuBatch, saveHuBatch, HU_STATUS } from "../hu/store.js";
|
|
7
7
|
import { emitProgress, makeEvent } from "../utils/events.js";
|
|
8
8
|
import { refineHuWithContext } from "../hu/lazy-planner.js";
|
|
9
|
+
import { join } from "node:path";
|
|
10
|
+
import { acquireSlot, releaseSlot } from "karajan-core/slot-registry";
|
|
11
|
+
import { getKarajanHome } from "karajan-core/paths";
|
|
9
12
|
import { findParallelGroups, createWorktree, mergeWorktree, removeWorktree } from "../hu/parallel-executor.js";
|
|
13
|
+
import { bootstrapWorktree } from "../hu/worktree-bootstrap.js";
|
|
10
14
|
import { partitionConflictFree } from "./hu-scheduler.js";
|
|
11
15
|
import { createParallelLimiter, planBudgetUsd } from "./parallel-limiter.js";
|
|
12
16
|
|
|
@@ -200,7 +204,7 @@ function buildHuOutcome({ story: _story, iterResult, status, startedAt, huBudget
|
|
|
200
204
|
* @param {object} params
|
|
201
205
|
* @returns {Promise<{huId: string, approved: boolean, result?: object, error?: string, blockedDependents?: string[]}>}
|
|
202
206
|
*/
|
|
203
|
-
async function runSingleHu({ storyId, batch, batchSessionId, runIterationFn, emitter, eventBase, logger, config, results, worktreePath, onStatusChange = null, onOutcome = null, budgetTracker = null }) {
|
|
207
|
+
async function runSingleHu({ storyId, batch, batchSessionId, runIterationFn, emitter, eventBase, logger, config, results, worktreePath, laneSlot = null, onStatusChange = null, onOutcome = null, budgetTracker = null }) {
|
|
204
208
|
// PR1 (live HU status): every saveHuBatch should also notify the
|
|
205
209
|
// plan JSON so the board reflects state in real time. Defined as a
|
|
206
210
|
// local helper so we don't repeat the try/null-check boilerplate.
|
|
@@ -287,7 +291,7 @@ async function runSingleHu({ storyId, batch, batchSessionId, runIterationFn, emi
|
|
|
287
291
|
try {
|
|
288
292
|
// PAR-E2 (KJC-TSK-0629): lanes handed a worktree must aim every git and
|
|
289
293
|
// filesystem touchpoint at it — laneOpts carries that path to the runner.
|
|
290
|
-
const iterResult = await runIterationFn(huTask, story, { worktreePath: worktreePath || null });
|
|
294
|
+
const iterResult = await runIterationFn(huTask, story, { worktreePath: worktreePath || null, laneSlot });
|
|
291
295
|
const approved = Boolean(iterResult?.approved);
|
|
292
296
|
|
|
293
297
|
// --- Transition to reviewing (post-coder, pre-reviewer evaluation) ---
|
|
@@ -527,12 +531,27 @@ export async function runHuSubPipeline({ huReviewerResult, runIterationFn, emitt
|
|
|
527
531
|
// Multiple HUs: create worktrees, run in parallel
|
|
528
532
|
const projectDir = config?.projectDir || process.cwd();
|
|
529
533
|
const worktrees = new Map();
|
|
534
|
+
// PAR-H (KJC-TSK-0631): every lane gets a stable numeric slot so
|
|
535
|
+
// services it starts can offset their ports (KJ_LANE_SLOT /
|
|
536
|
+
// KJ_PORT_OFFSET reach the coder and the acceptance tests).
|
|
537
|
+
const slotRegistryPath = join(getKarajanHome(), "worktree-slots.json");
|
|
538
|
+
const laneSlots = new Map();
|
|
530
539
|
|
|
531
540
|
// Create worktrees for each HU in the batch
|
|
532
541
|
for (const id of runnableIds) {
|
|
533
542
|
try {
|
|
534
543
|
const wtPath = await createWorktree(projectDir, id);
|
|
535
544
|
worktrees.set(id, wtPath);
|
|
545
|
+
// PAR-G (KJC-TSK-0630): a fresh worktree has no node_modules and
|
|
546
|
+
// no initialized submodules — make the lane operative before the
|
|
547
|
+
// coder lands. Best-effort: warnings never block the lane.
|
|
548
|
+
await bootstrapWorktree({ worktreePath: wtPath, setupCommand: config?.session?.worktree_setup || null, logger });
|
|
549
|
+
try {
|
|
550
|
+
const { slot } = await acquireSlot({ registryPath: slotRegistryPath, id: `${projectDir}::${id}` });
|
|
551
|
+
laneSlots.set(id, slot);
|
|
552
|
+
} catch (err) {
|
|
553
|
+
logger.warn(`Failed to acquire lane slot for HU ${id}: ${err.message} — lane runs without port offset`);
|
|
554
|
+
}
|
|
536
555
|
} catch (err) {
|
|
537
556
|
logger.warn(`Failed to create worktree for HU ${id}: ${err.message} — will run sequentially`);
|
|
538
557
|
}
|
|
@@ -547,6 +566,7 @@ export async function runHuSubPipeline({ huReviewerResult, runIterationFn, emitt
|
|
|
547
566
|
storyId, batch, batchSessionId, runIterationFn,
|
|
548
567
|
emitter, eventBase, logger, config, results,
|
|
549
568
|
worktreePath: worktrees.get(storyId),
|
|
569
|
+
laneSlot: laneSlots.get(storyId) ?? null,
|
|
550
570
|
onStatusChange, onOutcome, budgetTracker
|
|
551
571
|
});
|
|
552
572
|
} finally {
|
|
@@ -562,6 +582,7 @@ export async function runHuSubPipeline({ huReviewerResult, runIterationFn, emitt
|
|
|
562
582
|
if (res.approved && worktrees.has(res.huId)) {
|
|
563
583
|
try {
|
|
564
584
|
await mergeWorktree(projectDir, res.huId);
|
|
585
|
+
await releaseSlot({ registryPath: slotRegistryPath, id: `${projectDir}::${res.huId}` }).catch(() => {});
|
|
565
586
|
} catch (err) {
|
|
566
587
|
logger.warn(`Failed to merge worktree for HU ${res.huId}: ${err.message}`);
|
|
567
588
|
}
|
|
@@ -581,6 +602,7 @@ export async function runHuSubPipeline({ huReviewerResult, runIterationFn, emitt
|
|
|
581
602
|
// Clean up failed worktree
|
|
582
603
|
if (worktrees.has(res.huId)) {
|
|
583
604
|
try { await removeWorktree(projectDir, res.huId); } catch { /* ignore */ }
|
|
605
|
+
await releaseSlot({ registryPath: slotRegistryPath, id: `${projectDir}::${res.huId}` }).catch(() => {});
|
|
584
606
|
}
|
|
585
607
|
}
|
|
586
608
|
}
|
|
@@ -71,6 +71,9 @@ export async function runCoderStage({ coderRoleInstance, coderRole, config, logg
|
|
|
71
71
|
// PAR-E2 (KJC-TSK-0629): the stage's config wins over the role's own —
|
|
72
72
|
// worktree lanes pass a laneConfig whose projectDir is the worktree.
|
|
73
73
|
projectDir: config?.projectDir || null,
|
|
74
|
+
// PAR-H (KJC-TSK-0631): lane env (KJ_LANE_SLOT / KJ_PORT_OFFSET)
|
|
75
|
+
// reaches the coder subprocess so services it starts don't collide.
|
|
76
|
+
env: config?.lane_env || null,
|
|
74
77
|
onOutput: coderStall.onOutput,
|
|
75
78
|
// Lets Brain Recovery persist a standby snapshot if the coder's
|
|
76
79
|
// provider hits a quota cap mid-run (KJC hibernation wiring).
|
package/src/rag/embedder.js
CHANGED
|
@@ -1,81 +1,2 @@
|
|
|
1
|
-
//
|
|
2
|
-
|
|
3
|
-
// Float32Array vectors. Cero deps externas; usa fetch global.
|
|
4
|
-
//
|
|
5
|
-
// Defaults:
|
|
6
|
-
// url = process.env.KJ_OLLAMA_URL or "http://localhost:11434"
|
|
7
|
-
// model = process.env.KJ_OLLAMA_EMBED_MODEL or "nomic-embed-text"
|
|
8
|
-
// dim = 768 (matches nomic-embed-text)
|
|
9
|
-
// timeoutMs = 30000
|
|
10
|
-
|
|
11
|
-
const DEFAULT_URL = "http://localhost:11434";
|
|
12
|
-
const DEFAULT_MODEL = "nomic-embed-text";
|
|
13
|
-
const DEFAULT_DIM = 768;
|
|
14
|
-
const DEFAULT_TIMEOUT_MS = 30000;
|
|
15
|
-
|
|
16
|
-
export class OllamaEmbedderError extends Error {
|
|
17
|
-
constructor(message, { cause, status } = {}) {
|
|
18
|
-
super(message);
|
|
19
|
-
this.name = "OllamaEmbedderError";
|
|
20
|
-
if (cause) this.cause = cause;
|
|
21
|
-
if (status != null) this.status = status;
|
|
22
|
-
}
|
|
23
|
-
}
|
|
24
|
-
|
|
25
|
-
export class OllamaEmbedder {
|
|
26
|
-
constructor({
|
|
27
|
-
url = process.env.KJ_OLLAMA_URL || DEFAULT_URL,
|
|
28
|
-
model = process.env.KJ_OLLAMA_EMBED_MODEL || DEFAULT_MODEL,
|
|
29
|
-
dim = DEFAULT_DIM,
|
|
30
|
-
timeoutMs = DEFAULT_TIMEOUT_MS,
|
|
31
|
-
fetchFn = globalThis.fetch,
|
|
32
|
-
} = {}) {
|
|
33
|
-
this.url = url.replace(/\/$/, "");
|
|
34
|
-
this.model = model;
|
|
35
|
-
this.dim = dim;
|
|
36
|
-
this.timeoutMs = timeoutMs;
|
|
37
|
-
this.fetch = fetchFn;
|
|
38
|
-
}
|
|
39
|
-
|
|
40
|
-
/** Single-text embedding. Returns Float32Array(dim). */
|
|
41
|
-
async embed(text) {
|
|
42
|
-
if (typeof text !== "string" || text.length === 0) {
|
|
43
|
-
throw new OllamaEmbedderError("embed: text must be a non-empty string");
|
|
44
|
-
}
|
|
45
|
-
const ctrl = new AbortController();
|
|
46
|
-
const timer = setTimeout(() => ctrl.abort(), this.timeoutMs);
|
|
47
|
-
let res;
|
|
48
|
-
try {
|
|
49
|
-
res = await this.fetch(`${this.url}/api/embeddings`, {
|
|
50
|
-
method: "POST",
|
|
51
|
-
headers: { "Content-Type": "application/json" },
|
|
52
|
-
body: JSON.stringify({ model: this.model, prompt: text }),
|
|
53
|
-
signal: ctrl.signal,
|
|
54
|
-
});
|
|
55
|
-
} catch (err) {
|
|
56
|
-
throw new OllamaEmbedderError(`Ollama embed request failed (${this.url}): ${err.message}`, { cause: err });
|
|
57
|
-
} finally {
|
|
58
|
-
clearTimeout(timer);
|
|
59
|
-
}
|
|
60
|
-
if (!res.ok) {
|
|
61
|
-
throw new OllamaEmbedderError(`Ollama embed HTTP ${res.status} from ${this.url}`, { status: res.status });
|
|
62
|
-
}
|
|
63
|
-
const body = await res.json();
|
|
64
|
-
const arr = body?.embedding;
|
|
65
|
-
if (!Array.isArray(arr) || arr.length === 0) {
|
|
66
|
-
throw new OllamaEmbedderError("Ollama response missing 'embedding' array");
|
|
67
|
-
}
|
|
68
|
-
if (arr.length !== this.dim) {
|
|
69
|
-
throw new OllamaEmbedderError(`Ollama dim mismatch: got ${arr.length}, expected ${this.dim} for model ${this.model}`);
|
|
70
|
-
}
|
|
71
|
-
return Float32Array.from(arr);
|
|
72
|
-
}
|
|
73
|
-
|
|
74
|
-
/** Batch embedding (sequential — Ollama /api/embeddings is single-text). */
|
|
75
|
-
async embedBatch(texts) {
|
|
76
|
-
if (!Array.isArray(texts)) throw new OllamaEmbedderError("embedBatch: texts must be an array");
|
|
77
|
-
const out = [];
|
|
78
|
-
for (const t of texts) out.push(await this.embed(t));
|
|
79
|
-
return out;
|
|
80
|
-
}
|
|
81
|
-
}
|
|
1
|
+
// Shim: embedder now lives in karajan-core/rag (KJC-TSK-0632 PR3).
|
|
2
|
+
export * from "karajan-core/rag/embedder";
|
|
@@ -1,24 +1,2 @@
|
|
|
1
|
-
//
|
|
2
|
-
|
|
3
|
-
// extracted by caller, dim validated against the adapter's expected size.
|
|
4
|
-
export async function cloudEmbed(adapter, text, ErrCls, { provider, body, extract }) {
|
|
5
|
-
if (typeof text !== "string" || text.length === 0) throw new ErrCls("embed: text must be a non-empty string");
|
|
6
|
-
const ctrl = new AbortController();
|
|
7
|
-
const timer = setTimeout(() => ctrl.abort(), adapter.timeoutMs);
|
|
8
|
-
let res;
|
|
9
|
-
try {
|
|
10
|
-
res = await adapter.fetch(adapter.url, {
|
|
11
|
-
method: "POST",
|
|
12
|
-
headers: { "Content-Type": "application/json", Authorization: `Bearer ${adapter.apiKey}` },
|
|
13
|
-
body: JSON.stringify(body),
|
|
14
|
-
signal: ctrl.signal,
|
|
15
|
-
});
|
|
16
|
-
} catch (err) { throw new ErrCls(`${provider} embed request failed: ${err.message}`, { cause: err }); }
|
|
17
|
-
finally { clearTimeout(timer); }
|
|
18
|
-
if (!res.ok) throw new ErrCls(`${provider} embed HTTP ${res.status}`, { status: res.status });
|
|
19
|
-
const json = await res.json();
|
|
20
|
-
const arr = extract(json);
|
|
21
|
-
if (!Array.isArray(arr) || arr.length === 0) throw new ErrCls(`${provider} response missing embedding`);
|
|
22
|
-
if (arr.length !== adapter.dim) throw new ErrCls(`${provider} dim mismatch: got ${arr.length}, expected ${adapter.dim} for model ${adapter.model}`);
|
|
23
|
-
return Float32Array.from(arr);
|
|
24
|
-
}
|
|
1
|
+
// Shim: moved to karajan-core/rag/embedders (KJC-TSK-0632 PR3).
|
|
2
|
+
export * from "karajan-core/rag/embedders/_cloud-base";
|
|
@@ -1,28 +1,2 @@
|
|
|
1
|
-
// KJC-TSK-
|
|
2
|
-
|
|
3
|
-
// is requested, or { embeddings: [[...]] } in legacy v1. We extract the
|
|
4
|
-
// modern shape with a v1 fallback.
|
|
5
|
-
import { cloudEmbed } from "./_cloud-base.js";
|
|
6
|
-
|
|
7
|
-
const DEFAULTS = { url: "https://api.cohere.com/v2/embed", model: "embed-multilingual-v3.0", dim: 1024, timeoutMs: 30000 };
|
|
8
|
-
|
|
9
|
-
export class CohereEmbedderError extends Error {
|
|
10
|
-
constructor(message, { cause, status } = {}) { super(message); this.name = "CohereEmbedderError"; if (cause) this.cause = cause; if (status != null) this.status = status; }
|
|
11
|
-
}
|
|
12
|
-
|
|
13
|
-
export class CohereEmbedder {
|
|
14
|
-
constructor({ url = DEFAULTS.url, apiKey = process.env.KJ_COHERE_KEY, model = process.env.KJ_COHERE_EMBED_MODEL || DEFAULTS.model, dim = DEFAULTS.dim, timeoutMs = DEFAULTS.timeoutMs, fetchFn = globalThis.fetch } = {}) {
|
|
15
|
-
// KJC architecture invariant: scoped env var (KJ_COHERE_KEY), not the
|
|
16
|
-
// generic COHERE_API_KEY — same rule as OpenAI/Voyage adapters.
|
|
17
|
-
if (!apiKey) throw new CohereEmbedderError("Cohere embedder requires an api_key (config.rag.embedder.api_key or KJ_COHERE_KEY env)");
|
|
18
|
-
Object.assign(this, { url, apiKey, model, dim, timeoutMs, fetch: fetchFn });
|
|
19
|
-
}
|
|
20
|
-
async embed(text) {
|
|
21
|
-
return cloudEmbed(this, text, CohereEmbedderError, {
|
|
22
|
-
provider: "Cohere",
|
|
23
|
-
body: { model: this.model, texts: [text], input_type: "search_document", embedding_types: ["float"] },
|
|
24
|
-
extract: (b) => b?.embeddings?.float?.[0] || b?.embeddings?.[0],
|
|
25
|
-
});
|
|
26
|
-
}
|
|
27
|
-
async embedBatch(texts) { if (!Array.isArray(texts)) throw new CohereEmbedderError("embedBatch: texts must be an array"); const out = []; for (const t of texts) out.push(await this.embed(t)); return out; }
|
|
28
|
-
}
|
|
1
|
+
// Shim: moved to karajan-core/rag/embedders (KJC-TSK-0632 PR3).
|
|
2
|
+
export * from "karajan-core/rag/embedders/cohere";
|
|
@@ -1,29 +1,2 @@
|
|
|
1
|
-
//
|
|
2
|
-
|
|
3
|
-
// default ollama). Each provider has its own dim default; the factory wires
|
|
4
|
-
// it so the caller does not need to know.
|
|
5
|
-
import { OllamaEmbedder } from "../embedder.js";
|
|
6
|
-
import { OpenAIEmbedder } from "./openai.js";
|
|
7
|
-
import { VoyageEmbedder } from "./voyage.js";
|
|
8
|
-
import { CohereEmbedder } from "./cohere.js";
|
|
9
|
-
import { MistralEmbedder } from "./mistral.js";
|
|
10
|
-
import { ONNXEmbedder } from "./onnx.js";
|
|
11
|
-
|
|
12
|
-
const PROVIDERS = {
|
|
13
|
-
ollama: { cls: OllamaEmbedder, dim: 768 },
|
|
14
|
-
openai: { cls: OpenAIEmbedder, dim: 1536 },
|
|
15
|
-
voyage: { cls: VoyageEmbedder, dim: 1024 },
|
|
16
|
-
cohere: { cls: CohereEmbedder, dim: 1024 },
|
|
17
|
-
mistral: { cls: MistralEmbedder, dim: 1024 },
|
|
18
|
-
onnx: { cls: ONNXEmbedder, dim: 384 },
|
|
19
|
-
};
|
|
20
|
-
|
|
21
|
-
export function makeEmbedder(config = {}) {
|
|
22
|
-
const cfg = config?.rag?.embedder || {};
|
|
23
|
-
const provider = cfg.provider || "ollama";
|
|
24
|
-
const spec = PROVIDERS[provider];
|
|
25
|
-
if (!spec) throw new Error(`Unknown embedder provider: ${provider}. Supported: ${Object.keys(PROVIDERS).join(", ")}`);
|
|
26
|
-
const dim = cfg.dim || spec.dim;
|
|
27
|
-
return new spec.cls({ url: cfg.url, apiKey: cfg.api_key, model: cfg.model, dim, timeoutMs: cfg.timeout_ms });
|
|
28
|
-
}
|
|
29
|
-
|
|
1
|
+
// Shim: moved to karajan-core/rag/embedders (KJC-TSK-0632 PR3).
|
|
2
|
+
export * from "karajan-core/rag/embedders/factory";
|
|
@@ -1,26 +1,2 @@
|
|
|
1
|
-
// KJC-TSK-
|
|
2
|
-
|
|
3
|
-
// endpoints. Single model today (`mistral-embed`, 1024 dim).
|
|
4
|
-
import { cloudEmbed } from "./_cloud-base.js";
|
|
5
|
-
|
|
6
|
-
const DEFAULTS = { url: "https://api.mistral.ai/v1/embeddings", model: "mistral-embed", dim: 1024, timeoutMs: 30000 };
|
|
7
|
-
|
|
8
|
-
export class MistralEmbedderError extends Error {
|
|
9
|
-
constructor(message, { cause, status } = {}) { super(message); this.name = "MistralEmbedderError"; if (cause) this.cause = cause; if (status != null) this.status = status; }
|
|
10
|
-
}
|
|
11
|
-
|
|
12
|
-
export class MistralEmbedder {
|
|
13
|
-
constructor({ url = DEFAULTS.url, apiKey = process.env.KJ_MISTRAL_KEY, model = process.env.KJ_MISTRAL_EMBED_MODEL || DEFAULTS.model, dim = DEFAULTS.dim, timeoutMs = DEFAULTS.timeoutMs, fetchFn = globalThis.fetch } = {}) {
|
|
14
|
-
// KJC architecture invariant: scoped env var (KJ_MISTRAL_KEY).
|
|
15
|
-
if (!apiKey) throw new MistralEmbedderError("Mistral embedder requires an api_key (config.rag.embedder.api_key or KJ_MISTRAL_KEY env)");
|
|
16
|
-
Object.assign(this, { url, apiKey, model, dim, timeoutMs, fetch: fetchFn });
|
|
17
|
-
}
|
|
18
|
-
async embed(text) {
|
|
19
|
-
return cloudEmbed(this, text, MistralEmbedderError, {
|
|
20
|
-
provider: "Mistral",
|
|
21
|
-
body: { model: this.model, input: [text] },
|
|
22
|
-
extract: (b) => b?.data?.[0]?.embedding,
|
|
23
|
-
});
|
|
24
|
-
}
|
|
25
|
-
async embedBatch(texts) { if (!Array.isArray(texts)) throw new MistralEmbedderError("embedBatch: texts must be an array"); const out = []; for (const t of texts) out.push(await this.embed(t)); return out; }
|
|
26
|
-
}
|
|
1
|
+
// Shim: moved to karajan-core/rag/embedders (KJC-TSK-0632 PR3).
|
|
2
|
+
export * from "karajan-core/rag/embedders/mistral";
|
|
@@ -1,63 +1,2 @@
|
|
|
1
|
-
// KJC-TSK-
|
|
2
|
-
|
|
3
|
-
// directly in Node — no API key, no Docker, no Ollama.
|
|
4
|
-
//
|
|
5
|
-
// First call downloads the model weights to ~/.cache/huggingface/ (~80 MB
|
|
6
|
-
// for the default MiniLM). Subsequent calls reuse the cache, so the steady
|
|
7
|
-
// state is fully offline.
|
|
8
|
-
//
|
|
9
|
-
// Default model: `Xenova/all-MiniLM-L6-v2` (384 dim, ~80 MB, fast).
|
|
10
|
-
// Higher-quality alternative: `Xenova/jina-embeddings-v2-base-en`
|
|
11
|
-
// (768 dim, ~260 MB, slower but better for retrieval).
|
|
12
|
-
|
|
13
|
-
const DEFAULTS = { model: "Xenova/all-MiniLM-L6-v2", dim: 384, pooling: "mean", normalize: true };
|
|
14
|
-
|
|
15
|
-
export class ONNXEmbedderError extends Error {
|
|
16
|
-
constructor(message, { cause } = {}) { super(message); this.name = "ONNXEmbedderError"; if (cause) this.cause = cause; }
|
|
17
|
-
}
|
|
18
|
-
|
|
19
|
-
async function loadTransformers() {
|
|
20
|
-
// Prefer the modern @huggingface/transformers (post-2024 rebrand);
|
|
21
|
-
// fall back to @xenova/transformers for users still on the legacy package.
|
|
22
|
-
// Both are optional peer deps — Karajan does not ship them by default to
|
|
23
|
-
// keep the install size small (~500 MB with WASM + weights). The user
|
|
24
|
-
// installs whichever they prefer when they opt into provider: onnx.
|
|
25
|
-
/* eslint-disable import-x/no-unresolved */
|
|
26
|
-
try { return await import("@huggingface/transformers"); }
|
|
27
|
-
catch (e1) {
|
|
28
|
-
try { return await import("@xenova/transformers"); }
|
|
29
|
-
catch (e2) {
|
|
30
|
-
throw new ONNXEmbedderError(
|
|
31
|
-
"ONNX embedder requires @huggingface/transformers (or legacy @xenova/transformers). Install with `npm install @huggingface/transformers`.",
|
|
32
|
-
{ cause: e2 },
|
|
33
|
-
);
|
|
34
|
-
}
|
|
35
|
-
}
|
|
36
|
-
/* eslint-enable import-x/no-unresolved */
|
|
37
|
-
}
|
|
38
|
-
|
|
39
|
-
export class ONNXEmbedder {
|
|
40
|
-
constructor({ model = process.env.KJ_ONNX_EMBED_MODEL || DEFAULTS.model, dim = DEFAULTS.dim, pooling = DEFAULTS.pooling, normalize = DEFAULTS.normalize } = {}) {
|
|
41
|
-
Object.assign(this, { model, dim, pooling, normalize, _pipe: null });
|
|
42
|
-
}
|
|
43
|
-
async _ensurePipeline() {
|
|
44
|
-
if (this._pipe) return this._pipe;
|
|
45
|
-
const transformers = await loadTransformers();
|
|
46
|
-
this._pipe = await transformers.pipeline("feature-extraction", this.model);
|
|
47
|
-
return this._pipe;
|
|
48
|
-
}
|
|
49
|
-
async embed(text) {
|
|
50
|
-
if (typeof text !== "string" || text.length === 0) throw new ONNXEmbedderError("embed: text must be a non-empty string");
|
|
51
|
-
const pipe = await this._ensurePipeline();
|
|
52
|
-
const out = await pipe(text, { pooling: this.pooling, normalize: this.normalize });
|
|
53
|
-
const arr = Array.from(out.data || []);
|
|
54
|
-
if (arr.length !== this.dim) throw new ONNXEmbedderError(`ONNX dim mismatch: got ${arr.length}, expected ${this.dim} for model ${this.model}`);
|
|
55
|
-
return Float32Array.from(arr);
|
|
56
|
-
}
|
|
57
|
-
async embedBatch(texts) {
|
|
58
|
-
if (!Array.isArray(texts)) throw new ONNXEmbedderError("embedBatch: texts must be an array");
|
|
59
|
-
const out = [];
|
|
60
|
-
for (const t of texts) out.push(await this.embed(t));
|
|
61
|
-
return out;
|
|
62
|
-
}
|
|
63
|
-
}
|
|
1
|
+
// Shim: moved to karajan-core/rag/embedders (KJC-TSK-0632 PR3).
|
|
2
|
+
export * from "karajan-core/rag/embedders/onnx";
|
|
@@ -1,22 +1,2 @@
|
|
|
1
|
-
// KJC-TSK-
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
const DEFAULTS = { url: "https://api.openai.com/v1/embeddings", model: "text-embedding-3-small", dim: 1536, timeoutMs: 30000 };
|
|
5
|
-
|
|
6
|
-
export class OpenAIEmbedderError extends Error {
|
|
7
|
-
constructor(message, { cause, status } = {}) { super(message); this.name = "OpenAIEmbedderError"; if (cause) this.cause = cause; if (status != null) this.status = status; }
|
|
8
|
-
}
|
|
9
|
-
|
|
10
|
-
export class OpenAIEmbedder {
|
|
11
|
-
constructor({ url = DEFAULTS.url, apiKey = process.env.KJ_OPENAI_KEY, model = process.env.KJ_OPENAI_EMBED_MODEL || DEFAULTS.model, dim = DEFAULTS.dim, timeoutMs = DEFAULTS.timeoutMs, fetchFn = globalThis.fetch } = {}) {
|
|
12
|
-
// KJC architecture invariant: Karajan does not read provider API keys
|
|
13
|
-
// (ANTHROPIC_API_KEY, OPENAI_API_KEY, etc.) — those belong to the CLI
|
|
14
|
-
// agents Karajan spawns. RAG embedders are the one exception: they
|
|
15
|
-
// call the OpenAI endpoint directly, but use a Karajan-scoped env var
|
|
16
|
-
// (KJ_OPENAI_KEY) so the invariant stays clean.
|
|
17
|
-
if (!apiKey) throw new OpenAIEmbedderError("OpenAI embedder requires an api_key (config.rag.embedder.api_key or KJ_OPENAI_KEY env)");
|
|
18
|
-
Object.assign(this, { url, apiKey, model, dim, timeoutMs, fetch: fetchFn });
|
|
19
|
-
}
|
|
20
|
-
async embed(text) { return cloudEmbed(this, text, OpenAIEmbedderError, { provider: "OpenAI", body: { model: this.model, input: text }, extract: (b) => b?.data?.[0]?.embedding }); }
|
|
21
|
-
async embedBatch(texts) { if (!Array.isArray(texts)) throw new OpenAIEmbedderError("embedBatch: texts must be an array"); const out = []; for (const t of texts) out.push(await this.embed(t)); return out; }
|
|
22
|
-
}
|
|
1
|
+
// Shim: moved to karajan-core/rag/embedders (KJC-TSK-0632 PR3).
|
|
2
|
+
export * from "karajan-core/rag/embedders/openai";
|
|
@@ -1,18 +1,2 @@
|
|
|
1
|
-
// KJC-TSK-
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
const DEFAULTS = { url: "https://api.voyageai.com/v1/embeddings", model: "voyage-code-3", dim: 1024, timeoutMs: 30000 };
|
|
5
|
-
|
|
6
|
-
export class VoyageEmbedderError extends Error {
|
|
7
|
-
constructor(message, { cause, status } = {}) { super(message); this.name = "VoyageEmbedderError"; if (cause) this.cause = cause; if (status != null) this.status = status; }
|
|
8
|
-
}
|
|
9
|
-
|
|
10
|
-
export class VoyageEmbedder {
|
|
11
|
-
constructor({ url = DEFAULTS.url, apiKey = process.env.KJ_VOYAGE_KEY, model = process.env.KJ_VOYAGE_EMBED_MODEL || DEFAULTS.model, dim = DEFAULTS.dim, timeoutMs = DEFAULTS.timeoutMs, fetchFn = globalThis.fetch } = {}) {
|
|
12
|
-
// Same invariant as OpenAIEmbedder: Karajan-scoped env var.
|
|
13
|
-
if (!apiKey) throw new VoyageEmbedderError("Voyage embedder requires an api_key (config.rag.embedder.api_key or KJ_VOYAGE_KEY env)");
|
|
14
|
-
Object.assign(this, { url, apiKey, model, dim, timeoutMs, fetch: fetchFn });
|
|
15
|
-
}
|
|
16
|
-
async embed(text) { return cloudEmbed(this, text, VoyageEmbedderError, { provider: "Voyage", body: { model: this.model, input: [text], input_type: "document" }, extract: (b) => b?.data?.[0]?.embedding }); }
|
|
17
|
-
async embedBatch(texts) { if (!Array.isArray(texts)) throw new VoyageEmbedderError("embedBatch: texts must be an array"); const out = []; for (const t of texts) out.push(await this.embed(t)); return out; }
|
|
18
|
-
}
|
|
1
|
+
// Shim: moved to karajan-core/rag/embedders (KJC-TSK-0632 PR3).
|
|
2
|
+
export * from "karajan-core/rag/embedders/voyage";
|
package/src/rag/rerank.js
CHANGED
|
@@ -1,74 +1,4 @@
|
|
|
1
|
-
//
|
|
2
|
-
//
|
|
3
|
-
//
|
|
4
|
-
|
|
5
|
-
// encode query + passage instead of caching the passage), but they are
|
|
6
|
-
// substantially more precise for the final top-K ranking. Karajan calls
|
|
7
|
-
// the reranker only on the post-fusion candidates (≤topK*2), so the
|
|
8
|
-
// added latency is bounded.
|
|
9
|
-
//
|
|
10
|
-
// Model is loaded dynamically via @huggingface/transformers (same dep as
|
|
11
|
-
// the ONNX embedder, KJC-TSK-0447). Default: `Xenova/ms-marco-MiniLM-L-6-v2`,
|
|
12
|
-
// the de-facto standard sentence-transformers reranker.
|
|
13
|
-
|
|
14
|
-
const DEFAULT_MODEL = "Xenova/ms-marco-MiniLM-L-6-v2";
|
|
15
|
-
|
|
16
|
-
export class RerankError extends Error {
|
|
17
|
-
constructor(message, { cause } = {}) { super(message); this.name = "RerankError"; if (cause) this.cause = cause; }
|
|
18
|
-
}
|
|
19
|
-
|
|
20
|
-
async function loadTransformers() {
|
|
21
|
-
/* eslint-disable import-x/no-unresolved */
|
|
22
|
-
try { return await import("@huggingface/transformers"); }
|
|
23
|
-
catch (e1) {
|
|
24
|
-
try { return await import("@xenova/transformers"); }
|
|
25
|
-
catch (e2) {
|
|
26
|
-
throw new RerankError(
|
|
27
|
-
"rerank requires @huggingface/transformers (or legacy @xenova/transformers). Install with `npm install @huggingface/transformers`.",
|
|
28
|
-
{ cause: e2 },
|
|
29
|
-
);
|
|
30
|
-
}
|
|
31
|
-
}
|
|
32
|
-
/* eslint-enable import-x/no-unresolved */
|
|
33
|
-
}
|
|
34
|
-
|
|
35
|
-
/**
|
|
36
|
-
* Lazy singleton — first call downloads weights (~80 MB), every subsequent
|
|
37
|
-
* call reuses the same pipeline instance.
|
|
38
|
-
*/
|
|
39
|
-
let _pipelinePromise = null;
|
|
40
|
-
function getPipeline(modelName) {
|
|
41
|
-
if (!_pipelinePromise) _pipelinePromise = (async () => {
|
|
42
|
-
const t = await loadTransformers();
|
|
43
|
-
// text-classification pipeline returns the cross-encoder relevance score
|
|
44
|
-
// for a `[query, passage]` text pair when applied to a CE model.
|
|
45
|
-
return t.pipeline("text-classification", modelName);
|
|
46
|
-
})();
|
|
47
|
-
return _pipelinePromise;
|
|
48
|
-
}
|
|
49
|
-
|
|
50
|
-
/** Reset for tests. */
|
|
51
|
-
export function _resetPipeline() { _pipelinePromise = null; }
|
|
52
|
-
|
|
53
|
-
/**
|
|
54
|
-
* Rerank `hits` against `queryText` using the cross-encoder. Returns a NEW
|
|
55
|
-
* array sorted by descending rerank score (best first), `score` mutated
|
|
56
|
-
* so the existing downstream code (kindBoost, slicing) keeps working.
|
|
57
|
-
*
|
|
58
|
-
* Each input hit is treated as a [query, hit.text] pair. Truncates the
|
|
59
|
-
* passage to `maxChars` to avoid blowing past the CE model's context
|
|
60
|
-
* window — most CE models handle 512 tokens, ~2000 chars is safe.
|
|
61
|
-
*
|
|
62
|
-
* If `pipelineFn` is injected (tests), it bypasses model loading.
|
|
63
|
-
*/
|
|
64
|
-
export async function rerank(queryText, hits, { model = process.env.KJ_RERANK_MODEL || DEFAULT_MODEL, maxChars = 2000, pipelineFn = null } = {}) {
|
|
65
|
-
if (!queryText || typeof queryText !== "string") throw new RerankError("rerank: queryText must be a non-empty string");
|
|
66
|
-
if (!Array.isArray(hits)) throw new RerankError("rerank: hits must be an array");
|
|
67
|
-
if (hits.length === 0) return hits;
|
|
68
|
-
const pipe = pipelineFn || await getPipeline(model);
|
|
69
|
-
const pairs = hits.map((h) => ({ text: queryText, text_pair: String(h.text || "").slice(0, maxChars) }));
|
|
70
|
-
const results = await pipe(pairs);
|
|
71
|
-
return hits
|
|
72
|
-
.map((h, i) => ({ ...h, _rerank: Number(results[i]?.score) || 0, score: -Number(results[i]?.score) || 0 }))
|
|
73
|
-
.sort((a, b) => a.score - b.score);
|
|
74
|
-
}
|
|
1
|
+
// Shim: rerank now lives in karajan-core/rag so the hu-board workspace
|
|
2
|
+
// can consume it without a relative dep on the CLI src tree.
|
|
3
|
+
// KJC-TSK-0632 PR2.
|
|
4
|
+
export { RerankError, _resetPipeline, rerank } from "karajan-core/rag/rerank";
|
package/src/rag/retriever.js
CHANGED
|
@@ -1,127 +1,2 @@
|
|
|
1
|
-
//
|
|
2
|
-
|
|
3
|
-
import { parseWhere, buildWhereSql } from "./where-parser.js";
|
|
4
|
-
import { rerank } from "./rerank.js";
|
|
5
|
-
|
|
6
|
-
const DEFAULT_KIND_BOOST = { plan: 0.05, onboarding: 0.03, code: 0 };
|
|
7
|
-
|
|
8
|
-
// KJC-TSK-0440 — asymmetric source/test boost.
|
|
9
|
-
const SOURCE_BOOST_NON_TEST = 0.05;
|
|
10
|
-
const TEST_TERMS_RE = /\b(test|tests|spec|specs|expect|describe|it\(|jest|vitest|mocha)\b/i;
|
|
11
|
-
const TEST_PATH_RE = /[\\/](tests?|specs?|__tests__)[\\/]|\.test\.[jt]sx?$|\.spec\.[jt]sx?$/i;
|
|
12
|
-
|
|
13
|
-
function shouldBoostSources(queryText) { return !TEST_TERMS_RE.test(queryText); }
|
|
14
|
-
function isTestPath(source) { return TEST_PATH_RE.test(source || ""); }
|
|
15
|
-
|
|
16
|
-
// KJC-TSK-0443 — fuse semantic + keyword hits into a unified candidate set.
|
|
17
|
-
// For 'semantic'/'keyword' modes we surface that side. For 'hybrid' (default)
|
|
18
|
-
// we min-max normalise both scores to [0,1] (lower=better) and linear-combine
|
|
19
|
-
// via alpha * semantic + (1-alpha) * keyword. Result written back to
|
|
20
|
-
// `distance` so the kind+source boost pipeline keeps working unchanged.
|
|
21
|
-
function fuseHits(semantic, keyword, alpha, mode) {
|
|
22
|
-
if (mode === "semantic") return semantic;
|
|
23
|
-
if (mode === "keyword") return keyword.map((h) => ({ ...h, distance: h.bm25 }));
|
|
24
|
-
const byId = new Map();
|
|
25
|
-
for (const h of semantic) byId.set(h.id, { ...h, _sem: h.distance });
|
|
26
|
-
for (const h of keyword) {
|
|
27
|
-
const prev = byId.get(h.id);
|
|
28
|
-
if (prev) prev._kw = h.bm25;
|
|
29
|
-
else byId.set(h.id, { ...h, _kw: h.bm25, distance: h.bm25 });
|
|
30
|
-
}
|
|
31
|
-
const list = [...byId.values()];
|
|
32
|
-
const norm = (vals) => {
|
|
33
|
-
const xs = vals.filter((v) => Number.isFinite(v));
|
|
34
|
-
if (xs.length === 0) return () => 0.5;
|
|
35
|
-
const min = Math.min(...xs); const max = Math.max(...xs); const span = max - min || 1;
|
|
36
|
-
return (v) => Number.isFinite(v) ? (v - min) / span : 1;
|
|
37
|
-
};
|
|
38
|
-
const nSem = norm(list.map((h) => h._sem));
|
|
39
|
-
const nKw = norm(list.map((h) => h._kw));
|
|
40
|
-
for (const h of list) h.distance = alpha * nSem(h._sem) + (1 - alpha) * nKw(h._kw);
|
|
41
|
-
return list;
|
|
42
|
-
}
|
|
43
|
-
|
|
44
|
-
// KJC-TSK-0484 PR-B — Maximal Marginal Relevance. Given an ordered list of
|
|
45
|
-
// candidates (best-first by `score`), pick `topK` that balance relevance
|
|
46
|
-
// against intra-result diversity. `lambda=1` collapses to plain relevance;
|
|
47
|
-
// `lambda=0` maximises diversity. Cosine sim between candidate embeddings
|
|
48
|
-
// drives the redundancy penalty. Candidates without an embedding are kept
|
|
49
|
-
// at the end (no penalty applies).
|
|
50
|
-
function cosineSim(a, b) {
|
|
51
|
-
if (!a || !b || a.length !== b.length) return 0;
|
|
52
|
-
let dot = 0; let na = 0; let nb = 0;
|
|
53
|
-
for (let i = 0; i < a.length; i += 1) { dot += a[i] * b[i]; na += a[i] * a[i]; nb += b[i] * b[i]; }
|
|
54
|
-
const denom = Math.sqrt(na) * Math.sqrt(nb);
|
|
55
|
-
return denom === 0 ? 0 : dot / denom;
|
|
56
|
-
}
|
|
57
|
-
|
|
58
|
-
export function mmrRerank(candidates, embeddings, { topK, lambda = 0.7 }) {
|
|
59
|
-
const remaining = candidates.slice();
|
|
60
|
-
const picked = [];
|
|
61
|
-
const relevance = (c) => 1 / (1 + Math.max(0, c.score ?? c.distance ?? 0));
|
|
62
|
-
while (picked.length < topK && remaining.length > 0) {
|
|
63
|
-
let bestIdx = 0; let bestVal = -Infinity;
|
|
64
|
-
for (let i = 0; i < remaining.length; i += 1) {
|
|
65
|
-
const c = remaining[i];
|
|
66
|
-
const rel = relevance(c);
|
|
67
|
-
let maxSim = 0;
|
|
68
|
-
const ce = embeddings.get(c.id);
|
|
69
|
-
if (ce && picked.length > 0) {
|
|
70
|
-
for (const p of picked) {
|
|
71
|
-
const pe = embeddings.get(p.id);
|
|
72
|
-
if (pe) maxSim = Math.max(maxSim, cosineSim(ce, pe));
|
|
73
|
-
}
|
|
74
|
-
}
|
|
75
|
-
const score = lambda * rel - (1 - lambda) * maxSim;
|
|
76
|
-
if (score > bestVal) { bestVal = score; bestIdx = i; }
|
|
77
|
-
}
|
|
78
|
-
picked.push(remaining.splice(bestIdx, 1)[0]);
|
|
79
|
-
}
|
|
80
|
-
return picked;
|
|
81
|
-
}
|
|
82
|
-
|
|
83
|
-
export async function query(db, embedder, text, { topK = 5, scope = "all", kindBoost = DEFAULT_KIND_BOOST, project = null, mode = "hybrid", alpha = 0.6, where = null, rerankOpts = null, diversify = false, mmrLambda = 0.7 } = {}) {
|
|
84
|
-
if (!text || typeof text !== "string") throw new Error("query: text must be a non-empty string");
|
|
85
|
-
const fetchK = Math.min(50, topK * 2);
|
|
86
|
-
const scopeKind = scope === "plans" ? "plan" : scope === "code" ? "code" : scope === "onboarding" ? "onboarding" : null;
|
|
87
|
-
// KJC-TSK-0448 — metadata filter. `where` is a string like "symbol=Foo AND
|
|
88
|
-
// hu_id=HU-003"; we parse once and reuse the SQL fragment across both
|
|
89
|
-
// semantic and keyword searches.
|
|
90
|
-
const parsed = parseWhere(where);
|
|
91
|
-
if (!parsed.ok) throw new Error(`query: ${parsed.error}`);
|
|
92
|
-
const { sql: whereSql, params: whereParams } = buildWhereSql(parsed.clauses);
|
|
93
|
-
const opts = { kind: scopeKind, project, whereSql, whereParams };
|
|
94
|
-
const wantSemantic = mode !== "keyword";
|
|
95
|
-
const wantKeyword = mode !== "semantic";
|
|
96
|
-
const semanticHits = wantSemantic ? searchSimilar(db, await embedder.embed(text), fetchK, opts) : [];
|
|
97
|
-
const keywordHits = wantKeyword ? searchBM25(db, text, fetchK, opts) : [];
|
|
98
|
-
const raw = fuseHits(semanticHits, keywordHits, alpha, mode);
|
|
99
|
-
if (raw.length === 0) return [];
|
|
100
|
-
const boostSources = shouldBoostSources(text);
|
|
101
|
-
const scored = raw.map((r) => {
|
|
102
|
-
const kindB = kindBoost[r.kind] || 0;
|
|
103
|
-
const sourceB = (boostSources && r.kind === "code" && !isTestPath(r.source)) ? SOURCE_BOOST_NON_TEST : 0;
|
|
104
|
-
return { ...r, score: r.distance - kindB - sourceB };
|
|
105
|
-
}).sort((a, b) => a.score - b.score);
|
|
106
|
-
// KJC-TSK-0484 PR-B — MMR diversification over topK*2 candidates. Fetches
|
|
107
|
-
// embeddings only when requested; cosine between candidate vectors drives
|
|
108
|
-
// the redundancy penalty so near-duplicates that survived dedup (e.g. a
|
|
109
|
-
// boilerplate chunk repeated under slightly different wording) get pushed
|
|
110
|
-
// down. Skipped entirely when `diversify=false` (default).
|
|
111
|
-
let reranked;
|
|
112
|
-
if (diversify) {
|
|
113
|
-
const pool = scored.slice(0, Math.min(scored.length, topK * 2));
|
|
114
|
-
const embeddings = getEmbeddingsByIds(db, pool.map((c) => c.id));
|
|
115
|
-
reranked = mmrRerank(pool, embeddings, { topK, lambda: mmrLambda });
|
|
116
|
-
} else {
|
|
117
|
-
reranked = scored.slice(0, topK);
|
|
118
|
-
}
|
|
119
|
-
// KJC-TSK-0449 — optional cross-encoder rerank. When `rerankOpts` is set,
|
|
120
|
-
// we re-score the topK survivors with a (query, passage) cross-encoder;
|
|
121
|
-
// the kind+source boost has already been applied, so the rerank acts as
|
|
122
|
-
// a finer-grained quality lever on top.
|
|
123
|
-
if (rerankOpts) return await rerank(text, reranked, rerankOpts);
|
|
124
|
-
return reranked;
|
|
125
|
-
}
|
|
126
|
-
|
|
127
|
-
export { fuseHits, cosineSim };
|
|
1
|
+
// Shim: retriever now lives in karajan-core/rag (KJC-TSK-0632 PR3).
|
|
2
|
+
export * from "karajan-core/rag/retriever";
|
package/src/rag/where-parser.js
CHANGED
|
@@ -1,54 +1,4 @@
|
|
|
1
|
-
//
|
|
2
|
-
//
|
|
3
|
-
//
|
|
4
|
-
|
|
5
|
-
// pair := IDENT "=" VALUE
|
|
6
|
-
// IDENT := [A-Za-z_][A-Za-z0-9_.-]*
|
|
7
|
-
// VALUE := unquoted token | "quoted string"
|
|
8
|
-
//
|
|
9
|
-
// Examples:
|
|
10
|
-
// symbol=loadConfig
|
|
11
|
-
// hu_id=HU-003 AND kind=plan
|
|
12
|
-
// "headingPath.0"=Architecture
|
|
13
|
-
//
|
|
14
|
-
// Returns { ok: true, clauses: [["symbol","loadConfig"], ...] } or
|
|
15
|
-
// { ok: false, error: "...explanation..." }. The retriever turns each
|
|
16
|
-
// clause into a `json_extract(c.metadata, '$.<key>') = ?` SQL fragment;
|
|
17
|
-
// the `kind` key is special-cased to filter the column directly.
|
|
18
|
-
|
|
19
|
-
const IDENT_RE = /^[A-Za-z_][A-Za-z0-9_.-]*$/;
|
|
20
|
-
|
|
21
|
-
export function parseWhere(input) {
|
|
22
|
-
if (input == null) return { ok: true, clauses: [] };
|
|
23
|
-
if (typeof input !== "string") return { ok: false, error: "where: input must be a string" };
|
|
24
|
-
const s = input.trim();
|
|
25
|
-
if (!s) return { ok: true, clauses: [] };
|
|
26
|
-
const parts = s.split(/\s+AND\s+/i);
|
|
27
|
-
const clauses = [];
|
|
28
|
-
for (const p of parts) {
|
|
29
|
-
const eq = p.indexOf("=");
|
|
30
|
-
if (eq <= 0) return { ok: false, error: `where: bad pair "${p}" — expected key=value` };
|
|
31
|
-
const key = p.slice(0, eq).trim().replace(/^"|"$/g, "");
|
|
32
|
-
let val = p.slice(eq + 1).trim();
|
|
33
|
-
if (!IDENT_RE.test(key)) return { ok: false, error: `where: invalid key "${key}"` };
|
|
34
|
-
if ((val.startsWith('"') && val.endsWith('"')) || (val.startsWith("'") && val.endsWith("'"))) val = val.slice(1, -1);
|
|
35
|
-
if (val.length === 0) return { ok: false, error: `where: empty value for key "${key}"` };
|
|
36
|
-
clauses.push([key, val]);
|
|
37
|
-
}
|
|
38
|
-
return { ok: true, clauses };
|
|
39
|
-
}
|
|
40
|
-
|
|
41
|
-
/**
|
|
42
|
-
* Build a `(sqlFragment, params)` pair to append to a chunks-table query.
|
|
43
|
-
* `kind` filters the column directly; everything else goes through json_extract.
|
|
44
|
-
*/
|
|
45
|
-
export function buildWhereSql(clauses) {
|
|
46
|
-
if (!clauses?.length) return { sql: "", params: [] };
|
|
47
|
-
const fragments = [];
|
|
48
|
-
const params = [];
|
|
49
|
-
for (const [k, v] of clauses) {
|
|
50
|
-
if (k === "kind") { fragments.push("c.kind = ?"); params.push(v); }
|
|
51
|
-
else { fragments.push(`json_extract(c.metadata, '$.${k}') = ?`); params.push(v); }
|
|
52
|
-
}
|
|
53
|
-
return { sql: ` AND ${fragments.join(" AND ")}`, params };
|
|
54
|
-
}
|
|
1
|
+
// Shim: where-parser now lives in karajan-core/rag so the hu-board
|
|
2
|
+
// workspace can consume it without a relative dep on the CLI src tree.
|
|
3
|
+
// KJC-TSK-0632 PR2.
|
|
4
|
+
export { parseWhere, buildWhereSql } from "karajan-core/rag/where-parser";
|
package/src/roles/agent-role.js
CHANGED
|
@@ -129,6 +129,10 @@ export class AgentRole extends BaseRole {
|
|
|
129
129
|
|
|
130
130
|
const runArgs = { prompt, role: this.name };
|
|
131
131
|
if (onOutput) runArgs.onOutput = onOutput;
|
|
132
|
+
// PAR-H (KJC-TSK-0631): per-call env additions for the agent
|
|
133
|
+
// subprocess (lane slot vars). Agents that ignore task.env simply
|
|
134
|
+
// run without them.
|
|
135
|
+
if (extracted.env) runArgs.env = extracted.env;
|
|
132
136
|
// Φ1-D (KJC-PCS-0057): roles that build their prompt as a
|
|
133
137
|
// stable/volatile layout forward both buckets so cache-aware agents
|
|
134
138
|
// (ClaudeAgent) can ship the stable block as system prompt. Agents
|
package/src/roles/coder-role.js
CHANGED
|
@@ -56,6 +56,8 @@ export class CoderRole extends AgentRole {
|
|
|
56
56
|
// PAR-E2 (KJC-TSK-0629): per-call project root. Worktree lanes pass
|
|
57
57
|
// their lane dir here; the shared role instance keeps its own config.
|
|
58
58
|
projectDir: input?.projectDir || null,
|
|
59
|
+
// PAR-H (KJC-TSK-0631): per-call subprocess env (lane slot vars).
|
|
60
|
+
env: input?.env || null,
|
|
59
61
|
onOutput: input?.onOutput || null
|
|
60
62
|
};
|
|
61
63
|
}
|