@kolisachint/hoocode-agent 0.5.30 → 0.5.31
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +103 -0
- package/dist/core/agent-selection-eval.d.ts +91 -0
- package/dist/core/agent-selection-eval.d.ts.map +1 -0
- package/dist/core/agent-selection-eval.js +186 -0
- package/dist/core/agent-selection-eval.js.map +1 -0
- package/dist/core/builtin-skills.d.ts +61 -0
- package/dist/core/builtin-skills.d.ts.map +1 -0
- package/dist/core/builtin-skills.js +106 -0
- package/dist/core/builtin-skills.js.map +1 -0
- package/dist/core/extensions/plugins/default-marketplace/.agents-plugin/marketplace.json +5 -1
- package/dist/core/extensions/plugins/trigger-judge.d.ts +42 -0
- package/dist/core/extensions/plugins/trigger-judge.d.ts.map +1 -0
- package/dist/core/extensions/plugins/trigger-judge.js +121 -0
- package/dist/core/extensions/plugins/trigger-judge.js.map +1 -0
- package/dist/core/external-tools.d.ts +67 -0
- package/dist/core/external-tools.d.ts.map +1 -0
- package/dist/core/external-tools.js +120 -0
- package/dist/core/external-tools.js.map +1 -0
- package/dist/core/mode-prompts.d.ts +15 -3
- package/dist/core/mode-prompts.d.ts.map +1 -1
- package/dist/core/mode-prompts.js +17 -29
- package/dist/core/mode-prompts.js.map +1 -1
- package/dist/core/tools/propose-plugin.d.ts.map +1 -1
- package/dist/core/tools/propose-plugin.js +7 -7
- package/dist/core/tools/propose-plugin.js.map +1 -1
- package/dist/core/tools/subagent.d.ts +7 -1
- package/dist/core/tools/subagent.d.ts.map +1 -1
- package/dist/core/tools/subagent.js +12 -24
- package/dist/core/tools/subagent.js.map +1 -1
- package/dist/extensions/core/modes.d.ts.map +1 -1
- package/dist/extensions/core/modes.js +21 -23
- package/dist/extensions/core/modes.js.map +1 -1
- package/dist/extensions/core/scaffold.d.ts.map +1 -1
- package/dist/extensions/core/scaffold.js +28 -24
- package/dist/extensions/core/scaffold.js.map +1 -1
- package/dist/init-templates.generated.d.ts +3 -0
- package/dist/init-templates.generated.d.ts.map +1 -1
- package/dist/init-templates.generated.js +12 -0
- package/dist/init-templates.generated.js.map +1 -1
- package/dist/main.d.ts.map +1 -1
- package/dist/main.js +14 -1
- package/dist/main.js.map +1 -1
- package/dist/modes/interactive/components/settings-selector.d.ts +7 -0
- package/dist/modes/interactive/components/settings-selector.d.ts.map +1 -1
- package/dist/modes/interactive/components/settings-selector.js +152 -11
- package/dist/modes/interactive/components/settings-selector.js.map +1 -1
- package/dist/modes/interactive/interactive-mode.d.ts.map +1 -1
- package/dist/modes/interactive/interactive-mode.js +5 -0
- package/dist/modes/interactive/interactive-mode.js.map +1 -1
- package/dist/utils/tools-manager.d.ts +30 -0
- package/dist/utils/tools-manager.d.ts.map +1 -1
- package/dist/utils/tools-manager.js +35 -0
- package/dist/utils/tools-manager.js.map +1 -1
- package/docs/modes.md +4 -0
- package/docs/plugins.md +10 -0
- package/docs/settings.md +42 -0
- package/docs/skills.md +30 -0
- package/examples/extensions/custom-provider-anthropic/package.json +1 -1
- package/examples/extensions/custom-provider-gitlab-duo/package.json +1 -1
- package/examples/extensions/sandbox/package.json +1 -1
- package/examples/extensions/with-deps/package.json +1 -1
- package/package.json +5 -4
- package/templates/prompts/grill-bridge.md +1 -0
- package/templates/prompts/grill-me.md +7 -0
- package/templates/prompts/grill-plan.md +9 -0
- package/templates/prompts/task-background-agents.md +2 -0
- package/templates/prompts/task-background-none.md +1 -0
- package/templates/prompts/task-main.md +19 -0
- package/templates/skills/plugin-authoring/SKILL.md +81 -0
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"builtin-skills.js","sourceRoot":"","sources":["../../src/core/builtin-skills.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;GAoBG;AAEH,OAAO,EAAE,UAAU,EAAE,MAAM,aAAa,CAAC;AACzC,OAAO,EAAE,UAAU,EAAE,SAAS,EAAE,YAAY,EAAE,UAAU,EAAE,MAAM,EAAE,aAAa,EAAE,MAAM,SAAS,CAAC;AACjG,OAAO,EAAE,OAAO,EAAE,IAAI,EAAE,MAAM,WAAW,CAAC;AAC1C,OAAO,EAAE,WAAW,EAAE,MAAM,cAAc,CAAC;AAC3C,OAAO,EAAE,eAAe,EAAE,MAAM,gCAAgC,CAAC;AAuBjE,MAAM,CAAC,MAAM,cAAc,GAA4B;IACtD;QACC,IAAI,EAAE,kBAAkB;QACxB,OAAO,EACN,sJAAsJ;QACvJ,wEAAwE;QACxE,wEAAwE;QACxE,IAAI,EAAE,CAAC,OAAO,EAAE,EAAE,CAAC,OAAO,CAAC,iBAAiB;KAC5C;CACD,CAAC;AAEF,gFAAgF;AAChF,SAAS,WAAW,GAAW;IAC9B,MAAM,IAAI,GAAG,UAAU,CAAC,QAAQ,CAAC,CAAC;IAClC,KAAK,MAAM,GAAG,IAAI,MAAM,CAAC,IAAI,CAAC,eAAe,CAAC,CAAC,IAAI,EAAE,EAAE,CAAC;QACvD,IAAI,CAAC,MAAM,CAAC,GAAG,CAAC,CAAC;QACjB,IAAI,CAAC,MAAM,CAAC,IAAI,CAAC,CAAC;QAClB,IAAI,CAAC,MAAM,CAAC,eAAe,CAAC,GAAG,CAAC,IAAI,EAAE,CAAC,CAAC;QACxC,IAAI,CAAC,MAAM,CAAC,IAAI,CAAC,CAAC;IACnB,CAAC;IACD,OAAO,IAAI,CAAC,MAAM,CAAC,KAAK,CAAC,CAAC,KAAK,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC;AAAA,CACvC;AAED,6DAA6D;AAC7D,MAAM,UAAU,qBAAqB,CAAC,QAAQ,GAAW,WAAW,EAAE,EAAU;IAC/E,OAAO,IAAI,CAAC,QAAQ,EAAE,OAAO,EAAE,gBAAgB,EAAE,WAAW,EAAE,CAAC,CAAC;AAAA,CAChE;AAED;;;;;;;GAOG;AACH,MAAM,UAAU,wBAAwB,CAAC,QAAQ,GAAW,WAAW,EAAE,EAAiB;IACzF,MAAM,IAAI,GAAG,qBAAqB,CAAC,QAAQ,CAAC,CAAC;IAC7C,IAAI,CAAC;QACJ,KAAK,MAAM,CAAC,YAAY,EAAE,OAAO,CAAC,IAAI,MAAM,CAAC,OAAO,CAAC,eAAe,CAAC,EAAE,CAAC;YACvE,MAAM,MAAM,GAAG,IAAI,CAAC,IAAI,EAAE,YAAY,CAAC,CAAC;YACxC,wEAAwE;YACxE,iEAAiE;YACjE,IAAI,UAAU,CAAC,MAAM,CAAC,IAAI,YAAY,CAAC,MAAM,EAAE,OAAO,CAAC,KAAK,OAAO;gBAAE,SAAS;YAC9E,SAAS,CAAC,OAAO,CAAC,MAAM,CAAC,EAAE,EAAE,SAAS,EAAE,IAAI,EAAE,CAAC,CAAC;YAChD,oEAAoE;YACpE,kEAAkE;YAClE,MAAM,IAAI,GAAG,GAAG,MAAM,IAAI,OAAO,CAAC,GAAG,MAAM,CAAC;YAC5C,aAAa,CAAC,IAAI,EAAE,OAAO,EAAE,OAAO,CAAC,CAAC;YACtC,UAAU,CAAC,IAAI,EAAE,MAAM,CAAC,CAAC;QAC1B,CAAC;QACD,OAAO,IAAI,CAAC;IACb,CAAC;IAAC,MAAM,CAAC;QACR,wEAAwE;QACxE,uEAAuE;QACvE,mEAAmE;QACnE,sEAAoE;QACpE,4BAA4B;QAC5B,IAAI,CAAC;YACJ,MAAM,CAAC,IAAI,EAAE,EAAE,SAAS,EAAE,IAAI,EAAE,KAAK,EAAE,IAAI,EAAE,CAAC,CAAC;QAChD,CAAC;QAAC,MAAM,CAAC,CAAA,CAAC;QACV,OAAO,IAAI,CAAC;IACb,CAAC;AAAA,CACD;AAED;;;;;;GAMG;AACH,MAAM,UAAU,iBAAiB,CAAC,IAAsB,EAAE,QAAQ,GAAW,WAAW,EAAE,EAAY;IACrG,MAAM,OAAO,GAAG,cAAc,CAAC,MAAM,CAAC,CAAC,KAAK,EAAE,EAAE,CAAC,CAAC,KAAK,CAAC,IAAI,IAAI,KAAK,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC,CAAC;IAClF,IAAI,OAAO,CAAC,MAAM,KAAK,CAAC;QAAE,OAAO,EAAE,CAAC;IAEpC,MAAM,IAAI,GAAG,wBAAwB,CAAC,QAAQ,CAAC,CAAC;IAChD,IAAI,CAAC,IAAI;QAAE,OAAO,EAAE,CAAC;IAErB,OAAO,OAAO,CAAC,GAAG,CAAC,CAAC,KAAK,EAAE,EAAE,CAAC,IAAI,CAAC,IAAI,EAAE,KAAK,CAAC,IAAI,CAAC,CAAC,CAAC,MAAM,CAAC,CAAC,GAAG,EAAE,EAAE,CAAC,UAAU,CAAC,GAAG,CAAC,CAAC,CAAC;AAAA,CACvF","sourcesContent":["/**\n * Skills hoocode itself ships.\n *\n * hoocode reads skills from `~/.agents/skills`, `.hoocode/skills`, `.claude/skills`\n * and installed packages — every source except its own. That gap is why it\n * shipped three subagents and zero skills while telling users skills are the\n * extension unit: there was simply nowhere for a first-party skill to live.\n *\n * The obstacle is that a skill's `<location>` has to be a real readable path —\n * the model loads a skill by `read`ing it — and the Bun-compiled binary has no\n * `templates/` beside it. So rather than resolve the package directory (which\n * differs across npm/pnpm/source/binary layouts and would give the compiled\n * binary a silently degraded skill set), every install materializes the same\n * embedded copy into a cache directory. One code path, same behaviour\n * everywhere.\n *\n * The cache is keyed by content hash, so an upgrade writes a new directory and\n * a dev build that changes a skill without changing the version still takes\n * effect. It is a cache, not user-editable state: `~/.agents/skills` is where a\n * user's own skills go, and nothing here ever writes there.\n */\n\nimport { createHash } from \"node:crypto\";\nimport { existsSync, mkdirSync, readFileSync, renameSync, rmSync, writeFileSync } from \"node:fs\";\nimport { dirname, join } from \"node:path\";\nimport { getAgentDir } from \"../config.js\";\nimport { EMBEDDED_SKILLS } from \"../init-templates.generated.js\";\n\n/** What decides whether a built-in skill is registered this session. */\nexport interface BuiltinSkillGate {\n\t/** The `enablePluginTools` setting (the plugin system's master switch). */\n\tenablePluginTools: boolean;\n}\n\nexport interface BuiltinSkill {\n\t/** Directory name under `templates/skills`, and the skill's own name. */\n\tname: string;\n\t/** Why hoocode ships it. Documentation, and the catalog test reads it. */\n\tsummary: string;\n\t/**\n\t * Registered only when this returns true. A skill costs its description on\n\t * every turn, so one that only makes sense alongside a feature rides that\n\t * feature's switch rather than the default user's token budget.\n\t *\n\t * Omit for a skill that should always be available.\n\t */\n\tgate?: (options: BuiltinSkillGate) => boolean;\n}\n\nexport const BUILTIN_SKILLS: readonly BuiltinSkill[] = [\n\t{\n\t\tname: \"plugin-authoring\",\n\t\tsummary:\n\t\t\t\"The craft half of ProposePlugin/UpdatePlugin: when a capability is worth extracting, naming it so it triggers again, portability, and the hook trap.\",\n\t\t// Useless without the tools it describes, and those are off by default,\n\t\t// so this costs nothing for a user who never enables the plugin system.\n\t\tgate: (options) => options.enablePluginTools,\n\t},\n];\n\n/** Stable short hash of the embedded skill tree; the cache directory's name. */\nfunction contentHash(): string {\n\tconst hash = createHash(\"sha256\");\n\tfor (const key of Object.keys(EMBEDDED_SKILLS).sort()) {\n\t\thash.update(key);\n\t\thash.update(\"\\0\");\n\t\thash.update(EMBEDDED_SKILLS[key] ?? \"\");\n\t\thash.update(\"\\0\");\n\t}\n\treturn hash.digest(\"hex\").slice(0, 12);\n}\n\n/** Root of the materialized copy for the current content. */\nexport function builtinSkillsCacheDir(agentDir: string = getAgentDir()): string {\n\treturn join(agentDir, \"cache\", \"builtin-skills\", contentHash());\n}\n\n/**\n * Write the embedded skills to the cache directory if they are not already\n * there, and return its path.\n *\n * Returns null when nothing could be written — a read-only home, a full disk.\n * That is a degraded session, not a broken one: the caller contributes no skill\n * paths and hoocode runs exactly as it did before these existed.\n */\nexport function materializeBuiltinSkills(agentDir: string = getAgentDir()): string | null {\n\tconst root = builtinSkillsCacheDir(agentDir);\n\ttry {\n\t\tfor (const [relativePath, content] of Object.entries(EMBEDDED_SKILLS)) {\n\t\t\tconst target = join(root, relativePath);\n\t\t\t// Content is hash-addressed, so an existing file with the right size is\n\t\t\t// already correct; re-reading beats re-writing on every startup.\n\t\t\tif (existsSync(target) && readFileSync(target, \"utf-8\") === content) continue;\n\t\t\tmkdirSync(dirname(target), { recursive: true });\n\t\t\t// Write-then-rename so a killed process never leaves a half-written\n\t\t\t// SKILL.md that would parse as a malformed skill on the next run.\n\t\t\tconst temp = `${target}.${process.pid}.tmp`;\n\t\t\twriteFileSync(temp, content, \"utf-8\");\n\t\t\trenameSync(temp, target);\n\t\t}\n\t\treturn root;\n\t} catch {\n\t\t// Drop a partial tree so the next run rebuilds it rather than loading a\n\t\t// half-written skill. The cleanup gets its own guard: `force` swallows\n\t\t// ENOENT but not ENOTDIR, and a cleanup that throws would turn the\n\t\t// degraded path back into a crash — which is the failure this whole\n\t\t// branch exists to prevent.\n\t\ttry {\n\t\t\trmSync(root, { recursive: true, force: true });\n\t\t} catch {}\n\t\treturn null;\n\t}\n}\n\n/**\n * The skill directories to load this session, after gating.\n *\n * Returns per-skill directories rather than the root so a gated-off skill is\n * genuinely absent rather than loaded and filtered later — the load is what\n * costs the description on every turn.\n */\nexport function builtinSkillPaths(gate: BuiltinSkillGate, agentDir: string = getAgentDir()): string[] {\n\tconst enabled = BUILTIN_SKILLS.filter((skill) => !skill.gate || skill.gate(gate));\n\tif (enabled.length === 0) return [];\n\n\tconst root = materializeBuiltinSkills(agentDir);\n\tif (!root) return [];\n\n\treturn enabled.map((skill) => join(root, skill.name)).filter((dir) => existsSync(dir));\n}\n"]}
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The model call G4 was designed around and never had.
|
|
3
|
+
*
|
|
4
|
+
* `trigger-eval.ts` takes its judge as a parameter so scoring stays testable
|
|
5
|
+
* without a model, and nothing in the tree ever passed one — so every G4 run
|
|
6
|
+
* has reported `not-run` since it was written. This is that judge.
|
|
7
|
+
*
|
|
8
|
+
* It is deliberately in its own module rather than inside `trigger-eval.ts`:
|
|
9
|
+
* the eval is pure and testable, this reaches the network, and the same
|
|
10
|
+
* separation lets the agent-selection eval reuse the judge without pulling in
|
|
11
|
+
* the plugin gate machinery.
|
|
12
|
+
*
|
|
13
|
+
* ## Why one call for all prompts
|
|
14
|
+
*
|
|
15
|
+
* The judge sees every candidate and every prompt at once. Per-prompt calls
|
|
16
|
+
* would be cleaner to reason about, but the question being scored is
|
|
17
|
+
* comparative — "which of these fires" — and batching keeps the candidate list
|
|
18
|
+
* identical across prompts, which is the thing that must not vary. It also
|
|
19
|
+
* makes the cost proportional to the corpus rather than to the case count.
|
|
20
|
+
*/
|
|
21
|
+
import { type Model } from "@kolisachint/hoocode-ai";
|
|
22
|
+
import type { TriggerCandidate, TriggerJudge, TriggerJudgeVerdict } from "./trigger-eval.js";
|
|
23
|
+
/**
|
|
24
|
+
* Read the verdict list back, positionally.
|
|
25
|
+
*
|
|
26
|
+
* Returns one entry per prompt no matter what came back: a missing or
|
|
27
|
+
* unparseable row becomes null (read as "nothing fired") rather than shifting
|
|
28
|
+
* every later verdict onto the wrong prompt. `runTriggerEval` rejects a
|
|
29
|
+
* length mismatch outright, so the alternative to filling the gaps is
|
|
30
|
+
* discarding the whole run — and a hallucinated capability name is a real
|
|
31
|
+
* signal about the candidate list, not a reason to throw the batch away.
|
|
32
|
+
*/
|
|
33
|
+
export declare function parseTriggerVerdicts(response: string, candidates: readonly TriggerCandidate[], promptCount: number): TriggerJudgeVerdict[];
|
|
34
|
+
export interface TriggerJudgeDeps {
|
|
35
|
+
model: Model<any>;
|
|
36
|
+
apiKey?: string;
|
|
37
|
+
headers?: Record<string, string>;
|
|
38
|
+
signal?: AbortSignal;
|
|
39
|
+
}
|
|
40
|
+
/** A {@link TriggerJudge} backed by a real model. */
|
|
41
|
+
export declare function createLlmTriggerJudge(deps: TriggerJudgeDeps): TriggerJudge;
|
|
42
|
+
//# sourceMappingURL=trigger-judge.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"trigger-judge.d.ts","sourceRoot":"","sources":["../../../../src/core/extensions/plugins/trigger-judge.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;GAmBG;AAEH,OAAO,EAAkB,KAAK,KAAK,EAAE,MAAM,yBAAyB,CAAC;AACrE,OAAO,KAAK,EAAE,gBAAgB,EAAE,YAAY,EAAE,mBAAmB,EAAE,MAAM,mBAAmB,CAAC;AA+B7F;;;;;;;;;GASG;AACH,wBAAgB,oBAAoB,CACnC,QAAQ,EAAE,MAAM,EAChB,UAAU,EAAE,SAAS,gBAAgB,EAAE,EACvC,WAAW,EAAE,MAAM,GACjB,mBAAmB,EAAE,CAgCvB;AAED,MAAM,WAAW,gBAAgB;IAChC,KAAK,EAAE,KAAK,CAAC,GAAG,CAAC,CAAC;IAClB,MAAM,CAAC,EAAE,MAAM,CAAC;IAChB,OAAO,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;IACjC,MAAM,CAAC,EAAE,WAAW,CAAC;CACrB;AAED,qDAAqD;AACrD,wBAAgB,qBAAqB,CAAC,IAAI,EAAE,gBAAgB,GAAG,YAAY,CAgC1E","sourcesContent":["/**\n * The model call G4 was designed around and never had.\n *\n * `trigger-eval.ts` takes its judge as a parameter so scoring stays testable\n * without a model, and nothing in the tree ever passed one — so every G4 run\n * has reported `not-run` since it was written. This is that judge.\n *\n * It is deliberately in its own module rather than inside `trigger-eval.ts`:\n * the eval is pure and testable, this reaches the network, and the same\n * separation lets the agent-selection eval reuse the judge without pulling in\n * the plugin gate machinery.\n *\n * ## Why one call for all prompts\n *\n * The judge sees every candidate and every prompt at once. Per-prompt calls\n * would be cleaner to reason about, but the question being scored is\n * comparative — \"which of these fires\" — and batching keeps the candidate list\n * identical across prompts, which is the thing that must not vary. It also\n * makes the cost proportional to the corpus rather than to the case count.\n */\n\nimport { completeSimple, type Model } from \"@kolisachint/hoocode-ai\";\nimport type { TriggerCandidate, TriggerJudge, TriggerJudgeVerdict } from \"./trigger-eval.js\";\n\n/** Description characters per candidate. The opening states the trigger; past that it is filler. */\nconst DESCRIPTION_CHARS = 600;\nconst MAX_RESPONSE_TOKENS = 4_000;\n\nconst JUDGE_SYSTEM_PROMPT = `You simulate how a coding agent picks a capability.\n\nYou are given numbered CAPABILITIES (name and description) and numbered PROMPTS. For each prompt, answer with the ONE capability an agent would reach for, or null when none of them fits and the agent should just do the work itself.\n\nRules:\n- Judge ONLY from the descriptions. Do not use knowledge about what these names usually mean elsewhere.\n- Pick the single best fit. When two fit, pick the one whose description names the prompt's situation more specifically.\n- Answer null when no description covers the prompt. Do not stretch a description to make it fit; a wrong pick and a null are both wrong, but pretending coverage hides the real failure.\n- Answer every prompt exactly once, in the order given.\n\nOutput STRICT JSON, no markdown fence, no prose:\n{\"verdicts\":[{\"prompt\":0,\"capability\":\"explore\"},{\"prompt\":1,\"capability\":null}]}`;\n\nfunction buildPrompt(candidates: readonly TriggerCandidate[], prompts: readonly string[]): string {\n\tconst lines: string[] = [\"CAPABILITIES:\"];\n\tfor (const [i, candidate] of candidates.entries()) {\n\t\tlines.push(`${i}. ${candidate.name} — ${candidate.description.slice(0, DESCRIPTION_CHARS)}`);\n\t}\n\tlines.push(\"\", \"PROMPTS:\");\n\tfor (const [i, prompt] of prompts.entries()) {\n\t\tlines.push(`${i}. ${prompt}`);\n\t}\n\treturn lines.join(\"\\n\");\n}\n\n/**\n * Read the verdict list back, positionally.\n *\n * Returns one entry per prompt no matter what came back: a missing or\n * unparseable row becomes null (read as \"nothing fired\") rather than shifting\n * every later verdict onto the wrong prompt. `runTriggerEval` rejects a\n * length mismatch outright, so the alternative to filling the gaps is\n * discarding the whole run — and a hallucinated capability name is a real\n * signal about the candidate list, not a reason to throw the batch away.\n */\nexport function parseTriggerVerdicts(\n\tresponse: string,\n\tcandidates: readonly TriggerCandidate[],\n\tpromptCount: number,\n): TriggerJudgeVerdict[] {\n\tconst verdicts: TriggerJudgeVerdict[] = new Array(promptCount).fill(null);\n\tconst start = response.indexOf(\"{\");\n\tconst end = response.lastIndexOf(\"}\");\n\tif (start < 0 || end <= start) return verdicts;\n\n\tlet parsed: unknown;\n\ttry {\n\t\tparsed = JSON.parse(response.slice(start, end + 1));\n\t} catch {\n\t\treturn verdicts;\n\t}\n\n\tconst rows = (parsed as { verdicts?: unknown })?.verdicts;\n\tif (!Array.isArray(rows)) return verdicts;\n\n\tconst known = new Set(candidates.map((c) => c.name));\n\tfor (const row of rows) {\n\t\tif (!row || typeof row !== \"object\") continue;\n\t\tconst entry = row as Record<string, unknown>;\n\t\tconst index = typeof entry.prompt === \"number\" ? entry.prompt : Number.NaN;\n\t\tif (!Number.isInteger(index) || index < 0 || index >= promptCount) continue;\n\t\tconst capability = entry.capability;\n\t\tif (capability === null || capability === undefined) {\n\t\t\tverdicts[index] = null;\n\t\t\tcontinue;\n\t\t}\n\t\t// A name that is not on the candidate list is a hallucination. Recording it\n\t\t// as null keeps it wrong without letting it masquerade as a real pick.\n\t\tverdicts[index] = typeof capability === \"string\" && known.has(capability) ? capability : null;\n\t}\n\treturn verdicts;\n}\n\nexport interface TriggerJudgeDeps {\n\tmodel: Model<any>;\n\tapiKey?: string;\n\theaders?: Record<string, string>;\n\tsignal?: AbortSignal;\n}\n\n/** A {@link TriggerJudge} backed by a real model. */\nexport function createLlmTriggerJudge(deps: TriggerJudgeDeps): TriggerJudge {\n\treturn async ({ candidates, prompts }) => {\n\t\tif (candidates.length === 0 || prompts.length === 0) return prompts.map(() => null);\n\n\t\tconst response = await completeSimple(\n\t\t\tdeps.model,\n\t\t\t{\n\t\t\t\tsystemPrompt: JUDGE_SYSTEM_PROMPT,\n\t\t\t\tmessages: [\n\t\t\t\t\t{\n\t\t\t\t\t\trole: \"user\",\n\t\t\t\t\t\tcontent: [{ type: \"text\", text: buildPrompt(candidates, prompts) }],\n\t\t\t\t\t\ttimestamp: Date.now(),\n\t\t\t\t\t},\n\t\t\t\t],\n\t\t\t},\n\t\t\t{ maxTokens: MAX_RESPONSE_TOKENS, signal: deps.signal, apiKey: deps.apiKey, headers: deps.headers },\n\t\t);\n\n\t\tif (response.stopReason === \"error\") {\n\t\t\t// Thrown, not swallowed: runTriggerEval turns this into `not-run` with\n\t\t\t// the reason attached, which is the honest outcome. Returning nulls\n\t\t\t// would score every case as \"nothing fired\" and read like a real result.\n\t\t\tthrow new Error(response.errorMessage || \"trigger judge call failed\");\n\t\t}\n\n\t\tconst text = response.content\n\t\t\t.filter((c): c is { type: \"text\"; text: string } => c.type === \"text\")\n\t\t\t.map((c) => c.text)\n\t\t\t.join(\"\\n\");\n\t\treturn parseTriggerVerdicts(text, candidates, prompts.length);\n\t};\n}\n"]}
|
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The model call G4 was designed around and never had.
|
|
3
|
+
*
|
|
4
|
+
* `trigger-eval.ts` takes its judge as a parameter so scoring stays testable
|
|
5
|
+
* without a model, and nothing in the tree ever passed one — so every G4 run
|
|
6
|
+
* has reported `not-run` since it was written. This is that judge.
|
|
7
|
+
*
|
|
8
|
+
* It is deliberately in its own module rather than inside `trigger-eval.ts`:
|
|
9
|
+
* the eval is pure and testable, this reaches the network, and the same
|
|
10
|
+
* separation lets the agent-selection eval reuse the judge without pulling in
|
|
11
|
+
* the plugin gate machinery.
|
|
12
|
+
*
|
|
13
|
+
* ## Why one call for all prompts
|
|
14
|
+
*
|
|
15
|
+
* The judge sees every candidate and every prompt at once. Per-prompt calls
|
|
16
|
+
* would be cleaner to reason about, but the question being scored is
|
|
17
|
+
* comparative — "which of these fires" — and batching keeps the candidate list
|
|
18
|
+
* identical across prompts, which is the thing that must not vary. It also
|
|
19
|
+
* makes the cost proportional to the corpus rather than to the case count.
|
|
20
|
+
*/
|
|
21
|
+
import { completeSimple } from "@kolisachint/hoocode-ai";
|
|
22
|
+
/** Description characters per candidate. The opening states the trigger; past that it is filler. */
|
|
23
|
+
const DESCRIPTION_CHARS = 600;
|
|
24
|
+
const MAX_RESPONSE_TOKENS = 4_000;
|
|
25
|
+
const JUDGE_SYSTEM_PROMPT = `You simulate how a coding agent picks a capability.
|
|
26
|
+
|
|
27
|
+
You are given numbered CAPABILITIES (name and description) and numbered PROMPTS. For each prompt, answer with the ONE capability an agent would reach for, or null when none of them fits and the agent should just do the work itself.
|
|
28
|
+
|
|
29
|
+
Rules:
|
|
30
|
+
- Judge ONLY from the descriptions. Do not use knowledge about what these names usually mean elsewhere.
|
|
31
|
+
- Pick the single best fit. When two fit, pick the one whose description names the prompt's situation more specifically.
|
|
32
|
+
- Answer null when no description covers the prompt. Do not stretch a description to make it fit; a wrong pick and a null are both wrong, but pretending coverage hides the real failure.
|
|
33
|
+
- Answer every prompt exactly once, in the order given.
|
|
34
|
+
|
|
35
|
+
Output STRICT JSON, no markdown fence, no prose:
|
|
36
|
+
{"verdicts":[{"prompt":0,"capability":"explore"},{"prompt":1,"capability":null}]}`;
|
|
37
|
+
function buildPrompt(candidates, prompts) {
|
|
38
|
+
const lines = ["CAPABILITIES:"];
|
|
39
|
+
for (const [i, candidate] of candidates.entries()) {
|
|
40
|
+
lines.push(`${i}. ${candidate.name} — ${candidate.description.slice(0, DESCRIPTION_CHARS)}`);
|
|
41
|
+
}
|
|
42
|
+
lines.push("", "PROMPTS:");
|
|
43
|
+
for (const [i, prompt] of prompts.entries()) {
|
|
44
|
+
lines.push(`${i}. ${prompt}`);
|
|
45
|
+
}
|
|
46
|
+
return lines.join("\n");
|
|
47
|
+
}
|
|
48
|
+
/**
|
|
49
|
+
* Read the verdict list back, positionally.
|
|
50
|
+
*
|
|
51
|
+
* Returns one entry per prompt no matter what came back: a missing or
|
|
52
|
+
* unparseable row becomes null (read as "nothing fired") rather than shifting
|
|
53
|
+
* every later verdict onto the wrong prompt. `runTriggerEval` rejects a
|
|
54
|
+
* length mismatch outright, so the alternative to filling the gaps is
|
|
55
|
+
* discarding the whole run — and a hallucinated capability name is a real
|
|
56
|
+
* signal about the candidate list, not a reason to throw the batch away.
|
|
57
|
+
*/
|
|
58
|
+
export function parseTriggerVerdicts(response, candidates, promptCount) {
|
|
59
|
+
const verdicts = new Array(promptCount).fill(null);
|
|
60
|
+
const start = response.indexOf("{");
|
|
61
|
+
const end = response.lastIndexOf("}");
|
|
62
|
+
if (start < 0 || end <= start)
|
|
63
|
+
return verdicts;
|
|
64
|
+
let parsed;
|
|
65
|
+
try {
|
|
66
|
+
parsed = JSON.parse(response.slice(start, end + 1));
|
|
67
|
+
}
|
|
68
|
+
catch {
|
|
69
|
+
return verdicts;
|
|
70
|
+
}
|
|
71
|
+
const rows = parsed?.verdicts;
|
|
72
|
+
if (!Array.isArray(rows))
|
|
73
|
+
return verdicts;
|
|
74
|
+
const known = new Set(candidates.map((c) => c.name));
|
|
75
|
+
for (const row of rows) {
|
|
76
|
+
if (!row || typeof row !== "object")
|
|
77
|
+
continue;
|
|
78
|
+
const entry = row;
|
|
79
|
+
const index = typeof entry.prompt === "number" ? entry.prompt : Number.NaN;
|
|
80
|
+
if (!Number.isInteger(index) || index < 0 || index >= promptCount)
|
|
81
|
+
continue;
|
|
82
|
+
const capability = entry.capability;
|
|
83
|
+
if (capability === null || capability === undefined) {
|
|
84
|
+
verdicts[index] = null;
|
|
85
|
+
continue;
|
|
86
|
+
}
|
|
87
|
+
// A name that is not on the candidate list is a hallucination. Recording it
|
|
88
|
+
// as null keeps it wrong without letting it masquerade as a real pick.
|
|
89
|
+
verdicts[index] = typeof capability === "string" && known.has(capability) ? capability : null;
|
|
90
|
+
}
|
|
91
|
+
return verdicts;
|
|
92
|
+
}
|
|
93
|
+
/** A {@link TriggerJudge} backed by a real model. */
|
|
94
|
+
export function createLlmTriggerJudge(deps) {
|
|
95
|
+
return async ({ candidates, prompts }) => {
|
|
96
|
+
if (candidates.length === 0 || prompts.length === 0)
|
|
97
|
+
return prompts.map(() => null);
|
|
98
|
+
const response = await completeSimple(deps.model, {
|
|
99
|
+
systemPrompt: JUDGE_SYSTEM_PROMPT,
|
|
100
|
+
messages: [
|
|
101
|
+
{
|
|
102
|
+
role: "user",
|
|
103
|
+
content: [{ type: "text", text: buildPrompt(candidates, prompts) }],
|
|
104
|
+
timestamp: Date.now(),
|
|
105
|
+
},
|
|
106
|
+
],
|
|
107
|
+
}, { maxTokens: MAX_RESPONSE_TOKENS, signal: deps.signal, apiKey: deps.apiKey, headers: deps.headers });
|
|
108
|
+
if (response.stopReason === "error") {
|
|
109
|
+
// Thrown, not swallowed: runTriggerEval turns this into `not-run` with
|
|
110
|
+
// the reason attached, which is the honest outcome. Returning nulls
|
|
111
|
+
// would score every case as "nothing fired" and read like a real result.
|
|
112
|
+
throw new Error(response.errorMessage || "trigger judge call failed");
|
|
113
|
+
}
|
|
114
|
+
const text = response.content
|
|
115
|
+
.filter((c) => c.type === "text")
|
|
116
|
+
.map((c) => c.text)
|
|
117
|
+
.join("\n");
|
|
118
|
+
return parseTriggerVerdicts(text, candidates, prompts.length);
|
|
119
|
+
};
|
|
120
|
+
}
|
|
121
|
+
//# sourceMappingURL=trigger-judge.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"trigger-judge.js","sourceRoot":"","sources":["../../../../src/core/extensions/plugins/trigger-judge.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;GAmBG;AAEH,OAAO,EAAE,cAAc,EAAc,MAAM,yBAAyB,CAAC;AAGrE,oGAAoG;AACpG,MAAM,iBAAiB,GAAG,GAAG,CAAC;AAC9B,MAAM,mBAAmB,GAAG,KAAK,CAAC;AAElC,MAAM,mBAAmB,GAAG;;;;;;;;;;;kFAWsD,CAAC;AAEnF,SAAS,WAAW,CAAC,UAAuC,EAAE,OAA0B,EAAU;IACjG,MAAM,KAAK,GAAa,CAAC,eAAe,CAAC,CAAC;IAC1C,KAAK,MAAM,CAAC,CAAC,EAAE,SAAS,CAAC,IAAI,UAAU,CAAC,OAAO,EAAE,EAAE,CAAC;QACnD,KAAK,CAAC,IAAI,CAAC,GAAG,CAAC,KAAK,SAAS,CAAC,IAAI,QAAM,SAAS,CAAC,WAAW,CAAC,KAAK,CAAC,CAAC,EAAE,iBAAiB,CAAC,EAAE,CAAC,CAAC;IAC9F,CAAC;IACD,KAAK,CAAC,IAAI,CAAC,EAAE,EAAE,UAAU,CAAC,CAAC;IAC3B,KAAK,MAAM,CAAC,CAAC,EAAE,MAAM,CAAC,IAAI,OAAO,CAAC,OAAO,EAAE,EAAE,CAAC;QAC7C,KAAK,CAAC,IAAI,CAAC,GAAG,CAAC,KAAK,MAAM,EAAE,CAAC,CAAC;IAC/B,CAAC;IACD,OAAO,KAAK,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;AAAA,CACxB;AAED;;;;;;;;;GASG;AACH,MAAM,UAAU,oBAAoB,CACnC,QAAgB,EAChB,UAAuC,EACvC,WAAmB,EACK;IACxB,MAAM,QAAQ,GAA0B,IAAI,KAAK,CAAC,WAAW,CAAC,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;IAC1E,MAAM,KAAK,GAAG,QAAQ,CAAC,OAAO,CAAC,GAAG,CAAC,CAAC;IACpC,MAAM,GAAG,GAAG,QAAQ,CAAC,WAAW,CAAC,GAAG,CAAC,CAAC;IACtC,IAAI,KAAK,GAAG,CAAC,IAAI,GAAG,IAAI,KAAK;QAAE,OAAO,QAAQ,CAAC;IAE/C,IAAI,MAAe,CAAC;IACpB,IAAI,CAAC;QACJ,MAAM,GAAG,IAAI,CAAC,KAAK,CAAC,QAAQ,CAAC,KAAK,CAAC,KAAK,EAAE,GAAG,GAAG,CAAC,CAAC,CAAC,CAAC;IACrD,CAAC;IAAC,MAAM,CAAC;QACR,OAAO,QAAQ,CAAC;IACjB,CAAC;IAED,MAAM,IAAI,GAAI,MAAiC,EAAE,QAAQ,CAAC;IAC1D,IAAI,CAAC,KAAK,CAAC,OAAO,CAAC,IAAI,CAAC;QAAE,OAAO,QAAQ,CAAC;IAE1C,MAAM,KAAK,GAAG,IAAI,GAAG,CAAC,UAAU,CAAC,GAAG,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC;IACrD,KAAK,MAAM,GAAG,IAAI,IAAI,EAAE,CAAC;QACxB,IAAI,CAAC,GAAG,IAAI,OAAO,GAAG,KAAK,QAAQ;YAAE,SAAS;QAC9C,MAAM,KAAK,GAAG,GAA8B,CAAC;QAC7C,MAAM,KAAK,GAAG,OAAO,KAAK,CAAC,MAAM,KAAK,QAAQ,CAAC,CAAC,CAAC,KAAK,CAAC,MAAM,CAAC,CAAC,CAAC,MAAM,CAAC,GAAG,CAAC;QAC3E,IAAI,CAAC,MAAM,CAAC,SAAS,CAAC,KAAK,CAAC,IAAI,KAAK,GAAG,CAAC,IAAI,KAAK,IAAI,WAAW;YAAE,SAAS;QAC5E,MAAM,UAAU,GAAG,KAAK,CAAC,UAAU,CAAC;QACpC,IAAI,UAAU,KAAK,IAAI,IAAI,UAAU,KAAK,SAAS,EAAE,CAAC;YACrD,QAAQ,CAAC,KAAK,CAAC,GAAG,IAAI,CAAC;YACvB,SAAS;QACV,CAAC;QACD,4EAA4E;QAC5E,uEAAuE;QACvE,QAAQ,CAAC,KAAK,CAAC,GAAG,OAAO,UAAU,KAAK,QAAQ,IAAI,KAAK,CAAC,GAAG,CAAC,UAAU,CAAC,CAAC,CAAC,CAAC,UAAU,CAAC,CAAC,CAAC,IAAI,CAAC;IAC/F,CAAC;IACD,OAAO,QAAQ,CAAC;AAAA,CAChB;AASD,qDAAqD;AACrD,MAAM,UAAU,qBAAqB,CAAC,IAAsB,EAAgB;IAC3E,OAAO,KAAK,EAAE,EAAE,UAAU,EAAE,OAAO,EAAE,EAAE,EAAE,CAAC;QACzC,IAAI,UAAU,CAAC,MAAM,KAAK,CAAC,IAAI,OAAO,CAAC,MAAM,KAAK,CAAC;YAAE,OAAO,OAAO,CAAC,GAAG,CAAC,GAAG,EAAE,CAAC,IAAI,CAAC,CAAC;QAEpF,MAAM,QAAQ,GAAG,MAAM,cAAc,CACpC,IAAI,CAAC,KAAK,EACV;YACC,YAAY,EAAE,mBAAmB;YACjC,QAAQ,EAAE;gBACT;oBACC,IAAI,EAAE,MAAM;oBACZ,OAAO,EAAE,CAAC,EAAE,IAAI,EAAE,MAAM,EAAE,IAAI,EAAE,WAAW,CAAC,UAAU,EAAE,OAAO,CAAC,EAAE,CAAC;oBACnE,SAAS,EAAE,IAAI,CAAC,GAAG,EAAE;iBACrB;aACD;SACD,EACD,EAAE,SAAS,EAAE,mBAAmB,EAAE,MAAM,EAAE,IAAI,CAAC,MAAM,EAAE,MAAM,EAAE,IAAI,CAAC,MAAM,EAAE,OAAO,EAAE,IAAI,CAAC,OAAO,EAAE,CACnG,CAAC;QAEF,IAAI,QAAQ,CAAC,UAAU,KAAK,OAAO,EAAE,CAAC;YACrC,uEAAuE;YACvE,oEAAoE;YACpE,yEAAyE;YACzE,MAAM,IAAI,KAAK,CAAC,QAAQ,CAAC,YAAY,IAAI,2BAA2B,CAAC,CAAC;QACvE,CAAC;QAED,MAAM,IAAI,GAAG,QAAQ,CAAC,OAAO;aAC3B,MAAM,CAAC,CAAC,CAAC,EAAuC,EAAE,CAAC,CAAC,CAAC,IAAI,KAAK,MAAM,CAAC;aACrE,GAAG,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,IAAI,CAAC;aAClB,IAAI,CAAC,IAAI,CAAC,CAAC;QACb,OAAO,oBAAoB,CAAC,IAAI,EAAE,UAAU,EAAE,OAAO,CAAC,MAAM,CAAC,CAAC;IAAA,CAC9D,CAAC;AAAA,CACF","sourcesContent":["/**\n * The model call G4 was designed around and never had.\n *\n * `trigger-eval.ts` takes its judge as a parameter so scoring stays testable\n * without a model, and nothing in the tree ever passed one — so every G4 run\n * has reported `not-run` since it was written. This is that judge.\n *\n * It is deliberately in its own module rather than inside `trigger-eval.ts`:\n * the eval is pure and testable, this reaches the network, and the same\n * separation lets the agent-selection eval reuse the judge without pulling in\n * the plugin gate machinery.\n *\n * ## Why one call for all prompts\n *\n * The judge sees every candidate and every prompt at once. Per-prompt calls\n * would be cleaner to reason about, but the question being scored is\n * comparative — \"which of these fires\" — and batching keeps the candidate list\n * identical across prompts, which is the thing that must not vary. It also\n * makes the cost proportional to the corpus rather than to the case count.\n */\n\nimport { completeSimple, type Model } from \"@kolisachint/hoocode-ai\";\nimport type { TriggerCandidate, TriggerJudge, TriggerJudgeVerdict } from \"./trigger-eval.js\";\n\n/** Description characters per candidate. The opening states the trigger; past that it is filler. */\nconst DESCRIPTION_CHARS = 600;\nconst MAX_RESPONSE_TOKENS = 4_000;\n\nconst JUDGE_SYSTEM_PROMPT = `You simulate how a coding agent picks a capability.\n\nYou are given numbered CAPABILITIES (name and description) and numbered PROMPTS. For each prompt, answer with the ONE capability an agent would reach for, or null when none of them fits and the agent should just do the work itself.\n\nRules:\n- Judge ONLY from the descriptions. Do not use knowledge about what these names usually mean elsewhere.\n- Pick the single best fit. When two fit, pick the one whose description names the prompt's situation more specifically.\n- Answer null when no description covers the prompt. Do not stretch a description to make it fit; a wrong pick and a null are both wrong, but pretending coverage hides the real failure.\n- Answer every prompt exactly once, in the order given.\n\nOutput STRICT JSON, no markdown fence, no prose:\n{\"verdicts\":[{\"prompt\":0,\"capability\":\"explore\"},{\"prompt\":1,\"capability\":null}]}`;\n\nfunction buildPrompt(candidates: readonly TriggerCandidate[], prompts: readonly string[]): string {\n\tconst lines: string[] = [\"CAPABILITIES:\"];\n\tfor (const [i, candidate] of candidates.entries()) {\n\t\tlines.push(`${i}. ${candidate.name} — ${candidate.description.slice(0, DESCRIPTION_CHARS)}`);\n\t}\n\tlines.push(\"\", \"PROMPTS:\");\n\tfor (const [i, prompt] of prompts.entries()) {\n\t\tlines.push(`${i}. ${prompt}`);\n\t}\n\treturn lines.join(\"\\n\");\n}\n\n/**\n * Read the verdict list back, positionally.\n *\n * Returns one entry per prompt no matter what came back: a missing or\n * unparseable row becomes null (read as \"nothing fired\") rather than shifting\n * every later verdict onto the wrong prompt. `runTriggerEval` rejects a\n * length mismatch outright, so the alternative to filling the gaps is\n * discarding the whole run — and a hallucinated capability name is a real\n * signal about the candidate list, not a reason to throw the batch away.\n */\nexport function parseTriggerVerdicts(\n\tresponse: string,\n\tcandidates: readonly TriggerCandidate[],\n\tpromptCount: number,\n): TriggerJudgeVerdict[] {\n\tconst verdicts: TriggerJudgeVerdict[] = new Array(promptCount).fill(null);\n\tconst start = response.indexOf(\"{\");\n\tconst end = response.lastIndexOf(\"}\");\n\tif (start < 0 || end <= start) return verdicts;\n\n\tlet parsed: unknown;\n\ttry {\n\t\tparsed = JSON.parse(response.slice(start, end + 1));\n\t} catch {\n\t\treturn verdicts;\n\t}\n\n\tconst rows = (parsed as { verdicts?: unknown })?.verdicts;\n\tif (!Array.isArray(rows)) return verdicts;\n\n\tconst known = new Set(candidates.map((c) => c.name));\n\tfor (const row of rows) {\n\t\tif (!row || typeof row !== \"object\") continue;\n\t\tconst entry = row as Record<string, unknown>;\n\t\tconst index = typeof entry.prompt === \"number\" ? entry.prompt : Number.NaN;\n\t\tif (!Number.isInteger(index) || index < 0 || index >= promptCount) continue;\n\t\tconst capability = entry.capability;\n\t\tif (capability === null || capability === undefined) {\n\t\t\tverdicts[index] = null;\n\t\t\tcontinue;\n\t\t}\n\t\t// A name that is not on the candidate list is a hallucination. Recording it\n\t\t// as null keeps it wrong without letting it masquerade as a real pick.\n\t\tverdicts[index] = typeof capability === \"string\" && known.has(capability) ? capability : null;\n\t}\n\treturn verdicts;\n}\n\nexport interface TriggerJudgeDeps {\n\tmodel: Model<any>;\n\tapiKey?: string;\n\theaders?: Record<string, string>;\n\tsignal?: AbortSignal;\n}\n\n/** A {@link TriggerJudge} backed by a real model. */\nexport function createLlmTriggerJudge(deps: TriggerJudgeDeps): TriggerJudge {\n\treturn async ({ candidates, prompts }) => {\n\t\tif (candidates.length === 0 || prompts.length === 0) return prompts.map(() => null);\n\n\t\tconst response = await completeSimple(\n\t\t\tdeps.model,\n\t\t\t{\n\t\t\t\tsystemPrompt: JUDGE_SYSTEM_PROMPT,\n\t\t\t\tmessages: [\n\t\t\t\t\t{\n\t\t\t\t\t\trole: \"user\",\n\t\t\t\t\t\tcontent: [{ type: \"text\", text: buildPrompt(candidates, prompts) }],\n\t\t\t\t\t\ttimestamp: Date.now(),\n\t\t\t\t\t},\n\t\t\t\t],\n\t\t\t},\n\t\t\t{ maxTokens: MAX_RESPONSE_TOKENS, signal: deps.signal, apiKey: deps.apiKey, headers: deps.headers },\n\t\t);\n\n\t\tif (response.stopReason === \"error\") {\n\t\t\t// Thrown, not swallowed: runTriggerEval turns this into `not-run` with\n\t\t\t// the reason attached, which is the honest outcome. Returning nulls\n\t\t\t// would score every case as \"nothing fired\" and read like a real result.\n\t\t\tthrow new Error(response.errorMessage || \"trigger judge call failed\");\n\t\t}\n\n\t\tconst text = response.content\n\t\t\t.filter((c): c is { type: \"text\"; text: string } => c.type === \"text\")\n\t\t\t.map((c) => c.text)\n\t\t\t.join(\"\\n\");\n\t\treturn parseTriggerVerdicts(text, candidates, prompts.length);\n\t};\n}\n"]}
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The external (Rust) binaries hoocode can use, and what each one is worth.
|
|
3
|
+
*
|
|
4
|
+
* hoocode is self-sufficient without any of them: `grep`/`find`/`search` fall
|
|
5
|
+
* back to pure-JS implementations, and the features that have no fallback
|
|
6
|
+
* (web, voice) are off or inert rather than broken. These binaries are an
|
|
7
|
+
* *expansion* layer, which is exactly why they were invisible - nothing failed
|
|
8
|
+
* loudly enough to tell anyone they existed.
|
|
9
|
+
*
|
|
10
|
+
* This module is the single description of that layer. The `/settings` pane
|
|
11
|
+
* renders it; the same table says which settings rows are gated on a binary, so
|
|
12
|
+
* a row that cannot do anything today can say so instead of lying.
|
|
13
|
+
*/
|
|
14
|
+
import { type ManagedTool, type ManagedToolStatus } from "../utils/tools-manager.js";
|
|
15
|
+
/** How hoocode gets a binary it does not already have. */
|
|
16
|
+
export type Acquisition =
|
|
17
|
+
/** Fetched in the background at startup. */
|
|
18
|
+
"startup"
|
|
19
|
+
/** Fetched the first time the feature is actually used. */
|
|
20
|
+
| "on-demand"
|
|
21
|
+
/** Never fetched implicitly; install it yourself or point the env var at it. */
|
|
22
|
+
| "manual";
|
|
23
|
+
export interface ExternalToolDoc {
|
|
24
|
+
tool: ManagedTool;
|
|
25
|
+
/** Row label in the pane. */
|
|
26
|
+
label: string;
|
|
27
|
+
/** What the binary does for hoocode, in one line. */
|
|
28
|
+
summary: string;
|
|
29
|
+
/** What it turns on, feature by feature. */
|
|
30
|
+
enables: string[];
|
|
31
|
+
/** What hoocode does instead when it is missing. Never "nothing works". */
|
|
32
|
+
fallback: string;
|
|
33
|
+
acquisition: Acquisition;
|
|
34
|
+
/** Env vars that change how this binary is resolved or driven. */
|
|
35
|
+
env: string[];
|
|
36
|
+
/**
|
|
37
|
+
* `/settings` row ids whose effect depends on this binary. A row listed here
|
|
38
|
+
* is annotated (not hidden) while the binary is missing: hiding it would
|
|
39
|
+
* recreate the discoverability hole this whole surface exists to close.
|
|
40
|
+
*/
|
|
41
|
+
dependentRows: string[];
|
|
42
|
+
/** settings.json keys this binary gates, for docs and search. */
|
|
43
|
+
settingsKeys: string[];
|
|
44
|
+
}
|
|
45
|
+
/**
|
|
46
|
+
* Ordered so the two that silently make everything faster come first, then the
|
|
47
|
+
* three that add capability hoocode does not otherwise have.
|
|
48
|
+
*/
|
|
49
|
+
export declare const EXTERNAL_TOOLS: readonly ExternalToolDoc[];
|
|
50
|
+
export interface ExternalToolStatus extends ExternalToolDoc, ManagedToolStatus {
|
|
51
|
+
installed: boolean;
|
|
52
|
+
/**
|
|
53
|
+
* Whether hoocode would fetch it if the feature were used right now. False in
|
|
54
|
+
* offline mode, and on Android where the published Linux builds do not run.
|
|
55
|
+
*/
|
|
56
|
+
downloadable: boolean;
|
|
57
|
+
}
|
|
58
|
+
/** Short, fixed-width-ish status word for the pane's value column. */
|
|
59
|
+
export declare function statusLabel(status: ExternalToolStatus): string;
|
|
60
|
+
/**
|
|
61
|
+
* Resolve every external tool. Never downloads: this is what the pane shows
|
|
62
|
+
* *before* the user opts into anything.
|
|
63
|
+
*/
|
|
64
|
+
export declare function describeExternalTools(): ExternalToolStatus[];
|
|
65
|
+
/** Index of pane-row id -> the tool it needs, for annotating gated rows. */
|
|
66
|
+
export declare function buildRowGates(statuses: readonly ExternalToolStatus[]): Map<string, ExternalToolStatus>;
|
|
67
|
+
//# sourceMappingURL=external-tools.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"external-tools.d.ts","sourceRoot":"","sources":["../../src/core/external-tools.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;GAYG;AAEH,OAAO,EAAgC,KAAK,WAAW,EAAE,KAAK,iBAAiB,EAAE,MAAM,2BAA2B,CAAC;AAEnH,0DAA0D;AAC1D,MAAM,MAAM,WAAW;AACtB,4CAA4C;AAC1C,SAAS;AACX,2DAA2D;GACzD,WAAW;AACb,gFAAgF;GAC9E,QAAQ,CAAC;AAEZ,MAAM,WAAW,eAAe;IAC/B,IAAI,EAAE,WAAW,CAAC;IAClB,6BAA6B;IAC7B,KAAK,EAAE,MAAM,CAAC;IACd,qDAAqD;IACrD,OAAO,EAAE,MAAM,CAAC;IAChB,4CAA4C;IAC5C,OAAO,EAAE,MAAM,EAAE,CAAC;IAClB,2EAA2E;IAC3E,QAAQ,EAAE,MAAM,CAAC;IACjB,WAAW,EAAE,WAAW,CAAC;IACzB,kEAAkE;IAClE,GAAG,EAAE,MAAM,EAAE,CAAC;IACd;;;;OAIG;IACH,aAAa,EAAE,MAAM,EAAE,CAAC;IACxB,iEAAiE;IACjE,YAAY,EAAE,MAAM,EAAE,CAAC;CACvB;AAED;;;GAGG;AACH,eAAO,MAAM,cAAc,EAAE,SAAS,eAAe,EA+DpD,CAAC;AAEF,MAAM,WAAW,kBAAmB,SAAQ,eAAe,EAAE,iBAAiB;IAC7E,SAAS,EAAE,OAAO,CAAC;IACnB;;;OAGG;IACH,YAAY,EAAE,OAAO,CAAC;CACtB;AAED,sEAAsE;AACtE,wBAAgB,WAAW,CAAC,MAAM,EAAE,kBAAkB,GAAG,MAAM,CAY9D;AAED;;;GAGG;AACH,wBAAgB,qBAAqB,IAAI,kBAAkB,EAAE,CAY5D;AAED,4EAA4E;AAC5E,wBAAgB,aAAa,CAAC,QAAQ,EAAE,SAAS,kBAAkB,EAAE,GAAG,GAAG,CAAC,MAAM,EAAE,kBAAkB,CAAC,CAMtG","sourcesContent":["/**\n * The external (Rust) binaries hoocode can use, and what each one is worth.\n *\n * hoocode is self-sufficient without any of them: `grep`/`find`/`search` fall\n * back to pure-JS implementations, and the features that have no fallback\n * (web, voice) are off or inert rather than broken. These binaries are an\n * *expansion* layer, which is exactly why they were invisible - nothing failed\n * loudly enough to tell anyone they existed.\n *\n * This module is the single description of that layer. The `/settings` pane\n * renders it; the same table says which settings rows are gated on a binary, so\n * a row that cannot do anything today can say so instead of lying.\n */\n\nimport { getToolStatus, isOfflineMode, type ManagedTool, type ManagedToolStatus } from \"../utils/tools-manager.js\";\n\n/** How hoocode gets a binary it does not already have. */\nexport type Acquisition =\n\t/** Fetched in the background at startup. */\n\t| \"startup\"\n\t/** Fetched the first time the feature is actually used. */\n\t| \"on-demand\"\n\t/** Never fetched implicitly; install it yourself or point the env var at it. */\n\t| \"manual\";\n\nexport interface ExternalToolDoc {\n\ttool: ManagedTool;\n\t/** Row label in the pane. */\n\tlabel: string;\n\t/** What the binary does for hoocode, in one line. */\n\tsummary: string;\n\t/** What it turns on, feature by feature. */\n\tenables: string[];\n\t/** What hoocode does instead when it is missing. Never \"nothing works\". */\n\tfallback: string;\n\tacquisition: Acquisition;\n\t/** Env vars that change how this binary is resolved or driven. */\n\tenv: string[];\n\t/**\n\t * `/settings` row ids whose effect depends on this binary. A row listed here\n\t * is annotated (not hidden) while the binary is missing: hiding it would\n\t * recreate the discoverability hole this whole surface exists to close.\n\t */\n\tdependentRows: string[];\n\t/** settings.json keys this binary gates, for docs and search. */\n\tsettingsKeys: string[];\n}\n\n/**\n * Ordered so the two that silently make everything faster come first, then the\n * three that add capability hoocode does not otherwise have.\n */\nexport const EXTERNAL_TOOLS: readonly ExternalToolDoc[] = [\n\t{\n\t\ttool: \"rg\",\n\t\tlabel: \"ripgrep (rg)\",\n\t\tsummary: \"Fast path for content search.\",\n\t\tenables: [\"grep runs rg instead of the JS scanner\", \"the lexical half of search runs rg\"],\n\t\tfallback:\n\t\t\t\"A pure-JS scanner produces the same match shape, so results are identical - it is materially slower on large trees and respects fewer ignore-file edge cases.\",\n\t\tacquisition: \"startup\",\n\t\tenv: [\"HOOCODE_RG_BINARY\", \"HOOCODE_NATIVE_SEARCH=1 forces the JS path even when rg is present\"],\n\t\tdependentRows: [],\n\t\tsettingsKeys: [],\n\t},\n\t{\n\t\ttool: \"fd\",\n\t\tlabel: \"fd\",\n\t\tsummary: \"Fast path for filename search.\",\n\t\tenables: [\"find runs fd instead of the JS directory walker\"],\n\t\tfallback:\n\t\t\t\"A JS walker produces the same result shape - slower on large trees, and glob/ignore handling is the JS approximation rather than fd's.\",\n\t\tacquisition: \"startup\",\n\t\tenv: [\"HOOCODE_FD_BINARY\", \"HOOCODE_NATIVE_SEARCH=1 forces the JS path even when fd is present\"],\n\t\tdependentRows: [],\n\t\tsettingsKeys: [],\n\t},\n\t{\n\t\ttool: \"embsearch\",\n\t\tlabel: \"embsearch (semantic index)\",\n\t\tsummary: \"Local embedding index. The only source of semantic ranking in hoocode.\",\n\t\tenables: [\n\t\t\t\"search fuses semantic hits with its lexical hits\",\n\t\t\t\"MCP/capability deferral ranks tools by meaning rather than keyword\",\n\t\t],\n\t\tfallback:\n\t\t\t\"search is lexical-only and capability lookup ranks lexically. Nothing errors; queries phrased by intent rather than by token simply rank worse. Requires the ONNX build - the mock build is rejected on purpose, because it would rank at random while looking healthy.\",\n\t\tacquisition: \"on-demand\",\n\t\tenv: [\"HOOCODE_EMBSEARCH_BINARY\"],\n\t\tdependentRows: [\"group:embsearch\"],\n\t\tsettingsKeys: [\"enableEmbsearchTools\", \"embsearchBinaryPath\", \"embsearchThresholdBytes\"],\n\t},\n\t{\n\t\ttool: \"webtools\",\n\t\tlabel: \"webtools (webfetch/websearch)\",\n\t\tsummary: \"The network layer. Without it hoocode has no way to reach the internet.\",\n\t\tenables: [\"the webfetch tool\", \"the websearch tool\"],\n\t\tfallback:\n\t\t\t\"Both tools return an error when called. The web tool group is off by default, so a missing binary is invisible until you turn the group on.\",\n\t\tacquisition: \"on-demand\",\n\t\tenv: [\"HOOCODE_WEBTOOLS_BINARY\", \"HOOCODE_WEBTOOLS_TIMEOUT\"],\n\t\tdependentRows: [\"group:web\", \"webtools-timeout-secs\"],\n\t\tsettingsKeys: [\"enableWebTools\", \"webtools.timeoutSecs\"],\n\t},\n\t{\n\t\ttool: \"voicetools\",\n\t\tlabel: \"voicetools (voice input)\",\n\t\tsummary: \"Microphone capture and transcription for the TUI.\",\n\t\tenables: [\"push-to-talk voice input in the editor\"],\n\t\tfallback: \"Voice capture reports an error and never starts. Typing is unaffected.\",\n\t\tacquisition: \"on-demand\",\n\t\tenv: [\"VOICETOOLS_BIN\", \"HOOCODE_VOICETOOLS_BINARY\", \"VOICETOOLS_SILENCE_MS\"],\n\t\tdependentRows: [\"voice-silence-ms\"],\n\t\tsettingsKeys: [\"voice.silenceMs\"],\n\t},\n];\n\nexport interface ExternalToolStatus extends ExternalToolDoc, ManagedToolStatus {\n\tinstalled: boolean;\n\t/**\n\t * Whether hoocode would fetch it if the feature were used right now. False in\n\t * offline mode, and on Android where the published Linux builds do not run.\n\t */\n\tdownloadable: boolean;\n}\n\n/** Short, fixed-width-ish status word for the pane's value column. */\nexport function statusLabel(status: ExternalToolStatus): string {\n\tif (!status.installed) return status.downloadable ? \"not installed\" : \"unavailable\";\n\tswitch (status.source) {\n\t\tcase \"override\":\n\t\t\treturn \"env override\";\n\t\tcase \"managed\":\n\t\t\treturn \"installed\";\n\t\tcase \"path\":\n\t\t\treturn \"system\";\n\t\tdefault:\n\t\t\treturn \"installed\";\n\t}\n}\n\n/**\n * Resolve every external tool. Never downloads: this is what the pane shows\n * *before* the user opts into anything.\n */\nexport function describeExternalTools(): ExternalToolStatus[] {\n\tconst offline = isOfflineMode();\n\tconst android = process.platform === \"android\";\n\treturn EXTERNAL_TOOLS.map((doc) => {\n\t\tconst status = getToolStatus(doc.tool);\n\t\treturn {\n\t\t\t...doc,\n\t\t\t...status,\n\t\t\tinstalled: status.path !== null,\n\t\t\tdownloadable: !offline && !android && doc.acquisition !== \"manual\",\n\t\t};\n\t});\n}\n\n/** Index of pane-row id -> the tool it needs, for annotating gated rows. */\nexport function buildRowGates(statuses: readonly ExternalToolStatus[]): Map<string, ExternalToolStatus> {\n\tconst gates = new Map<string, ExternalToolStatus>();\n\tfor (const status of statuses) {\n\t\tfor (const rowId of status.dependentRows) gates.set(rowId, status);\n\t}\n\treturn gates;\n}\n"]}
|
|
@@ -0,0 +1,120 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The external (Rust) binaries hoocode can use, and what each one is worth.
|
|
3
|
+
*
|
|
4
|
+
* hoocode is self-sufficient without any of them: `grep`/`find`/`search` fall
|
|
5
|
+
* back to pure-JS implementations, and the features that have no fallback
|
|
6
|
+
* (web, voice) are off or inert rather than broken. These binaries are an
|
|
7
|
+
* *expansion* layer, which is exactly why they were invisible - nothing failed
|
|
8
|
+
* loudly enough to tell anyone they existed.
|
|
9
|
+
*
|
|
10
|
+
* This module is the single description of that layer. The `/settings` pane
|
|
11
|
+
* renders it; the same table says which settings rows are gated on a binary, so
|
|
12
|
+
* a row that cannot do anything today can say so instead of lying.
|
|
13
|
+
*/
|
|
14
|
+
import { getToolStatus, isOfflineMode } from "../utils/tools-manager.js";
|
|
15
|
+
/**
|
|
16
|
+
* Ordered so the two that silently make everything faster come first, then the
|
|
17
|
+
* three that add capability hoocode does not otherwise have.
|
|
18
|
+
*/
|
|
19
|
+
export const EXTERNAL_TOOLS = [
|
|
20
|
+
{
|
|
21
|
+
tool: "rg",
|
|
22
|
+
label: "ripgrep (rg)",
|
|
23
|
+
summary: "Fast path for content search.",
|
|
24
|
+
enables: ["grep runs rg instead of the JS scanner", "the lexical half of search runs rg"],
|
|
25
|
+
fallback: "A pure-JS scanner produces the same match shape, so results are identical - it is materially slower on large trees and respects fewer ignore-file edge cases.",
|
|
26
|
+
acquisition: "startup",
|
|
27
|
+
env: ["HOOCODE_RG_BINARY", "HOOCODE_NATIVE_SEARCH=1 forces the JS path even when rg is present"],
|
|
28
|
+
dependentRows: [],
|
|
29
|
+
settingsKeys: [],
|
|
30
|
+
},
|
|
31
|
+
{
|
|
32
|
+
tool: "fd",
|
|
33
|
+
label: "fd",
|
|
34
|
+
summary: "Fast path for filename search.",
|
|
35
|
+
enables: ["find runs fd instead of the JS directory walker"],
|
|
36
|
+
fallback: "A JS walker produces the same result shape - slower on large trees, and glob/ignore handling is the JS approximation rather than fd's.",
|
|
37
|
+
acquisition: "startup",
|
|
38
|
+
env: ["HOOCODE_FD_BINARY", "HOOCODE_NATIVE_SEARCH=1 forces the JS path even when fd is present"],
|
|
39
|
+
dependentRows: [],
|
|
40
|
+
settingsKeys: [],
|
|
41
|
+
},
|
|
42
|
+
{
|
|
43
|
+
tool: "embsearch",
|
|
44
|
+
label: "embsearch (semantic index)",
|
|
45
|
+
summary: "Local embedding index. The only source of semantic ranking in hoocode.",
|
|
46
|
+
enables: [
|
|
47
|
+
"search fuses semantic hits with its lexical hits",
|
|
48
|
+
"MCP/capability deferral ranks tools by meaning rather than keyword",
|
|
49
|
+
],
|
|
50
|
+
fallback: "search is lexical-only and capability lookup ranks lexically. Nothing errors; queries phrased by intent rather than by token simply rank worse. Requires the ONNX build - the mock build is rejected on purpose, because it would rank at random while looking healthy.",
|
|
51
|
+
acquisition: "on-demand",
|
|
52
|
+
env: ["HOOCODE_EMBSEARCH_BINARY"],
|
|
53
|
+
dependentRows: ["group:embsearch"],
|
|
54
|
+
settingsKeys: ["enableEmbsearchTools", "embsearchBinaryPath", "embsearchThresholdBytes"],
|
|
55
|
+
},
|
|
56
|
+
{
|
|
57
|
+
tool: "webtools",
|
|
58
|
+
label: "webtools (webfetch/websearch)",
|
|
59
|
+
summary: "The network layer. Without it hoocode has no way to reach the internet.",
|
|
60
|
+
enables: ["the webfetch tool", "the websearch tool"],
|
|
61
|
+
fallback: "Both tools return an error when called. The web tool group is off by default, so a missing binary is invisible until you turn the group on.",
|
|
62
|
+
acquisition: "on-demand",
|
|
63
|
+
env: ["HOOCODE_WEBTOOLS_BINARY", "HOOCODE_WEBTOOLS_TIMEOUT"],
|
|
64
|
+
dependentRows: ["group:web", "webtools-timeout-secs"],
|
|
65
|
+
settingsKeys: ["enableWebTools", "webtools.timeoutSecs"],
|
|
66
|
+
},
|
|
67
|
+
{
|
|
68
|
+
tool: "voicetools",
|
|
69
|
+
label: "voicetools (voice input)",
|
|
70
|
+
summary: "Microphone capture and transcription for the TUI.",
|
|
71
|
+
enables: ["push-to-talk voice input in the editor"],
|
|
72
|
+
fallback: "Voice capture reports an error and never starts. Typing is unaffected.",
|
|
73
|
+
acquisition: "on-demand",
|
|
74
|
+
env: ["VOICETOOLS_BIN", "HOOCODE_VOICETOOLS_BINARY", "VOICETOOLS_SILENCE_MS"],
|
|
75
|
+
dependentRows: ["voice-silence-ms"],
|
|
76
|
+
settingsKeys: ["voice.silenceMs"],
|
|
77
|
+
},
|
|
78
|
+
];
|
|
79
|
+
/** Short, fixed-width-ish status word for the pane's value column. */
|
|
80
|
+
export function statusLabel(status) {
|
|
81
|
+
if (!status.installed)
|
|
82
|
+
return status.downloadable ? "not installed" : "unavailable";
|
|
83
|
+
switch (status.source) {
|
|
84
|
+
case "override":
|
|
85
|
+
return "env override";
|
|
86
|
+
case "managed":
|
|
87
|
+
return "installed";
|
|
88
|
+
case "path":
|
|
89
|
+
return "system";
|
|
90
|
+
default:
|
|
91
|
+
return "installed";
|
|
92
|
+
}
|
|
93
|
+
}
|
|
94
|
+
/**
|
|
95
|
+
* Resolve every external tool. Never downloads: this is what the pane shows
|
|
96
|
+
* *before* the user opts into anything.
|
|
97
|
+
*/
|
|
98
|
+
export function describeExternalTools() {
|
|
99
|
+
const offline = isOfflineMode();
|
|
100
|
+
const android = process.platform === "android";
|
|
101
|
+
return EXTERNAL_TOOLS.map((doc) => {
|
|
102
|
+
const status = getToolStatus(doc.tool);
|
|
103
|
+
return {
|
|
104
|
+
...doc,
|
|
105
|
+
...status,
|
|
106
|
+
installed: status.path !== null,
|
|
107
|
+
downloadable: !offline && !android && doc.acquisition !== "manual",
|
|
108
|
+
};
|
|
109
|
+
});
|
|
110
|
+
}
|
|
111
|
+
/** Index of pane-row id -> the tool it needs, for annotating gated rows. */
|
|
112
|
+
export function buildRowGates(statuses) {
|
|
113
|
+
const gates = new Map();
|
|
114
|
+
for (const status of statuses) {
|
|
115
|
+
for (const rowId of status.dependentRows)
|
|
116
|
+
gates.set(rowId, status);
|
|
117
|
+
}
|
|
118
|
+
return gates;
|
|
119
|
+
}
|
|
120
|
+
//# sourceMappingURL=external-tools.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"external-tools.js","sourceRoot":"","sources":["../../src/core/external-tools.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;GAYG;AAEH,OAAO,EAAE,aAAa,EAAE,aAAa,EAA4C,MAAM,2BAA2B,CAAC;AAkCnH;;;GAGG;AACH,MAAM,CAAC,MAAM,cAAc,GAA+B;IACzD;QACC,IAAI,EAAE,IAAI;QACV,KAAK,EAAE,cAAc;QACrB,OAAO,EAAE,+BAA+B;QACxC,OAAO,EAAE,CAAC,wCAAwC,EAAE,oCAAoC,CAAC;QACzF,QAAQ,EACP,+JAA+J;QAChK,WAAW,EAAE,SAAS;QACtB,GAAG,EAAE,CAAC,mBAAmB,EAAE,oEAAoE,CAAC;QAChG,aAAa,EAAE,EAAE;QACjB,YAAY,EAAE,EAAE;KAChB;IACD;QACC,IAAI,EAAE,IAAI;QACV,KAAK,EAAE,IAAI;QACX,OAAO,EAAE,gCAAgC;QACzC,OAAO,EAAE,CAAC,iDAAiD,CAAC;QAC5D,QAAQ,EACP,wIAAwI;QACzI,WAAW,EAAE,SAAS;QACtB,GAAG,EAAE,CAAC,mBAAmB,EAAE,oEAAoE,CAAC;QAChG,aAAa,EAAE,EAAE;QACjB,YAAY,EAAE,EAAE;KAChB;IACD;QACC,IAAI,EAAE,WAAW;QACjB,KAAK,EAAE,4BAA4B;QACnC,OAAO,EAAE,wEAAwE;QACjF,OAAO,EAAE;YACR,kDAAkD;YAClD,oEAAoE;SACpE;QACD,QAAQ,EACP,yQAAyQ;QAC1Q,WAAW,EAAE,WAAW;QACxB,GAAG,EAAE,CAAC,0BAA0B,CAAC;QACjC,aAAa,EAAE,CAAC,iBAAiB,CAAC;QAClC,YAAY,EAAE,CAAC,sBAAsB,EAAE,qBAAqB,EAAE,yBAAyB,CAAC;KACxF;IACD;QACC,IAAI,EAAE,UAAU;QAChB,KAAK,EAAE,+BAA+B;QACtC,OAAO,EAAE,yEAAyE;QAClF,OAAO,EAAE,CAAC,mBAAmB,EAAE,oBAAoB,CAAC;QACpD,QAAQ,EACP,6IAA6I;QAC9I,WAAW,EAAE,WAAW;QACxB,GAAG,EAAE,CAAC,yBAAyB,EAAE,0BAA0B,CAAC;QAC5D,aAAa,EAAE,CAAC,WAAW,EAAE,uBAAuB,CAAC;QACrD,YAAY,EAAE,CAAC,gBAAgB,EAAE,sBAAsB,CAAC;KACxD;IACD;QACC,IAAI,EAAE,YAAY;QAClB,KAAK,EAAE,0BAA0B;QACjC,OAAO,EAAE,mDAAmD;QAC5D,OAAO,EAAE,CAAC,wCAAwC,CAAC;QACnD,QAAQ,EAAE,wEAAwE;QAClF,WAAW,EAAE,WAAW;QACxB,GAAG,EAAE,CAAC,gBAAgB,EAAE,2BAA2B,EAAE,uBAAuB,CAAC;QAC7E,aAAa,EAAE,CAAC,kBAAkB,CAAC;QACnC,YAAY,EAAE,CAAC,iBAAiB,CAAC;KACjC;CACD,CAAC;AAWF,sEAAsE;AACtE,MAAM,UAAU,WAAW,CAAC,MAA0B,EAAU;IAC/D,IAAI,CAAC,MAAM,CAAC,SAAS;QAAE,OAAO,MAAM,CAAC,YAAY,CAAC,CAAC,CAAC,eAAe,CAAC,CAAC,CAAC,aAAa,CAAC;IACpF,QAAQ,MAAM,CAAC,MAAM,EAAE,CAAC;QACvB,KAAK,UAAU;YACd,OAAO,cAAc,CAAC;QACvB,KAAK,SAAS;YACb,OAAO,WAAW,CAAC;QACpB,KAAK,MAAM;YACV,OAAO,QAAQ,CAAC;QACjB;YACC,OAAO,WAAW,CAAC;IACrB,CAAC;AAAA,CACD;AAED;;;GAGG;AACH,MAAM,UAAU,qBAAqB,GAAyB;IAC7D,MAAM,OAAO,GAAG,aAAa,EAAE,CAAC;IAChC,MAAM,OAAO,GAAG,OAAO,CAAC,QAAQ,KAAK,SAAS,CAAC;IAC/C,OAAO,cAAc,CAAC,GAAG,CAAC,CAAC,GAAG,EAAE,EAAE,CAAC;QAClC,MAAM,MAAM,GAAG,aAAa,CAAC,GAAG,CAAC,IAAI,CAAC,CAAC;QACvC,OAAO;YACN,GAAG,GAAG;YACN,GAAG,MAAM;YACT,SAAS,EAAE,MAAM,CAAC,IAAI,KAAK,IAAI;YAC/B,YAAY,EAAE,CAAC,OAAO,IAAI,CAAC,OAAO,IAAI,GAAG,CAAC,WAAW,KAAK,QAAQ;SAClE,CAAC;IAAA,CACF,CAAC,CAAC;AAAA,CACH;AAED,4EAA4E;AAC5E,MAAM,UAAU,aAAa,CAAC,QAAuC,EAAmC;IACvG,MAAM,KAAK,GAAG,IAAI,GAAG,EAA8B,CAAC;IACpD,KAAK,MAAM,MAAM,IAAI,QAAQ,EAAE,CAAC;QAC/B,KAAK,MAAM,KAAK,IAAI,MAAM,CAAC,aAAa;YAAE,KAAK,CAAC,GAAG,CAAC,KAAK,EAAE,MAAM,CAAC,CAAC;IACpE,CAAC;IACD,OAAO,KAAK,CAAC;AAAA,CACb","sourcesContent":["/**\n * The external (Rust) binaries hoocode can use, and what each one is worth.\n *\n * hoocode is self-sufficient without any of them: `grep`/`find`/`search` fall\n * back to pure-JS implementations, and the features that have no fallback\n * (web, voice) are off or inert rather than broken. These binaries are an\n * *expansion* layer, which is exactly why they were invisible - nothing failed\n * loudly enough to tell anyone they existed.\n *\n * This module is the single description of that layer. The `/settings` pane\n * renders it; the same table says which settings rows are gated on a binary, so\n * a row that cannot do anything today can say so instead of lying.\n */\n\nimport { getToolStatus, isOfflineMode, type ManagedTool, type ManagedToolStatus } from \"../utils/tools-manager.js\";\n\n/** How hoocode gets a binary it does not already have. */\nexport type Acquisition =\n\t/** Fetched in the background at startup. */\n\t| \"startup\"\n\t/** Fetched the first time the feature is actually used. */\n\t| \"on-demand\"\n\t/** Never fetched implicitly; install it yourself or point the env var at it. */\n\t| \"manual\";\n\nexport interface ExternalToolDoc {\n\ttool: ManagedTool;\n\t/** Row label in the pane. */\n\tlabel: string;\n\t/** What the binary does for hoocode, in one line. */\n\tsummary: string;\n\t/** What it turns on, feature by feature. */\n\tenables: string[];\n\t/** What hoocode does instead when it is missing. Never \"nothing works\". */\n\tfallback: string;\n\tacquisition: Acquisition;\n\t/** Env vars that change how this binary is resolved or driven. */\n\tenv: string[];\n\t/**\n\t * `/settings` row ids whose effect depends on this binary. A row listed here\n\t * is annotated (not hidden) while the binary is missing: hiding it would\n\t * recreate the discoverability hole this whole surface exists to close.\n\t */\n\tdependentRows: string[];\n\t/** settings.json keys this binary gates, for docs and search. */\n\tsettingsKeys: string[];\n}\n\n/**\n * Ordered so the two that silently make everything faster come first, then the\n * three that add capability hoocode does not otherwise have.\n */\nexport const EXTERNAL_TOOLS: readonly ExternalToolDoc[] = [\n\t{\n\t\ttool: \"rg\",\n\t\tlabel: \"ripgrep (rg)\",\n\t\tsummary: \"Fast path for content search.\",\n\t\tenables: [\"grep runs rg instead of the JS scanner\", \"the lexical half of search runs rg\"],\n\t\tfallback:\n\t\t\t\"A pure-JS scanner produces the same match shape, so results are identical - it is materially slower on large trees and respects fewer ignore-file edge cases.\",\n\t\tacquisition: \"startup\",\n\t\tenv: [\"HOOCODE_RG_BINARY\", \"HOOCODE_NATIVE_SEARCH=1 forces the JS path even when rg is present\"],\n\t\tdependentRows: [],\n\t\tsettingsKeys: [],\n\t},\n\t{\n\t\ttool: \"fd\",\n\t\tlabel: \"fd\",\n\t\tsummary: \"Fast path for filename search.\",\n\t\tenables: [\"find runs fd instead of the JS directory walker\"],\n\t\tfallback:\n\t\t\t\"A JS walker produces the same result shape - slower on large trees, and glob/ignore handling is the JS approximation rather than fd's.\",\n\t\tacquisition: \"startup\",\n\t\tenv: [\"HOOCODE_FD_BINARY\", \"HOOCODE_NATIVE_SEARCH=1 forces the JS path even when fd is present\"],\n\t\tdependentRows: [],\n\t\tsettingsKeys: [],\n\t},\n\t{\n\t\ttool: \"embsearch\",\n\t\tlabel: \"embsearch (semantic index)\",\n\t\tsummary: \"Local embedding index. The only source of semantic ranking in hoocode.\",\n\t\tenables: [\n\t\t\t\"search fuses semantic hits with its lexical hits\",\n\t\t\t\"MCP/capability deferral ranks tools by meaning rather than keyword\",\n\t\t],\n\t\tfallback:\n\t\t\t\"search is lexical-only and capability lookup ranks lexically. Nothing errors; queries phrased by intent rather than by token simply rank worse. Requires the ONNX build - the mock build is rejected on purpose, because it would rank at random while looking healthy.\",\n\t\tacquisition: \"on-demand\",\n\t\tenv: [\"HOOCODE_EMBSEARCH_BINARY\"],\n\t\tdependentRows: [\"group:embsearch\"],\n\t\tsettingsKeys: [\"enableEmbsearchTools\", \"embsearchBinaryPath\", \"embsearchThresholdBytes\"],\n\t},\n\t{\n\t\ttool: \"webtools\",\n\t\tlabel: \"webtools (webfetch/websearch)\",\n\t\tsummary: \"The network layer. Without it hoocode has no way to reach the internet.\",\n\t\tenables: [\"the webfetch tool\", \"the websearch tool\"],\n\t\tfallback:\n\t\t\t\"Both tools return an error when called. The web tool group is off by default, so a missing binary is invisible until you turn the group on.\",\n\t\tacquisition: \"on-demand\",\n\t\tenv: [\"HOOCODE_WEBTOOLS_BINARY\", \"HOOCODE_WEBTOOLS_TIMEOUT\"],\n\t\tdependentRows: [\"group:web\", \"webtools-timeout-secs\"],\n\t\tsettingsKeys: [\"enableWebTools\", \"webtools.timeoutSecs\"],\n\t},\n\t{\n\t\ttool: \"voicetools\",\n\t\tlabel: \"voicetools (voice input)\",\n\t\tsummary: \"Microphone capture and transcription for the TUI.\",\n\t\tenables: [\"push-to-talk voice input in the editor\"],\n\t\tfallback: \"Voice capture reports an error and never starts. Typing is unaffected.\",\n\t\tacquisition: \"on-demand\",\n\t\tenv: [\"VOICETOOLS_BIN\", \"HOOCODE_VOICETOOLS_BINARY\", \"VOICETOOLS_SILENCE_MS\"],\n\t\tdependentRows: [\"voice-silence-ms\"],\n\t\tsettingsKeys: [\"voice.silenceMs\"],\n\t},\n];\n\nexport interface ExternalToolStatus extends ExternalToolDoc, ManagedToolStatus {\n\tinstalled: boolean;\n\t/**\n\t * Whether hoocode would fetch it if the feature were used right now. False in\n\t * offline mode, and on Android where the published Linux builds do not run.\n\t */\n\tdownloadable: boolean;\n}\n\n/** Short, fixed-width-ish status word for the pane's value column. */\nexport function statusLabel(status: ExternalToolStatus): string {\n\tif (!status.installed) return status.downloadable ? \"not installed\" : \"unavailable\";\n\tswitch (status.source) {\n\t\tcase \"override\":\n\t\t\treturn \"env override\";\n\t\tcase \"managed\":\n\t\t\treturn \"installed\";\n\t\tcase \"path\":\n\t\t\treturn \"system\";\n\t\tdefault:\n\t\t\treturn \"installed\";\n\t}\n}\n\n/**\n * Resolve every external tool. Never downloads: this is what the pane shows\n * *before* the user opts into anything.\n */\nexport function describeExternalTools(): ExternalToolStatus[] {\n\tconst offline = isOfflineMode();\n\tconst android = process.platform === \"android\";\n\treturn EXTERNAL_TOOLS.map((doc) => {\n\t\tconst status = getToolStatus(doc.tool);\n\t\treturn {\n\t\t\t...doc,\n\t\t\t...status,\n\t\t\tinstalled: status.path !== null,\n\t\t\tdownloadable: !offline && !android && doc.acquisition !== \"manual\",\n\t\t};\n\t});\n}\n\n/** Index of pane-row id -> the tool it needs, for annotating gated rows. */\nexport function buildRowGates(statuses: readonly ExternalToolStatus[]): Map<string, ExternalToolStatus> {\n\tconst gates = new Map<string, ExternalToolStatus>();\n\tfor (const status of statuses) {\n\t\tfor (const rowId of status.dependentRows) gates.set(rowId, status);\n\t}\n\treturn gates;\n}\n"]}
|
|
@@ -5,12 +5,24 @@
|
|
|
5
5
|
* modes (ask / plan / build / debug). They are resolved only when a project or
|
|
6
6
|
* user has not supplied a `modes/{name}/system.md` override.
|
|
7
7
|
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
8
|
+
* The prompts themselves live in `templates/modes/<mode>/system.md` and reach
|
|
9
|
+
* this module through the build-time embed. They used to be a second, hand-
|
|
10
|
+
* written copy right here, and the two had already drifted: `/init` scaffolds
|
|
11
|
+
* the template copy into a project, so a user who ran it got one set of mode
|
|
12
|
+
* rules and a user who did not got another, differently worded set. Prose has
|
|
13
|
+
* one home now; this module is the typed accessor for it.
|
|
14
|
+
*
|
|
15
|
+
* Kept as a standalone module (rather than living inside the internal `hoo-core`
|
|
16
|
+
* extension's `modes` module) so downstream apps embedding hoocode can import,
|
|
10
17
|
* inspect, and extend the shipped prompts without copy-pasting them.
|
|
11
18
|
*/
|
|
12
19
|
/** Mode used when no `active_mode` is configured. */
|
|
13
20
|
export declare const DEFAULT_MODE = "build";
|
|
14
|
-
/**
|
|
21
|
+
/**
|
|
22
|
+
* Built-in fallback system prompts keyed by mode name.
|
|
23
|
+
*
|
|
24
|
+
* The plan-mode prompt carries a `{{PLAN_PATH}}` token that the mode extension
|
|
25
|
+
* substitutes per session; a consumer rendering these itself must do the same.
|
|
26
|
+
*/
|
|
15
27
|
export declare const DEFAULT_MODE_PROMPTS: Record<string, string>;
|
|
16
28
|
//# sourceMappingURL=mode-prompts.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"mode-prompts.d.ts","sourceRoot":"","sources":["../../src/core/mode-prompts.ts"],"names":[],"mappings":"AAAA
|
|
1
|
+
{"version":3,"file":"mode-prompts.d.ts","sourceRoot":"","sources":["../../src/core/mode-prompts.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;GAiBG;AAIH,qDAAqD;AACrD,eAAO,MAAM,YAAY,UAAU,CAAC;AAEpC;;;;;GAKG;AACH,eAAO,MAAM,oBAAoB,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAkB,CAAC","sourcesContent":["/**\n * Canonical built-in mode prompts and the default active mode.\n *\n * These are the fallback system prompts hoocode ships for its four built-in\n * modes (ask / plan / build / debug). They are resolved only when a project or\n * user has not supplied a `modes/{name}/system.md` override.\n *\n * The prompts themselves live in `templates/modes/<mode>/system.md` and reach\n * this module through the build-time embed. They used to be a second, hand-\n * written copy right here, and the two had already drifted: `/init` scaffolds\n * the template copy into a project, so a user who ran it got one set of mode\n * rules and a user who did not got another, differently worded set. Prose has\n * one home now; this module is the typed accessor for it.\n *\n * Kept as a standalone module (rather than living inside the internal `hoo-core`\n * extension's `modes` module) so downstream apps embedding hoocode can import,\n * inspect, and extend the shipped prompts without copy-pasting them.\n */\n\nimport { EMBEDDED_MODES } from \"../init-templates.generated.js\";\n\n/** Mode used when no `active_mode` is configured. */\nexport const DEFAULT_MODE = \"build\";\n\n/**\n * Built-in fallback system prompts keyed by mode name.\n *\n * The plan-mode prompt carries a `{{PLAN_PATH}}` token that the mode extension\n * substitutes per session; a consumer rendering these itself must do the same.\n */\nexport const DEFAULT_MODE_PROMPTS: Record<string, string> = EMBEDDED_MODES;\n"]}
|
|
@@ -5,37 +5,25 @@
|
|
|
5
5
|
* modes (ask / plan / build / debug). They are resolved only when a project or
|
|
6
6
|
* user has not supplied a `modes/{name}/system.md` override.
|
|
7
7
|
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
8
|
+
* The prompts themselves live in `templates/modes/<mode>/system.md` and reach
|
|
9
|
+
* this module through the build-time embed. They used to be a second, hand-
|
|
10
|
+
* written copy right here, and the two had already drifted: `/init` scaffolds
|
|
11
|
+
* the template copy into a project, so a user who ran it got one set of mode
|
|
12
|
+
* rules and a user who did not got another, differently worded set. Prose has
|
|
13
|
+
* one home now; this module is the typed accessor for it.
|
|
14
|
+
*
|
|
15
|
+
* Kept as a standalone module (rather than living inside the internal `hoo-core`
|
|
16
|
+
* extension's `modes` module) so downstream apps embedding hoocode can import,
|
|
10
17
|
* inspect, and extend the shipped prompts without copy-pasting them.
|
|
11
18
|
*/
|
|
19
|
+
import { EMBEDDED_MODES } from "../init-templates.generated.js";
|
|
12
20
|
/** Mode used when no `active_mode` is configured. */
|
|
13
21
|
export const DEFAULT_MODE = "build";
|
|
14
|
-
/**
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
plan: `You are in PLAN mode — exploration and planning.
|
|
22
|
-
Explore the codebase thoroughly. Understand the current structure.
|
|
23
|
-
Draft a complete plan with sections: Goal, Files to modify, New files, Tests, Verification.
|
|
24
|
-
Write the plan to {{PLAN_PATH}}.
|
|
25
|
-
When the plan is complete, tell the user their options: /grill to stress-test it
|
|
26
|
-
first, then /approve to execute it step by step or /goal to work toward it
|
|
27
|
-
autonomously. Recommend /grill when the plan carries real risk.`,
|
|
28
|
-
build: `You are in BUILD mode — careful implementation.
|
|
29
|
-
Read files before editing them. Show diffs before non-trivial changes.
|
|
30
|
-
Ask for confirmation before destructive operations (delete, reformat).
|
|
31
|
-
Run tests after every logical unit of work.
|
|
32
|
-
Prefer the smallest change that achieves the goal.
|
|
33
|
-
Follow existing code patterns and conventions.`,
|
|
34
|
-
debug: `You are in DEBUG mode — root cause analysis.
|
|
35
|
-
Gather evidence: read files, check logs, reproduce the issue.
|
|
36
|
-
Trace the call path from entry to failure point.
|
|
37
|
-
State the root cause in one sentence.
|
|
38
|
-
Describe the fix precisely but do NOT apply it.
|
|
39
|
-
To fix, switch to /mode build.`,
|
|
40
|
-
};
|
|
22
|
+
/**
|
|
23
|
+
* Built-in fallback system prompts keyed by mode name.
|
|
24
|
+
*
|
|
25
|
+
* The plan-mode prompt carries a `{{PLAN_PATH}}` token that the mode extension
|
|
26
|
+
* substitutes per session; a consumer rendering these itself must do the same.
|
|
27
|
+
*/
|
|
28
|
+
export const DEFAULT_MODE_PROMPTS = EMBEDDED_MODES;
|
|
41
29
|
//# sourceMappingURL=mode-prompts.js.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"mode-prompts.js","sourceRoot":"","sources":["../../src/core/mode-prompts.ts"],"names":[],"mappings":"AAAA
|
|
1
|
+
{"version":3,"file":"mode-prompts.js","sourceRoot":"","sources":["../../src/core/mode-prompts.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;GAiBG;AAEH,OAAO,EAAE,cAAc,EAAE,MAAM,gCAAgC,CAAC;AAEhE,qDAAqD;AACrD,MAAM,CAAC,MAAM,YAAY,GAAG,OAAO,CAAC;AAEpC;;;;;GAKG;AACH,MAAM,CAAC,MAAM,oBAAoB,GAA2B,cAAc,CAAC","sourcesContent":["/**\n * Canonical built-in mode prompts and the default active mode.\n *\n * These are the fallback system prompts hoocode ships for its four built-in\n * modes (ask / plan / build / debug). They are resolved only when a project or\n * user has not supplied a `modes/{name}/system.md` override.\n *\n * The prompts themselves live in `templates/modes/<mode>/system.md` and reach\n * this module through the build-time embed. They used to be a second, hand-\n * written copy right here, and the two had already drifted: `/init` scaffolds\n * the template copy into a project, so a user who ran it got one set of mode\n * rules and a user who did not got another, differently worded set. Prose has\n * one home now; this module is the typed accessor for it.\n *\n * Kept as a standalone module (rather than living inside the internal `hoo-core`\n * extension's `modes` module) so downstream apps embedding hoocode can import,\n * inspect, and extend the shipped prompts without copy-pasting them.\n */\n\nimport { EMBEDDED_MODES } from \"../init-templates.generated.js\";\n\n/** Mode used when no `active_mode` is configured. */\nexport const DEFAULT_MODE = \"build\";\n\n/**\n * Built-in fallback system prompts keyed by mode name.\n *\n * The plan-mode prompt carries a `{{PLAN_PATH}}` token that the mode extension\n * substitutes per session; a consumer rendering these itself must do the same.\n */\nexport const DEFAULT_MODE_PROMPTS: Record<string, string> = EMBEDDED_MODES;\n"]}
|