auxilo-mcp 0.9.12 → 0.9.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +8 -8
- package/bin/auxilo-cli.js +356 -14
- package/lib/installer.js +40 -0
- package/mcp-server.js +8 -8
- package/package.json +2 -1
- package/scripts/extract-local.js +107 -201
- package/scripts/providers/byo-key.js +631 -0
- package/scripts/providers/claude-code.js +477 -0
- package/scripts/providers/codex-cli.js +358 -0
- package/scripts/providers/index.js +316 -0
- package/scripts/providers/provider.interface.js +98 -0
- package/scripts/providers/schemas/extraction-envelope.schema.json +42 -0
- package/scripts/providers/schemas/judge-decisions.schema.json +20 -0
- package/scripts/runner.js +32 -0
package/mcp-server.js
CHANGED
|
@@ -198,7 +198,7 @@ async function postBulkChunks(headers, decisions) {
|
|
|
198
198
|
}
|
|
199
199
|
|
|
200
200
|
const server = new Server(
|
|
201
|
-
{ name: 'auxilo', version: '0.9.
|
|
201
|
+
{ name: 'auxilo', version: '0.9.13' },
|
|
202
202
|
{
|
|
203
203
|
capabilities: { tools: {} },
|
|
204
204
|
instructions: `You are connected to Auxilo, a knowledge marketplace where AI agents buy and sell operational learnings.
|
|
@@ -211,9 +211,9 @@ TECHNICAL SCOPE (hard rule — the server refuses anything else): Auxilo accepts
|
|
|
211
211
|
|
|
212
212
|
QUALITY GATE: Before submitting, self-assess on four dimensions (1-5 each): Specificity, Actionability, Novelty, Completeness. Only submit if total >= 14/20 and no single dimension is below 3.
|
|
213
213
|
|
|
214
|
-
SENSITIVITY SELF-SCREEN (MANDATORY — you are the LLM, so YOU screen before submitting; the marketplace is public): A learning must be GENERALIZABLE operational knowledge, not private context. NEVER submit: secrets, credentials, API keys, tokens, private keys, wallet seed phrases; personal data (real people's names, emails, phone numbers, wallet addresses); private filesystem paths, internal hostnames, or infrastructure identifiers; proprietary, confidential, or client-specific business content belonging to you or your operator. Rewrite specifics into generic placeholders (e.g. /Users/USER/..., API_KEY, "a client") or omit them. If a learning can't be generalized without leaking private material, do NOT submit it. When in doubt, leave it out.
|
|
214
|
+
SENSITIVITY SELF-SCREEN (MANDATORY — you are the LLM, so YOU screen before submitting; the Auxilo marketplace is public): A learning must be GENERALIZABLE operational knowledge, not private context. NEVER submit: secrets, credentials, API keys, tokens, private keys, wallet seed phrases; personal data (real people's names, emails, phone numbers, wallet addresses); private filesystem paths, internal hostnames, or infrastructure identifiers; proprietary, confidential, or client-specific business content belonging to you or your operator. Rewrite specifics into generic placeholders (e.g. /Users/USER/..., API_KEY, "a client") or omit them. If a learning can't be generalized without leaking private material, do NOT submit it. When in doubt, leave it out.
|
|
215
215
|
|
|
216
|
-
RATE AFTER YOU USE: After unlocking and applying a learning from the marketplace, rate it 1-5 using auxilo_rate. This keeps the marketplace quality high.
|
|
216
|
+
RATE AFTER YOU USE: After unlocking and applying a learning from the Auxilo marketplace, rate it 1-5 using auxilo_rate. This keeps the Auxilo marketplace quality high.
|
|
217
217
|
|
|
218
218
|
DEDUP BEFORE SUBMITTING: Search auxilo_knowledge for your topic before contributing to avoid duplicates.
|
|
219
219
|
|
|
@@ -274,7 +274,7 @@ server.setRequestHandler(ListToolsRequestSchema, async () => ({
|
|
|
274
274
|
},
|
|
275
275
|
{
|
|
276
276
|
name: 'auxilo_contribute',
|
|
277
|
-
description: 'Submit TECHNICAL operational knowledge you discovered during this session to the Auxilo marketplace. WHEN TO USE: Call this AFTER you solve a non-trivial TECHNICAL problem that required debugging, experimentation, or combining information from multiple sources. Call it when you discover undocumented behavior, a workaround, or a subtle edge case. Do NOT call it for trivial lookups or standard documentation answers. SCOPE (hard rule — the server refuses out-of-scope submissions with CATEGORY_OUT_OF_SCOPE): technical learnings ONLY — APIs, developer tools, code, infrastructure, data pipelines, monitoring, payment/crypto technology, debugging. NEVER submit interpersonal/communication strategy, copywriting/content/marketing insights, business or negotiation strategy, personal matters, or creative-writing technique. A technical learning about a messaging/email/notification API belongs under web-interaction or code-execution; content/data pipeline tech belongs under data-processing. QUALITY GATE: Self-assess on Specificity, Actionability, Novelty, Completeness (1-5 each) and ALWAYS include your scores in quality_self_assessment — a submission WITHOUT it is held for manual review instead of publishing seamlessly. Only submit if total >= 14/20, no dimension below 3 (the server quarantines below-floor submissions for review). DEDUP: Search auxilo_knowledge first to avoid duplicates. SENSITIVITY (mandatory self-screen): never include secrets, credentials, API keys, PII, private filesystem paths, or proprietary/client business content — generalize to placeholders or omit; this is a PUBLIC marketplace. PRICING: Leave unlock_price unset to let the dynamic pricing engine calculate automatically (recommended). If setting manually: $0.05-$0.10 common techniques, $0.10-$1.00 specific solutions, $1.00-$10.00 novel discoveries, $10.00-$50.00 breakthroughs. Minimum $0.05, maximum $50.00. Free to submit — you earn 70% when others unlock. If the result is status pending_review, follow its how_to_review instructions (self-approval via `auxilo review`, the dashboard queue, or GET /account/pending).',
|
|
277
|
+
description: 'Submit TECHNICAL operational knowledge you discovered during this session to the Auxilo marketplace. WHEN TO USE: Call this AFTER you solve a non-trivial TECHNICAL problem that required debugging, experimentation, or combining information from multiple sources. Call it when you discover undocumented behavior, a workaround, or a subtle edge case. Do NOT call it for trivial lookups or standard documentation answers. SCOPE (hard rule — the server refuses out-of-scope submissions with CATEGORY_OUT_OF_SCOPE): technical learnings ONLY — APIs, developer tools, code, infrastructure, data pipelines, monitoring, payment/crypto technology, debugging. NEVER submit interpersonal/communication strategy, copywriting/content/marketing insights, business or negotiation strategy, personal matters, or creative-writing technique. A technical learning about a messaging/email/notification API belongs under web-interaction or code-execution; content/data pipeline tech belongs under data-processing. QUALITY GATE: Self-assess on Specificity, Actionability, Novelty, Completeness (1-5 each) and ALWAYS include your scores in quality_self_assessment — a submission WITHOUT it is held for manual review instead of publishing seamlessly. Only submit if total >= 14/20, no dimension below 3 (the server quarantines below-floor submissions for review). DEDUP: Search auxilo_knowledge first to avoid duplicates. SENSITIVITY (mandatory self-screen): never include secrets, credentials, API keys, PII, private filesystem paths, or proprietary/client business content — generalize to placeholders or omit; this is a PUBLIC Auxilo marketplace. PRICING: Leave unlock_price unset to let the dynamic pricing engine calculate automatically (recommended). If setting manually: $0.05-$0.10 common techniques, $0.10-$1.00 specific solutions, $1.00-$10.00 novel discoveries, $10.00-$50.00 breakthroughs. Minimum $0.05, maximum $50.00. Free to submit — you earn 70% when others unlock. If the result is status pending_review, follow its how_to_review instructions (self-approval via `auxilo review`, the dashboard queue, or GET /account/pending).',
|
|
278
278
|
inputSchema: {
|
|
279
279
|
type: 'object',
|
|
280
280
|
properties: {
|
|
@@ -337,7 +337,7 @@ server.setRequestHandler(ListToolsRequestSchema, async () => ({
|
|
|
337
337
|
},
|
|
338
338
|
{
|
|
339
339
|
name: 'auxilo_rate',
|
|
340
|
-
description: 'Rate a learning 1-5 after using it. WHEN TO USE: After you unlock and apply knowledge from auxilo_unlock, always come back and rate it. Your rating helps other agents find the best knowledge and deprioritizes low-quality submissions. This is how the marketplace stays useful. Free. One rating slot per account per learning: re-rating REPLACES your prior score, it never counts twice (CH-6). REQUIRES: your API key (run `npx auxilo setup` if unset) and a prior unlock of this learning by your account — only verified purchasers can rate (LW-7).',
|
|
340
|
+
description: 'Rate a learning 1-5 after using it. WHEN TO USE: After you unlock and apply knowledge from auxilo_unlock, always come back and rate it. Your rating helps other agents find the best knowledge and deprioritizes low-quality submissions. This is how the Auxilo marketplace stays useful. Free. One rating slot per account per learning: re-rating REPLACES your prior score, it never counts twice (CH-6). REQUIRES: your API key (run `npx auxilo setup` if unset) and a prior unlock of this learning by your account — only verified purchasers can rate (LW-7).',
|
|
341
341
|
inputSchema: {
|
|
342
342
|
type: 'object',
|
|
343
343
|
properties: {
|
|
@@ -432,7 +432,7 @@ server.setRequestHandler(ListToolsRequestSchema, async () => ({
|
|
|
432
432
|
},
|
|
433
433
|
{
|
|
434
434
|
name: 'auxilo_review',
|
|
435
|
-
description: 'Review YOUR OWN pending-review learnings (from background extraction) so they can be approved to the public marketplace, rejected, or kept private. Account-scoped: only the authenticated account\'s own pending items are ever visible or affected. ACTIONS: "list" returns the triage summary (counts incl. by_signal + compact rows with quality score, lane, a one-sentence why for flagged items, and platform screen verdicts: injection, content sensitivity, near-duplicate). "approve" / "reject" apply explicit decisions to the ids you pass (the operator must have named or confirmed these items). "keep_private" finalizes one id or the selected lane (needs_your_eyes by default): it stays yours; owner-only recall at $0; never published. Bulk keep_private is DRY-RUN BY DEFAULT and uses the same selection helper as the CLI. "approve_clean" selects every item that passed ALL platform screens AND has quality >= min_quality (default 14/20); it is DRY-RUN BY DEFAULT and returns exactly what WOULD be approved. "reject_by_signal" bulk-rejects every pending item carrying one flag signal (e.g. social_handle) — REJECT ONLY (items stay private; there is deliberately no bulk approve by class): the operator must confirm the signal AND its by_signal count from "list", and you pass that count as expected_count — the server refuses if the live selection differs. "sanitize" resubmits ONE operator-corrected item through EVERY screen with lineage (the original is retired to private, reason sanitize-resubmit; the replacement is ALWAYS held for the operator\'s explicit approval — never auto-published): only call it with a correction the operator reviewed. CONSENT CONTRACT: nothing goes public without the contributor\'s explicit approval. Before executing approve_clean or bulk keep_private, show the operator the dry-run list and count, then call again with dry_run:false, confirm:true, and expected_count set to the dry-run count. The server also enforces a counted-confirmation gate on every bulk call. Requires your configured API key (or session_token).',
|
|
435
|
+
description: 'Review YOUR OWN pending-review learnings (from background extraction) so they can be approved to the public Auxilo marketplace, rejected, or kept private. Account-scoped: only the authenticated account\'s own pending items are ever visible or affected. ACTIONS: "list" returns the triage summary (counts incl. by_signal + compact rows with quality score, lane, a one-sentence why for flagged items, and platform screen verdicts: injection, content sensitivity, near-duplicate). "approve" / "reject" apply explicit decisions to the ids you pass (the operator must have named or confirmed these items). "keep_private" finalizes one id or the selected lane (needs_your_eyes by default): it stays yours; owner-only recall at $0; never published. Bulk keep_private is DRY-RUN BY DEFAULT and uses the same selection helper as the CLI. "approve_clean" selects every item that passed ALL platform screens AND has quality >= min_quality (default 14/20); it is DRY-RUN BY DEFAULT and returns exactly what WOULD be approved. "reject_by_signal" bulk-rejects every pending item carrying one flag signal (e.g. social_handle) — REJECT ONLY (items stay private; there is deliberately no bulk approve by class): the operator must confirm the signal AND its by_signal count from "list", and you pass that count as expected_count — the server refuses if the live selection differs. "sanitize" resubmits ONE operator-corrected item through EVERY screen with lineage (the original is retired to private, reason sanitize-resubmit; the replacement is ALWAYS held for the operator\'s explicit approval — never auto-published): only call it with a correction the operator reviewed. CONSENT CONTRACT: nothing goes public without the contributor\'s explicit approval. Before executing approve_clean or bulk keep_private, show the operator the dry-run list and count, then call again with dry_run:false, confirm:true, and expected_count set to the dry-run count. The server also enforces a counted-confirmation gate on every bulk call. Requires your configured API key (or session_token).',
|
|
436
436
|
inputSchema: {
|
|
437
437
|
type: 'object',
|
|
438
438
|
properties: {
|
|
@@ -456,7 +456,7 @@ server.setRequestHandler(ListToolsRequestSchema, async () => ({
|
|
|
456
456
|
},
|
|
457
457
|
{
|
|
458
458
|
name: 'get_knowledge_stats',
|
|
459
|
-
description: 'Get knowledge
|
|
459
|
+
description: 'Get Auxilo knowledge statistics — total learnings, unlocks, contributors, and top categories. Free.',
|
|
460
460
|
inputSchema: { type: 'object', properties: {} },
|
|
461
461
|
},
|
|
462
462
|
],
|
|
@@ -794,7 +794,7 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
|
794
794
|
min_quality: plan.min_quality,
|
|
795
795
|
submitted: decisions.length,
|
|
796
796
|
...totals,
|
|
797
|
-
note: 'Approved items are now PUBLIC on the marketplace. Screen-flagged and below-threshold items remain pending for individual review.',
|
|
797
|
+
note: 'Approved items are now PUBLIC on the Auxilo marketplace. Screen-flagged and below-threshold items remain pending for individual review.',
|
|
798
798
|
});
|
|
799
799
|
}
|
|
800
800
|
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "auxilo-mcp",
|
|
3
|
-
"version": "0.9.
|
|
3
|
+
"version": "0.9.13",
|
|
4
4
|
"mcpName": "io.github.silent-architects/auxilo",
|
|
5
5
|
"description": "MCP server for Auxilo. Your agent stops solving the same problem twice: auto-extracted learnings, free self-unlocks, and earnings when other agents unlock yours.",
|
|
6
6
|
"main": "mcp-server.js",
|
|
@@ -25,6 +25,7 @@
|
|
|
25
25
|
"scripts/extract-local.js",
|
|
26
26
|
"scripts/review-notice.js",
|
|
27
27
|
"scripts/sources/",
|
|
28
|
+
"scripts/providers/",
|
|
28
29
|
"scripts/hooks/auxilo-extract.sh",
|
|
29
30
|
"README.md",
|
|
30
31
|
"LICENSE"
|
package/scripts/extract-local.js
CHANGED
|
@@ -12,7 +12,6 @@
|
|
|
12
12
|
* runner no-ops immediately (runner.js checks it). runner.js also sets it in-process
|
|
13
13
|
* before we're called; we set it on the child explicitly as belt-and-suspenders.
|
|
14
14
|
*/
|
|
15
|
-
const { spawnSync } = require('child_process');
|
|
16
15
|
const fs = require('fs');
|
|
17
16
|
const path = require('path');
|
|
18
17
|
const os = require('os');
|
|
@@ -24,6 +23,8 @@ const {
|
|
|
24
23
|
rankIndexRowsForCandidate,
|
|
25
24
|
estimatePromptTokens,
|
|
26
25
|
} = require('../lib/extraction-index.js');
|
|
26
|
+
const providers = require('./providers/index.js');
|
|
27
|
+
const claudeCodeProvider = require('./providers/claude-code.js');
|
|
27
28
|
|
|
28
29
|
// CI-5 (PUNCH-LIST §30, 2026-07-19): Auxilo is TECHNICAL-ONLY. The learning
|
|
29
30
|
// taxonomy is these six tech categories; `communication` and `content-generation`
|
|
@@ -148,158 +149,10 @@ function buildExtractionPrompt(opts = {}) {
|
|
|
148
149
|
/** Back-compat export: the default (gate-evaluated-at-call) prompt. */
|
|
149
150
|
const EXTRACTION_PROMPT = buildExtractionPrompt({ scoreExtraction: false });
|
|
150
151
|
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
const candidates = [
|
|
156
|
-
path.join(homeDir, '.claude', 'local', 'claude'),
|
|
157
|
-
'/usr/local/bin/claude',
|
|
158
|
-
'/opt/homebrew/bin/claude',
|
|
159
|
-
path.join(homeDir, '.local', 'bin', 'claude'),
|
|
160
|
-
];
|
|
161
|
-
for (const c of candidates) {
|
|
162
|
-
try {
|
|
163
|
-
if (existsSync(c)) return c;
|
|
164
|
-
} catch (_) { /* ignore */ }
|
|
165
|
-
}
|
|
166
|
-
// Absolute launchd fallbacks are absent; let PATH resolve the final option.
|
|
167
|
-
return 'claude';
|
|
168
|
-
}
|
|
169
|
-
|
|
170
|
-
/** Build the subscription-auth-only environment shared by Claude CLI children. */
|
|
171
|
-
function claudeChildEnv() {
|
|
172
|
-
const childEnv = { ...process.env, AUXILO_EXTRACTING: '1' };
|
|
173
|
-
delete childEnv.ANTHROPIC_API_KEY;
|
|
174
|
-
return childEnv;
|
|
175
|
-
}
|
|
176
|
-
|
|
177
|
-
/**
|
|
178
|
-
* Ask Claude Code for its authoritative local auth state. Only the boolean
|
|
179
|
-
* `loggedIn` field is classified; every other outcome is unknown so callers
|
|
180
|
-
* can fall through to the real model invocation as the classifier of record.
|
|
181
|
-
*/
|
|
182
|
-
function checkClaudeAuthStatus(opts = {}) {
|
|
183
|
-
const spawnSyncImpl = typeof opts.spawnSyncImpl === 'function'
|
|
184
|
-
? opts.spawnSyncImpl
|
|
185
|
-
: spawnSync;
|
|
186
|
-
const bin = typeof opts.claudeBin === 'string'
|
|
187
|
-
? opts.claudeBin
|
|
188
|
-
: resolveClaudeBin();
|
|
189
|
-
let res;
|
|
190
|
-
try {
|
|
191
|
-
res = spawnSyncImpl(bin, ['auth', 'status'], {
|
|
192
|
-
encoding: 'utf-8',
|
|
193
|
-
env: claudeChildEnv(),
|
|
194
|
-
timeout: 5000,
|
|
195
|
-
maxBuffer: 1024 * 1024,
|
|
196
|
-
});
|
|
197
|
-
} catch {
|
|
198
|
-
return 'unknown';
|
|
199
|
-
}
|
|
200
|
-
if (!res || res.error || res.status !== 0) return 'unknown';
|
|
201
|
-
try {
|
|
202
|
-
const status = JSON.parse(String(res.stdout || ''));
|
|
203
|
-
if (!status || typeof status.loggedIn !== 'boolean') return 'unknown';
|
|
204
|
-
return status.loggedIn ? 'logged-in' : 'logged-out';
|
|
205
|
-
} catch {
|
|
206
|
-
return 'unknown';
|
|
207
|
-
}
|
|
208
|
-
}
|
|
209
|
-
|
|
210
|
-
/**
|
|
211
|
-
* Invoke the local Claude Code model headlessly (uses the USER's subscription auth).
|
|
212
|
-
* Prompt+transcript go via stdin. Returns { ok, out, reason } plus auth/cause
|
|
213
|
-
* metadata — never throws, so the SessionEnd hook degrades gracefully (the
|
|
214
|
-
* proactive auxilo_contribute path is the reliable primary; this deterministic
|
|
215
|
-
* hook is best-effort).
|
|
216
|
-
*/
|
|
217
|
-
function extractWithClaudeCode(transcript, opts = {}) {
|
|
218
|
-
const spawnSyncImpl = typeof opts.spawnSyncImpl === 'function'
|
|
219
|
-
? opts.spawnSyncImpl
|
|
220
|
-
: spawnSync;
|
|
221
|
-
const bin = typeof opts.claudeBin === 'string'
|
|
222
|
-
? opts.claudeBin
|
|
223
|
-
: resolveClaudeBin();
|
|
224
|
-
const authStatus = checkClaudeAuthStatus({ spawnSyncImpl, claudeBin: bin });
|
|
225
|
-
if (authStatus === 'logged-out') {
|
|
226
|
-
return {
|
|
227
|
-
ok: false,
|
|
228
|
-
out: '',
|
|
229
|
-
reason: 'local model not authenticated in this context (run `claude auth login` once); skipping deterministic extraction',
|
|
230
|
-
reasonCode: 'cli-unauthenticated',
|
|
231
|
-
authStatus,
|
|
232
|
-
};
|
|
233
|
-
}
|
|
234
|
-
const prompt = typeof opts.prompt === 'string'
|
|
235
|
-
? opts.prompt
|
|
236
|
-
: buildExtractionPrompt({
|
|
237
|
-
previousLessonsSection: opts.previousLessonsSection,
|
|
238
|
-
captureVisibility: opts.captureVisibility,
|
|
239
|
-
...(opts.scoreExtraction !== undefined && { scoreExtraction: opts.scoreExtraction }),
|
|
240
|
-
});
|
|
241
|
-
const input = prompt + String(transcript).slice(0, 200000);
|
|
242
|
-
// Do NOT pass ANTHROPIC_API_KEY through — we want the user's logged-in Claude
|
|
243
|
-
// subscription (OAuth), not an API key (which would bill someone). Delete it.
|
|
244
|
-
let res;
|
|
245
|
-
try {
|
|
246
|
-
res = spawnSyncImpl(bin, ['-p'], {
|
|
247
|
-
input,
|
|
248
|
-
encoding: 'utf-8',
|
|
249
|
-
env: claudeChildEnv(),
|
|
250
|
-
timeout: 120000,
|
|
251
|
-
maxBuffer: 20 * 1024 * 1024,
|
|
252
|
-
});
|
|
253
|
-
} catch (error) {
|
|
254
|
-
return {
|
|
255
|
-
ok: false,
|
|
256
|
-
out: '',
|
|
257
|
-
reason: `spawn failed (${bin}): ${error.message}`,
|
|
258
|
-
reasonCode: 'unknown',
|
|
259
|
-
authStatus,
|
|
260
|
-
};
|
|
261
|
-
}
|
|
262
|
-
if (!res) {
|
|
263
|
-
return {
|
|
264
|
-
ok: false,
|
|
265
|
-
out: '',
|
|
266
|
-
reason: `spawn failed (${bin}): no process result`,
|
|
267
|
-
reasonCode: 'unknown',
|
|
268
|
-
authStatus,
|
|
269
|
-
};
|
|
270
|
-
}
|
|
271
|
-
const out = String(res.stdout || '');
|
|
272
|
-
if (res.error) {
|
|
273
|
-
return {
|
|
274
|
-
ok: false,
|
|
275
|
-
out: '',
|
|
276
|
-
reason: `spawn failed (${bin}): ${res.error.message}`,
|
|
277
|
-
reasonCode: 'unknown',
|
|
278
|
-
authStatus,
|
|
279
|
-
};
|
|
280
|
-
}
|
|
281
|
-
// Claude prints auth failures ("API Error: 401 ... Please run /login") to stdout.
|
|
282
|
-
if (/Please run \/login|authentication_error|401/i.test(out) || /Please run \/login|authentication_error/i.test(String(res.stderr || ''))) {
|
|
283
|
-
return {
|
|
284
|
-
ok: false,
|
|
285
|
-
out,
|
|
286
|
-
reason: 'local model not authenticated in this context (run `claude auth login` once); skipping deterministic extraction',
|
|
287
|
-
reasonCode: 'cli-unauthenticated',
|
|
288
|
-
authStatus,
|
|
289
|
-
...(authStatus === 'logged-in' && { authDiscrepancy: true }),
|
|
290
|
-
};
|
|
291
|
-
}
|
|
292
|
-
if (res.status !== 0) {
|
|
293
|
-
return {
|
|
294
|
-
ok: false,
|
|
295
|
-
out,
|
|
296
|
-
reason: `local model exited ${res.status}: ${(out || String(res.stderr || '')).slice(0, 160)}`,
|
|
297
|
-
reasonCode: 'model-error',
|
|
298
|
-
authStatus,
|
|
299
|
-
};
|
|
300
|
-
}
|
|
301
|
-
return { ok: true, out, reason: null, authStatus };
|
|
302
|
-
}
|
|
152
|
+
// resolveClaudeBin, claudeChildEnv, checkClaudeAuthStatus, extractWithClaudeCode
|
|
153
|
+
// MOVED to scripts/providers/claude-code.js (EXTRACT-PER-CLIENT W1 PART A).
|
|
154
|
+
// Re-exported below (module.exports) from claudeCodeProvider — required because
|
|
155
|
+
// test/ext-0806b-silent-skip.test.js imports them directly from this module.
|
|
303
156
|
|
|
304
157
|
/**
|
|
305
158
|
* A1: validate a model-emitted quality_self_assessment. Returns the normalized
|
|
@@ -523,59 +376,97 @@ ${JSON.stringify(payload)}`;
|
|
|
523
376
|
return { prompt, rankings };
|
|
524
377
|
}
|
|
525
378
|
|
|
379
|
+
// invokeJudgeWithClaudeCode MOVED into scripts/providers/claude-code.js's
|
|
380
|
+
// runModel(mode:'judge') (EXTRACT-PER-CLIENT W1 PART A). Not re-exported here —
|
|
381
|
+
// grep confirmed no test imports it directly from this module (no dead surface
|
|
382
|
+
// carried).
|
|
383
|
+
//
|
|
384
|
+
// judgeUsage's Claude/Anthropic-specific summing (input_tokens +
|
|
385
|
+
// cache_creation_input_tokens + cache_read_input_tokens) moved with it, into
|
|
386
|
+
// claude-code.js's normalizeJudgeUsage() — that's the provider-specific half.
|
|
387
|
+
// This slimmer, provider-agnostic remainder is what's left once a provider's
|
|
388
|
+
// runModel already returns usage normalized to {input_tokens, output_tokens}
|
|
389
|
+
// (or null): apply the SAME text-length-estimate fallback the original judgeUsage
|
|
390
|
+
// used, generically, for whichever provider ran.
|
|
526
391
|
function judgeUsage(usage, prompt, completion) {
|
|
527
392
|
const numeric = usage && typeof usage === 'object' ? usage : {};
|
|
528
393
|
const directInput = Number(numeric.input_tokens) || 0;
|
|
529
|
-
const cacheCreation = Number(numeric.cache_creation_input_tokens) || 0;
|
|
530
|
-
const cacheRead = Number(numeric.cache_read_input_tokens) || 0;
|
|
531
394
|
const output = Number(numeric.output_tokens) || 0;
|
|
532
395
|
return {
|
|
533
|
-
prompt_tokens: directInput
|
|
534
|
-
estimatePromptTokens(prompt),
|
|
396
|
+
prompt_tokens: directInput || estimatePromptTokens(prompt),
|
|
535
397
|
completion_tokens: output || estimatePromptTokens(completion),
|
|
536
398
|
};
|
|
537
399
|
}
|
|
538
400
|
|
|
539
|
-
|
|
540
|
-
|
|
541
|
-
|
|
542
|
-
|
|
543
|
-
|
|
544
|
-
|
|
545
|
-
|
|
546
|
-
|
|
547
|
-
|
|
548
|
-
|
|
549
|
-
|
|
550
|
-
|
|
551
|
-
|
|
552
|
-
|
|
553
|
-
);
|
|
554
|
-
const stdout = String(res.stdout || '');
|
|
555
|
-
if (res.error) {
|
|
556
|
-
return { ok: false, out: '', reason: `judge spawn failed (${bin}): ${res.error.message}` };
|
|
557
|
-
}
|
|
558
|
-
if (/Please run \/login|authentication_error|401/i.test(stdout) ||
|
|
559
|
-
/Please run \/login|authentication_error/i.test(String(res.stderr || ''))) {
|
|
560
|
-
return { ok: false, out: '', reason: 'local judge model is not authenticated' };
|
|
561
|
-
}
|
|
562
|
-
if (res.status !== 0) {
|
|
563
|
-
return {
|
|
564
|
-
ok: false,
|
|
565
|
-
out: '',
|
|
566
|
-
reason: `local judge exited ${res.status}: ${(stdout || String(res.stderr || '')).slice(0, 160)}`,
|
|
567
|
-
};
|
|
401
|
+
/**
|
|
402
|
+
* PART C — resolve the extraction_model identity for a runModel result.
|
|
403
|
+
* Prefers the additive `identity` field a provider's runModel result may
|
|
404
|
+
* carry (byo-key.js always sets one: {provider:'byo-key', model, version,
|
|
405
|
+
* vendor}). claude-code.js/codex-cli.js shipped (PART A/B) before this stamp
|
|
406
|
+
* existed and don't set one — rather than touch those provider modules
|
|
407
|
+
* (outside this part's disjoint file scope, AGENTS.md's one-build rule),
|
|
408
|
+
* this falls back to the resolved provider id alone (model/version/vendor
|
|
409
|
+
* null) so every provider gets SOME stamp, never silently none. Best-effort:
|
|
410
|
+
* a resolution failure here must never block extraction itself.
|
|
411
|
+
*/
|
|
412
|
+
async function resolveExtractionModelIdentity(runModelResult, opts) {
|
|
413
|
+
if (runModelResult && runModelResult.identity && typeof runModelResult.identity === 'object') {
|
|
414
|
+
return runModelResult.identity;
|
|
568
415
|
}
|
|
569
|
-
let wrapper;
|
|
570
416
|
try {
|
|
571
|
-
|
|
572
|
-
|
|
573
|
-
|
|
574
|
-
|
|
575
|
-
|
|
576
|
-
|
|
577
|
-
|
|
578
|
-
|
|
417
|
+
const resolved = await providers.resolveProvider(opts);
|
|
418
|
+
if (resolved && resolved.ok && resolved.id) {
|
|
419
|
+
return { provider: resolved.id, model: null, version: null, vendor: null };
|
|
420
|
+
}
|
|
421
|
+
} catch { /* identity is best-effort; never block extraction on it */ }
|
|
422
|
+
return null;
|
|
423
|
+
}
|
|
424
|
+
|
|
425
|
+
/**
|
|
426
|
+
* Default invokeModel (extractLocally) — routes through the resolved provider's
|
|
427
|
+
* runModel(mode:'extract'), mapped back to the legacy {ok, out, reason,
|
|
428
|
+
* reasonCode, authStatus} shape extractLocally's caller already expects. opts
|
|
429
|
+
* (spawnSyncImpl/claudeBin/etc.) thread straight through so the injection
|
|
430
|
+
* pattern used by every extractLocally test still works when a test exercises
|
|
431
|
+
* this default path instead of supplying its own opts.invokeModel.
|
|
432
|
+
*/
|
|
433
|
+
async function defaultInvokeModel(transcript, invokeOpts, opts) {
|
|
434
|
+
const result = await providers.runModel({
|
|
435
|
+
...opts,
|
|
436
|
+
prompt: invokeOpts.prompt,
|
|
437
|
+
input: transcript,
|
|
438
|
+
timeoutMs: 120000,
|
|
439
|
+
log: opts.log,
|
|
440
|
+
mode: 'extract',
|
|
441
|
+
});
|
|
442
|
+
return {
|
|
443
|
+
ok: result.ok,
|
|
444
|
+
out: result.text,
|
|
445
|
+
reason: result.reason,
|
|
446
|
+
reasonCode: result.reasonCode,
|
|
447
|
+
authStatus: result.authStatus,
|
|
448
|
+
extractionModel: await resolveExtractionModelIdentity(result, opts),
|
|
449
|
+
...(result.authDiscrepancy !== undefined && { authDiscrepancy: result.authDiscrepancy }),
|
|
450
|
+
};
|
|
451
|
+
}
|
|
452
|
+
|
|
453
|
+
/**
|
|
454
|
+
* Default invokeJudge (runAnchoredJudge) — routes through the resolved
|
|
455
|
+
* provider's runModel(mode:'judge'), mapped back to the legacy {ok, out, usage,
|
|
456
|
+
* reason} shape runAnchoredJudge already consumes. Same opts thread-through as
|
|
457
|
+
* defaultInvokeModel above.
|
|
458
|
+
*/
|
|
459
|
+
function defaultInvokeJudge(opts) {
|
|
460
|
+
return async (prompt) => {
|
|
461
|
+
const result = await providers.runModel({
|
|
462
|
+
...opts,
|
|
463
|
+
prompt,
|
|
464
|
+
timeoutMs: 120000,
|
|
465
|
+
log: opts.log,
|
|
466
|
+
mode: 'judge',
|
|
467
|
+
});
|
|
468
|
+
return { ok: result.ok, out: result.text, usage: result.usage, reason: result.reason };
|
|
469
|
+
};
|
|
579
470
|
}
|
|
580
471
|
|
|
581
472
|
function parseJudgeDecisions(raw, candidates, rankings) {
|
|
@@ -630,7 +521,7 @@ async function runAnchoredJudge(candidates, indexState, opts = {}) {
|
|
|
630
521
|
if (rankings.some((rows) => rows.length === 0)) return empty;
|
|
631
522
|
const invokeJudge = typeof opts.invokeJudge === 'function'
|
|
632
523
|
? opts.invokeJudge
|
|
633
|
-
:
|
|
524
|
+
: defaultInvokeJudge(opts);
|
|
634
525
|
let result;
|
|
635
526
|
try {
|
|
636
527
|
result = await invokeJudge(prompt, { candidates: input, rankings });
|
|
@@ -765,7 +656,7 @@ async function extractLocally(transcript, sourceType, opts = {}) {
|
|
|
765
656
|
});
|
|
766
657
|
const invokeModel = typeof opts.invokeModel === 'function'
|
|
767
658
|
? opts.invokeModel
|
|
768
|
-
: (text, invokeOpts) =>
|
|
659
|
+
: (text, invokeOpts) => defaultInvokeModel(text, invokeOpts, opts);
|
|
769
660
|
const modelResult = await invokeModel(transcript, { prompt });
|
|
770
661
|
const { ok, out, reason } = modelResult;
|
|
771
662
|
if (!ok) {
|
|
@@ -779,13 +670,21 @@ async function extractLocally(transcript, sourceType, opts = {}) {
|
|
|
779
670
|
}),
|
|
780
671
|
};
|
|
781
672
|
}
|
|
673
|
+
// PART C: which provider/model actually ran this extraction, stamped onto
|
|
674
|
+
// every learning it produces (below). Absent when the caller supplied its
|
|
675
|
+
// own opts.invokeModel that doesn't report one (every existing test does
|
|
676
|
+
// this) — extraction_model then simply never appears, byte-identical to
|
|
677
|
+
// pre-PART-C behavior.
|
|
678
|
+
const extractionModel = modelResult.extractionModel || null;
|
|
782
679
|
const parsed = parseExtractionOutput(out, opts);
|
|
783
680
|
const promptDropResult = applyPromptMemoryDrops(
|
|
784
681
|
parsed.prompt_drops,
|
|
785
682
|
promptMemory,
|
|
786
683
|
opts
|
|
787
684
|
);
|
|
788
|
-
const candidates = [...parsed.learnings, ...promptDropResult.restored]
|
|
685
|
+
const candidates = [...parsed.learnings, ...promptDropResult.restored].map((l) => (
|
|
686
|
+
extractionModel ? { ...l, extraction_model: extractionModel } : l
|
|
687
|
+
));
|
|
789
688
|
|
|
790
689
|
// Re-read immediately before the lexical filter. If the index is deleted,
|
|
791
690
|
// becomes unreadable, or is corrupted while the model runs, the filter is
|
|
@@ -840,11 +739,18 @@ async function extractLocally(transcript, sourceType, opts = {}) {
|
|
|
840
739
|
}
|
|
841
740
|
|
|
842
741
|
module.exports = {
|
|
843
|
-
extractLocally,
|
|
844
|
-
parseLearnings, parseExtractionOutput,
|
|
742
|
+
extractLocally, EXTRACTABLE_SOURCES, EXTRACTABLE_SOURCE_IDS,
|
|
743
|
+
parseLearnings, parseExtractionOutput,
|
|
845
744
|
CATEGORIES, PRIVATE_CATEGORIES, RETIRED_CATEGORIES,
|
|
846
745
|
EXTRACTION_PROMPT, buildExtractionPrompt, scoreExtractionEnabled,
|
|
847
746
|
validateQualityAssessment, QUALITY_DIMENSIONS,
|
|
848
747
|
buildAnchoredJudgePrompt, parseJudgeDecisions, runAnchoredJudge,
|
|
849
|
-
recordDropAudit,
|
|
748
|
+
recordDropAudit, JUDGE_TOP_K,
|
|
749
|
+
// Re-exported from scripts/providers/claude-code.js (moved there in PART A) —
|
|
750
|
+
// kept here ONLY because test/ext-0806b-silent-skip.test.js imports these
|
|
751
|
+
// three directly from this module (grep-confirmed; invokeJudgeWithClaudeCode
|
|
752
|
+
// and judgeUsage had no such importer and are NOT carried forward).
|
|
753
|
+
extractWithClaudeCode: claudeCodeProvider.extractWithClaudeCode,
|
|
754
|
+
checkClaudeAuthStatus: claudeCodeProvider.checkClaudeAuthStatus,
|
|
755
|
+
resolveClaudeBin: claudeCodeProvider.resolveClaudeBin,
|
|
850
756
|
};
|