champollion 0.5.2 → 0.5.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/lib/command-help.js +2 -1
- package/lib/commands/models.js +50 -0
- package/lib/config.js +3 -1
- package/lib/fallback.js +23 -13
- package/lib/methods/anthropic.js +2 -1
- package/lib/methods/gemini.js +2 -1
- package/lib/methods/openai.js +4 -3
- package/lib/model-defaults.js +113 -0
- package/package.json +1 -1
- package/shared/docent/corpus.json +13573 -0
- package/shared/model-defaults.json +47 -0
package/lib/command-help.js
CHANGED
|
@@ -711,7 +711,7 @@ const COMMAND_HELP = {
|
|
|
711
711
|
},
|
|
712
712
|
|
|
713
713
|
models: {
|
|
714
|
-
usage: 'champollion models --method <provider>',
|
|
714
|
+
usage: 'champollion models --method <provider> | champollion models check',
|
|
715
715
|
description: [
|
|
716
716
|
'Lists available models from a translation provider\'s API.',
|
|
717
717
|
'Queries the provider\'s live model endpoint and displays all',
|
|
@@ -719,6 +719,7 @@ const COMMAND_HELP = {
|
|
|
719
719
|
],
|
|
720
720
|
options: [
|
|
721
721
|
['--method <name>', 'Provider to query: gemini, openai, or anthropic (required)'],
|
|
722
|
+
['check', 'Check every default model (shared/model-defaults.json) against the providers\' live lists; exit 1 if one is gone. Free (list endpoints are not billed)'],
|
|
722
723
|
['--json', 'Machine-readable JSON output (single document)'],
|
|
723
724
|
],
|
|
724
725
|
examples: [
|
package/lib/commands/models.js
CHANGED
|
@@ -21,6 +21,8 @@ import { resolveConfig } from '../config.js';
|
|
|
21
21
|
import { fetchAvailableModels, resolveProviderApiKey, getProviderLabel, getProviderEnvVar, isListableProvider, getListableProviders } from '../models.js';
|
|
22
22
|
import { missingKeyAdvice } from '../missing-key.js';
|
|
23
23
|
import { output } from '../output.js';
|
|
24
|
+
import { checkModelDefaults } from '../model-defaults.js';
|
|
25
|
+
import { fetchModelPricing } from '../methods/openrouter-pricing.js';
|
|
24
26
|
|
|
25
27
|
/**
|
|
26
28
|
* @param {import('../types.js').CLIArgs} args - Parsed CLI arguments
|
|
@@ -38,6 +40,8 @@ async function run(args, cwd) {
|
|
|
38
40
|
return 0;
|
|
39
41
|
}
|
|
40
42
|
|
|
43
|
+
if (args._[1] === 'check') return runCheck(args, cwd, json);
|
|
44
|
+
|
|
41
45
|
const method = args.method;
|
|
42
46
|
if (!method) {
|
|
43
47
|
if (json) {
|
|
@@ -182,3 +186,49 @@ function showHelp() {
|
|
|
182
186
|
}
|
|
183
187
|
|
|
184
188
|
export { run };
|
|
189
|
+
|
|
190
|
+
/**
|
|
191
|
+
* `champollion models check` — every default model (shared/model-defaults.json)
|
|
192
|
+
* against the providers' live lists. Read-only and free: list endpoints are
|
|
193
|
+
* not billed. Exit 1 when a pinned default is no longer listed (a run would
|
|
194
|
+
* fail on it); a newer model matching a role's rule is reported, not a
|
|
195
|
+
* failure — moving a default is a reviewed change, never automatic.
|
|
196
|
+
*
|
|
197
|
+
* @param {object} args
|
|
198
|
+
* @param {string} cwd
|
|
199
|
+
* @param {boolean} json
|
|
200
|
+
* @returns {Promise<number>}
|
|
201
|
+
*/
|
|
202
|
+
async function runCheck(args, cwd, json) {
|
|
203
|
+
const lists = { openrouter: null, openai: null, gemini: null, anthropic: null };
|
|
204
|
+
try {
|
|
205
|
+
const pricing = await fetchModelPricing();
|
|
206
|
+
if (pricing && pricing.size > 0) lists.openrouter = [...pricing.keys()];
|
|
207
|
+
} catch { /* unchecked */ }
|
|
208
|
+
for (const provider of ['openai', 'gemini', 'anthropic']) {
|
|
209
|
+
const key = resolveProviderApiKey(provider, cwd);
|
|
210
|
+
if (key) lists[provider] = await fetchAvailableModels(provider, key);
|
|
211
|
+
}
|
|
212
|
+
const rows = checkModelDefaults(lists);
|
|
213
|
+
const gone = rows.filter(r => r.status === 'gone');
|
|
214
|
+
if (json) {
|
|
215
|
+
console.log(JSON.stringify({ command: 'models check', ok: gone.length === 0, roles: rows }, null, 2));
|
|
216
|
+
return gone.length > 0 ? 1 : 0;
|
|
217
|
+
}
|
|
218
|
+
output.raw('\n Default models (shared/model-defaults.json) against the providers\' live lists:\n');
|
|
219
|
+
for (const r of rows) {
|
|
220
|
+
const mark = { ok: '[OK] ', newer: '[NEW] ', gone: '[GONE]', unchecked: '[--] ' }[r.status];
|
|
221
|
+
output.raw(` ${mark} ${r.role.padEnd(10)} ${r.model.padEnd(34)} ${r.note}`);
|
|
222
|
+
}
|
|
223
|
+
const unchecked = rows.filter(r => r.status === 'unchecked').map(r => r.provider);
|
|
224
|
+
if (unchecked.length > 0) {
|
|
225
|
+
output.raw(`\n Not checked (no key for ${[...new Set(unchecked)].map(p => getProviderEnvVar(p) || p).join(', ')}): set it to check that provider's list.`);
|
|
226
|
+
}
|
|
227
|
+
output.raw('');
|
|
228
|
+
if (gone.length > 0) {
|
|
229
|
+
output.error(`${gone.length} default model(s) are no longer listed by their provider — a run on them would fail. `
|
|
230
|
+
+ 'Update shared/model-defaults.json (in the repo: node cli/scripts/check-model-defaults.mjs --update) and release.');
|
|
231
|
+
return 1;
|
|
232
|
+
}
|
|
233
|
+
return 0;
|
|
234
|
+
}
|
package/lib/config.js
CHANGED
|
@@ -11,6 +11,7 @@
|
|
|
11
11
|
* complex setups with custom registers, models, and batch sizes.
|
|
12
12
|
*/
|
|
13
13
|
|
|
14
|
+
import { defaultModel } from './model-defaults.js';
|
|
14
15
|
import fs from 'node:fs';
|
|
15
16
|
import path from 'node:path';
|
|
16
17
|
import { DEFAULT_REGISTERS, getLanguageCard, getRegister, resolveCode } from './registers.js';
|
|
@@ -25,7 +26,8 @@ const CONFIG_FILENAMES = ['champollion.config.json'];
|
|
|
25
26
|
|
|
26
27
|
// Canonical defaults — import these in any module that needs a fallback
|
|
27
28
|
// instead of hardcoding the string/number inline.
|
|
28
|
-
|
|
29
|
+
// shared/model-defaults.json, role "translate" — never written here (lib/model-defaults.js).
|
|
30
|
+
const DEFAULT_OPENROUTER_MODEL = defaultModel('translate');
|
|
29
31
|
const DEFAULT_BATCH_SIZE = 80;
|
|
30
32
|
// Max parallel API calls for JSON key-value translation. 50 is kind to
|
|
31
33
|
// free/low-tier keys on a zero-config first run (200 would hammer 429s).
|
package/lib/fallback.js
CHANGED
|
@@ -720,8 +720,9 @@ function sameAnswer(a, b) {
|
|
|
720
720
|
* reason, the model either translates it or returns it unchanged again — and
|
|
721
721
|
* a keep-as-written refusal (isKeepAsWrittenFault) answered the same way
|
|
722
722
|
* twice is taken at its word, as the key-value lane takes a name. Every other
|
|
723
|
-
* fault must clear the gate outright. One ask,
|
|
724
|
-
* it is billed like any other
|
|
723
|
+
* fault must clear the gate outright. One ask per block, alone (not in the
|
|
724
|
+
* page's batch), never a loop; under --max-cost it is billed like any other
|
|
725
|
+
* call (budget.approve).
|
|
725
726
|
*
|
|
726
727
|
* @param {object} p
|
|
727
728
|
* @param {number[]} p.idx - Indices (into `missed`) the gate refused
|
|
@@ -752,17 +753,26 @@ async function retryRefusedBlocks({ idx, texts, missed, blocks, pairConfig, runB
|
|
|
752
753
|
if (!verdict.ok) return accepted;
|
|
753
754
|
}
|
|
754
755
|
const retryNotes = new Map(idx.map(i => [texts[i], reasons.get(i)]));
|
|
755
|
-
|
|
756
|
-
|
|
757
|
-
|
|
758
|
-
|
|
759
|
-
|
|
760
|
-
|
|
761
|
-
|
|
762
|
-
|
|
763
|
-
|
|
764
|
-
|
|
765
|
-
|
|
756
|
+
// Each block alone: away from the page's other segments, the model has one
|
|
757
|
+
// thing to do (and one reason to read). Refusals are rare, so the extra
|
|
758
|
+
// requests cost next to nothing; at most 4 at a time.
|
|
759
|
+
const answers1 = new Map();
|
|
760
|
+
let next = 0;
|
|
761
|
+
const worker = async () => {
|
|
762
|
+
while (next < idx.length) {
|
|
763
|
+
const i = idx[next++];
|
|
764
|
+
try {
|
|
765
|
+
const res = await runBatch([texts[i]], { ...pairConfig, retryNotes });
|
|
766
|
+
if (res && !(res.fellBack || []).includes(0) && typeof res.blocks[0] === 'string') answers1.set(i, res.blocks[0]);
|
|
767
|
+
} catch { /* no second answer for this block */ }
|
|
768
|
+
}
|
|
769
|
+
};
|
|
770
|
+
await Promise.all(Array.from({ length: Math.min(4, idx.length) }, worker));
|
|
771
|
+
idx.forEach((i) => {
|
|
772
|
+
if (!answers1.has(i)) return;
|
|
773
|
+
const protectedOut = answers1.get(i);
|
|
774
|
+
const restored = restoreBlocks(protectedOut, blocks);
|
|
775
|
+
const fault = blockFault(texts[i], protectedOut, restored, missed[i].source, pairConfig);
|
|
766
776
|
const insisted = isKeepAsWrittenFault(fault) && isKeepAsWrittenFault(reasons.get(i))
|
|
767
777
|
&& [firstOut.get(i), missed[i].source].some(v => sameAnswer(v, restored));
|
|
768
778
|
if (!fault || insisted) accepted.set(i, restored);
|
package/lib/methods/anthropic.js
CHANGED
|
@@ -13,6 +13,7 @@
|
|
|
13
13
|
* lives in DirectLLMMethod.
|
|
14
14
|
*/
|
|
15
15
|
|
|
16
|
+
import { defaultModel } from '../model-defaults.js';
|
|
16
17
|
import { DirectLLMMethod } from './direct-llm.js';
|
|
17
18
|
import { fetchAvailableModels } from '../models.js';
|
|
18
19
|
import { estimateLlmCost } from './provider-pricing.js';
|
|
@@ -21,7 +22,7 @@ import { estimateLlmCost } from './provider-pricing.js';
|
|
|
21
22
|
// _getDefaultModel() and again as an inline `pairConfig.model || '...'`
|
|
22
23
|
// fallback in estimateCost() — so the two could drift and price a run
|
|
23
24
|
// against a model it did not use.
|
|
24
|
-
const DEFAULT_MODEL = '
|
|
25
|
+
const DEFAULT_MODEL = defaultModel('anthropic'); // shared/model-defaults.json
|
|
25
26
|
|
|
26
27
|
class AnthropicMethod extends DirectLLMMethod {
|
|
27
28
|
constructor(options = {}) {
|
package/lib/methods/gemini.js
CHANGED
|
@@ -12,13 +12,14 @@
|
|
|
12
12
|
* lives in DirectLLMMethod.
|
|
13
13
|
*/
|
|
14
14
|
|
|
15
|
+
import { defaultModel } from '../model-defaults.js';
|
|
15
16
|
import { DirectLLMMethod } from './direct-llm.js';
|
|
16
17
|
import { fetchAvailableModels } from '../models.js';
|
|
17
18
|
import { estimateLlmCost } from './provider-pricing.js';
|
|
18
19
|
|
|
19
20
|
// The default model, named ONCE — it was previously written twice (here and
|
|
20
21
|
// as an inline fallback in estimateCost()) and the two could drift.
|
|
21
|
-
const DEFAULT_MODEL = 'gemini-
|
|
22
|
+
const DEFAULT_MODEL = defaultModel('gemini'); // shared/model-defaults.json
|
|
22
23
|
|
|
23
24
|
class GeminiMethod extends DirectLLMMethod {
|
|
24
25
|
constructor(options = {}) {
|
package/lib/methods/openai.js
CHANGED
|
@@ -13,15 +13,16 @@
|
|
|
13
13
|
* lives in DirectLLMMethod.
|
|
14
14
|
*/
|
|
15
15
|
|
|
16
|
+
import { defaultModel } from '../model-defaults.js';
|
|
16
17
|
import { DirectLLMMethod } from './direct-llm.js';
|
|
17
18
|
import { fetchAvailableModels } from '../models.js';
|
|
18
19
|
import { estimateLlmCost } from './provider-pricing.js';
|
|
19
20
|
|
|
20
21
|
// The default model, named ONCE — it was previously written twice (here and
|
|
21
22
|
// as an inline fallback in estimateCost()) and the two could drift.
|
|
22
|
-
//
|
|
23
|
-
// ruling 2026-10-05: exact model ids only
|
|
24
|
-
const DEFAULT_MODEL = '
|
|
23
|
+
// shared/model-defaults.json: a dated snapshot (plain 'gpt-4o' was an alias
|
|
24
|
+
// OpenAI repoints — founder ruling 2026-10-05: exact model ids only).
|
|
25
|
+
const DEFAULT_MODEL = defaultModel('openai');
|
|
25
26
|
|
|
26
27
|
class OpenAIMethod extends DirectLLMMethod {
|
|
27
28
|
constructor(options = {}) {
|
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* model-defaults.js — the ONE place a default model comes from.
|
|
3
|
+
*
|
|
4
|
+
* WHY: the toolset's defaults were ten hand-written ids in seven files across
|
|
5
|
+
* five components (CLI, its three direct providers, the harness, forge, the
|
|
6
|
+
* MCP server, the site docent). Nothing kept them in step, so they drifted —
|
|
7
|
+
* the docent still named an undated `claude-haiku-4-5` after every other
|
|
8
|
+
* default had been made exact (founder, 2026-10-07: "we're gonna be getting
|
|
9
|
+
* our wires crossed no matter what doing it that way"). Every default now
|
|
10
|
+
* reads shared/model-defaults.json (bundled here as shared/model-defaults.json).
|
|
11
|
+
*
|
|
12
|
+
* A default is an exact id, fixed in that file — never resolved at run time:
|
|
13
|
+
* a run must say which model translated, and the translation memory is keyed
|
|
14
|
+
* by model, so a default that moved by itself would re-bill every cached
|
|
15
|
+
* string. What IS live is the check: `champollion models check` reads each
|
|
16
|
+
* provider's current list, fails when a pinned id is gone, and names the
|
|
17
|
+
* newest model each role's rule matches today.
|
|
18
|
+
*/
|
|
19
|
+
|
|
20
|
+
import fs from 'node:fs';
|
|
21
|
+
import path from 'node:path';
|
|
22
|
+
import { fileURLToPath } from 'node:url';
|
|
23
|
+
|
|
24
|
+
const __dirname = path.dirname(fileURLToPath(import.meta.url));
|
|
25
|
+
const FILE = path.join(__dirname, '..', 'shared', 'model-defaults.json');
|
|
26
|
+
|
|
27
|
+
let _data = null;
|
|
28
|
+
|
|
29
|
+
/** The parsed file. Missing or malformed is a broken install: fail loud. */
|
|
30
|
+
export function modelDefaultsData() {
|
|
31
|
+
if (_data) return _data;
|
|
32
|
+
let raw;
|
|
33
|
+
try {
|
|
34
|
+
raw = fs.readFileSync(FILE, 'utf-8');
|
|
35
|
+
} catch (err) {
|
|
36
|
+
throw new Error(`${FILE} is missing (${err.code || err.message}): it names every default model. Reinstall champollion.`);
|
|
37
|
+
}
|
|
38
|
+
const data = JSON.parse(raw);
|
|
39
|
+
if (!data || typeof data.roles !== 'object') throw new Error(`${FILE} has no "roles" — it is not a model-defaults file.`);
|
|
40
|
+
_data = data;
|
|
41
|
+
return data;
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
/**
|
|
45
|
+
* The exact default model id for a role ("translate", "openai", "gemini",
|
|
46
|
+
* "anthropic", "harness", "docent"). Unknown role: a programming error.
|
|
47
|
+
*
|
|
48
|
+
* @param {string} role
|
|
49
|
+
* @returns {string}
|
|
50
|
+
*/
|
|
51
|
+
export function defaultModel(role) {
|
|
52
|
+
const r = modelDefaultsData().roles[role];
|
|
53
|
+
if (!r || typeof r.model !== 'string' || !r.model) throw new Error(`model-defaults.json has no "${role}" role.`);
|
|
54
|
+
return r.model;
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
/** Version string → comparable number array ("4-6" / "4.6" → [4, 6]). */
|
|
58
|
+
function versionKey(v) {
|
|
59
|
+
return String(v || '0').split(/[.-]/).map((n) => Number(n) || 0);
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
function compareIds(a, b) {
|
|
63
|
+
const [va, vb] = [versionKey(a.v), versionKey(b.v)];
|
|
64
|
+
for (let i = 0; i < Math.max(va.length, vb.length); i++) {
|
|
65
|
+
if ((va[i] || 0) !== (vb[i] || 0)) return (va[i] || 0) - (vb[i] || 0);
|
|
66
|
+
}
|
|
67
|
+
return String(a.date || '').localeCompare(String(b.date || ''));
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
/**
|
|
71
|
+
* The newest id in a provider's live list that a role's rule matches, or null.
|
|
72
|
+
*
|
|
73
|
+
* @param {object} role - A roles[] entry ({ rule: { match } })
|
|
74
|
+
* @param {string[]} ids - The provider's live model ids
|
|
75
|
+
* @returns {string|null}
|
|
76
|
+
*/
|
|
77
|
+
export function ruleCandidate(role, ids) {
|
|
78
|
+
const re = new RegExp(role.rule.match);
|
|
79
|
+
const hits = [];
|
|
80
|
+
for (const id of ids || []) {
|
|
81
|
+
const m = re.exec(id);
|
|
82
|
+
if (m) hits.push({ id, v: m.groups?.v, date: m.groups?.date });
|
|
83
|
+
}
|
|
84
|
+
if (hits.length === 0) return null;
|
|
85
|
+
hits.sort(compareIds);
|
|
86
|
+
return hits[hits.length - 1].id;
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
/**
|
|
90
|
+
* Check every role against live lists.
|
|
91
|
+
*
|
|
92
|
+
* @param {Record<string, string[]|null>} lists - provider → live ids (null = could not be read)
|
|
93
|
+
* @returns {Array<{ role: string, provider: string, model: string, status: 'ok'|'newer'|'gone'|'unchecked', candidate: string|null, note: string }>}
|
|
94
|
+
*/
|
|
95
|
+
export function checkModelDefaults(lists) {
|
|
96
|
+
const out = [];
|
|
97
|
+
for (const [name, role] of Object.entries(modelDefaultsData().roles)) {
|
|
98
|
+
const ids = lists[role.provider];
|
|
99
|
+
if (!ids) {
|
|
100
|
+
out.push({ role: name, provider: role.provider, model: role.model, status: 'unchecked', candidate: null, note: `${role.provider}'s model list could not be read` });
|
|
101
|
+
continue;
|
|
102
|
+
}
|
|
103
|
+
const candidate = ruleCandidate(role, ids);
|
|
104
|
+
if (!ids.includes(role.model)) {
|
|
105
|
+
out.push({ role: name, provider: role.provider, model: role.model, status: 'gone', candidate, note: `${role.model} is no longer in ${role.provider}'s list${candidate ? ` — the rule's model today: ${candidate}` : ''}` });
|
|
106
|
+
} else if (candidate && candidate !== role.model) {
|
|
107
|
+
out.push({ role: name, provider: role.provider, model: role.model, status: 'newer', candidate, note: `a newer model matches the rule: ${candidate}` });
|
|
108
|
+
} else {
|
|
109
|
+
out.push({ role: name, provider: role.provider, model: role.model, status: 'ok', candidate, note: 'listed' });
|
|
110
|
+
}
|
|
111
|
+
}
|
|
112
|
+
return out;
|
|
113
|
+
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "champollion",
|
|
3
|
-
"version": "0.5.
|
|
3
|
+
"version": "0.5.3",
|
|
4
4
|
"description": "Research-grade translation engine for i18n projects. Pluggable methods, per-pair quality tiers, and deterministic script converters. Supports JSON (next-intl, i18next), TOML, and YAML (Hugo).",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"bin": {
|