tiergear 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +294 -0
- package/dist/cli/bin.js +9 -0
- package/dist/cli/files.js +43 -0
- package/dist/cli/judge.js +19 -0
- package/dist/cli/launch.js +50 -0
- package/dist/cli/main.js +145 -0
- package/dist/cli/node.js +53 -0
- package/dist/cli/orca.js +105 -0
- package/dist/cli/spawn.js +70 -0
- package/dist/cli/stats.js +19 -0
- package/dist/cli/status.js +78 -0
- package/dist/core/config.js +58 -0
- package/dist/core/decide.js +253 -0
- package/dist/core/failures.js +10 -0
- package/dist/core/floor.js +44 -0
- package/dist/core/judge.js +13 -0
- package/dist/core/judges/presets.js +27 -0
- package/dist/core/judges/systemone.js +85 -0
- package/dist/core/log.js +50 -0
- package/dist/core/state.js +72 -0
- package/dist/core/status.js +65 -0
- package/dist/core/tables.js +85 -0
- package/dist/core/tiers.js +53 -0
- package/package.json +38 -0
package/dist/cli/orca.js
ADDED
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
import { execFile } from 'node:child_process';
|
|
2
|
+
export class OrcaError extends Error {
|
|
3
|
+
code;
|
|
4
|
+
constructor(message, code) {
|
|
5
|
+
super(message);
|
|
6
|
+
this.code = code;
|
|
7
|
+
}
|
|
8
|
+
}
|
|
9
|
+
export function pick(json, paths) {
|
|
10
|
+
for (const path of paths) {
|
|
11
|
+
let value = json;
|
|
12
|
+
for (const part of path.split('.')) {
|
|
13
|
+
value = value && typeof value === 'object' ? value[part] : undefined;
|
|
14
|
+
}
|
|
15
|
+
if (value !== undefined && value !== null)
|
|
16
|
+
return value;
|
|
17
|
+
}
|
|
18
|
+
return undefined;
|
|
19
|
+
}
|
|
20
|
+
// A long --json call streams one-line keepalive objects before its result, so drop those first.
|
|
21
|
+
export function parseOrcaOutput(stdout) {
|
|
22
|
+
const body = stdout
|
|
23
|
+
.split('\n')
|
|
24
|
+
.filter((line) => !/^\{"_keepalive":true\b/.test(line.trim()))
|
|
25
|
+
.join('\n')
|
|
26
|
+
.trim();
|
|
27
|
+
return body ? safeJson(body) : null;
|
|
28
|
+
}
|
|
29
|
+
function safeJson(text) {
|
|
30
|
+
try {
|
|
31
|
+
return JSON.parse(text);
|
|
32
|
+
}
|
|
33
|
+
catch {
|
|
34
|
+
return null;
|
|
35
|
+
}
|
|
36
|
+
}
|
|
37
|
+
export function errorCode(json) {
|
|
38
|
+
const code = pick(json, ['error.code', 'code']);
|
|
39
|
+
return typeof code === 'string' ? code : null;
|
|
40
|
+
}
|
|
41
|
+
// Arguments go to execFile as an array: no shell, so a brief's quotes and $ stay literal.
|
|
42
|
+
export function createOrcaExec(binary = process.env['ORCA_CLI_COMMAND'] || 'orca') {
|
|
43
|
+
return (args, cwd) => new Promise((resolve, reject) => {
|
|
44
|
+
execFile(binary, [...args, '--json'], { cwd, maxBuffer: 10 * 1024 * 1024 }, (error, stdout, stderr) => {
|
|
45
|
+
const parsed = parseOrcaOutput(stdout);
|
|
46
|
+
if (error) {
|
|
47
|
+
const detail = (stderr || stdout || error.message).trim().slice(0, 300);
|
|
48
|
+
reject(new OrcaError(`orca ${args.slice(0, 2).join(' ')} failed: ${detail}`, errorCode(parsed) ?? errorCode(safeJson(stderr))));
|
|
49
|
+
return;
|
|
50
|
+
}
|
|
51
|
+
resolve(parsed);
|
|
52
|
+
});
|
|
53
|
+
});
|
|
54
|
+
}
|
|
55
|
+
export function parseWorktree(json) {
|
|
56
|
+
const id = pick(json, ['result.worktree.id', 'worktree.id']);
|
|
57
|
+
if (typeof id !== 'string' || !id.includes('::'))
|
|
58
|
+
throw new Error('orca worktree create returned no worktree.id');
|
|
59
|
+
return { id, path: id.slice(id.indexOf('::') + 2) };
|
|
60
|
+
}
|
|
61
|
+
export function parseHandle(json) {
|
|
62
|
+
const handle = pick(json, ['result.terminal.handle', 'terminal.handle', 'handle']);
|
|
63
|
+
if (typeof handle !== 'string' || !handle)
|
|
64
|
+
throw new Error('orca terminal create returned no handle');
|
|
65
|
+
return handle;
|
|
66
|
+
}
|
|
67
|
+
export function parseRunId(json) {
|
|
68
|
+
const id = pick(json, ['result.run.id', 'run.id']);
|
|
69
|
+
return typeof id === 'string' && id ? id : null;
|
|
70
|
+
}
|
|
71
|
+
export function parseWorkerStart(json) {
|
|
72
|
+
if (pick(json, ['ok']) === false) {
|
|
73
|
+
const code = errorCode(json);
|
|
74
|
+
const message = pick(json, ['error.message']);
|
|
75
|
+
throw new OrcaError(`orca orchestration worker-start failed: ${code ?? 'error'}: ${typeof message === 'string' ? message : ''}`.trim(), code);
|
|
76
|
+
}
|
|
77
|
+
const runId = pick(json, ['result.runId']);
|
|
78
|
+
const taskId = pick(json, ['result.taskId']);
|
|
79
|
+
const dispatchId = pick(json, ['result.dispatchId']);
|
|
80
|
+
const effects = pick(json, ['result.effects']);
|
|
81
|
+
const terminal = Array.isArray(effects)
|
|
82
|
+
? effects.find((e) => !!e && typeof e === 'object' && e.kind === 'terminal' && e.role === 'agent')
|
|
83
|
+
: undefined;
|
|
84
|
+
const handle = terminal?.id;
|
|
85
|
+
if (typeof runId !== 'string' || typeof taskId !== 'string' || typeof dispatchId !== 'string' || typeof handle !== 'string') {
|
|
86
|
+
throw new Error('orca orchestration worker-start returned no dispatch ids or agent terminal');
|
|
87
|
+
}
|
|
88
|
+
return { runId, taskId, dispatchId, handle };
|
|
89
|
+
}
|
|
90
|
+
export function parseSatisfied(json) {
|
|
91
|
+
return pick(json, ['result.wait.satisfied', 'wait.satisfied', 'satisfied']) === true;
|
|
92
|
+
}
|
|
93
|
+
export function parseBlockedReason(json) {
|
|
94
|
+
const reason = pick(json, ['result.wait.blockedReason', 'wait.blockedReason']);
|
|
95
|
+
return typeof reason === 'string' ? reason : null;
|
|
96
|
+
}
|
|
97
|
+
// Orca renames a terminal to its agent, so match on worktree + agent identity first and the title second.
|
|
98
|
+
export function findAgentHandle(json, worktreeId, agent, title) {
|
|
99
|
+
const list = pick(json, ['result.terminals', 'terminals']);
|
|
100
|
+
if (!Array.isArray(list))
|
|
101
|
+
return null;
|
|
102
|
+
const entries = list.filter((t) => !!t && typeof t === 'object');
|
|
103
|
+
const match = entries.find((t) => t.worktreeId === worktreeId && t.agentIdentity === agent) ?? entries.find((t) => t.title === title);
|
|
104
|
+
return typeof match?.handle === 'string' ? match.handle : null;
|
|
105
|
+
}
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
import { OrcaError, findAgentHandle, parseBlockedReason, parseHandle, parseRunId, parseSatisfied, parseWorkerStart, parseWorktree, } from './orca.js';
|
|
2
|
+
export const WAIT_FIRST_MS = 60_000;
|
|
3
|
+
export const WAIT_RETRY_MS = 120_000;
|
|
4
|
+
// The direct path types the brief into the agent's prompt, where a leading `!` runs a shell command and `/` a command.
|
|
5
|
+
export function briefProblem(brief) {
|
|
6
|
+
const start = brief.trimStart()[0];
|
|
7
|
+
if (start === '!' || start === '/')
|
|
8
|
+
return `the brief starts with '${start}', which a Claude Code prompt runs as a command; start it with text`;
|
|
9
|
+
// Line breaks and tabs only: a carriage return submits early and an escape drives the terminal.
|
|
10
|
+
if (/[\u0000-\u0008\u000b-\u001f\u007f-\u009f]/.test(brief))
|
|
11
|
+
return 'the brief contains control characters; only line breaks and tabs are allowed';
|
|
12
|
+
return null;
|
|
13
|
+
}
|
|
14
|
+
export async function spawnWorker(p) {
|
|
15
|
+
// worker-start only works for the coordinator of a bound Run; without one, launch the agent directly.
|
|
16
|
+
const runId = parseRunId(await p.orca(['orchestration', 'run-current']));
|
|
17
|
+
const base = p.baseBranch ? ['--base-branch', p.baseBranch] : [];
|
|
18
|
+
const worktree = parseWorktree(await p.orca(['worktree', 'create', '--name', p.name, '--no-parent', ...base], p.repoDir));
|
|
19
|
+
p.log(`worktree ${worktree.path}`);
|
|
20
|
+
// The floor must exist before the agent reads its first prompt, which worker-start sends itself.
|
|
21
|
+
if (p.harness === 'claude') {
|
|
22
|
+
if (p.plan.floor)
|
|
23
|
+
await p.writeFloor(worktree.path, p.plan.floor);
|
|
24
|
+
else
|
|
25
|
+
p.log('no floor written (judge fallback)');
|
|
26
|
+
}
|
|
27
|
+
const selector = `id:${worktree.id}`;
|
|
28
|
+
if (runId) {
|
|
29
|
+
const { model, effort } = p.plan.target;
|
|
30
|
+
const dispatch = parseWorkerStart(await p.orca([
|
|
31
|
+
'orchestration', 'worker-start', '--spec', p.brief, '--task-title', p.name, '--worktree', selector,
|
|
32
|
+
'--agent', p.harness, '--model', model, ...(effort ? ['--effort', effort] : []),
|
|
33
|
+
// Same budget as the direct path, so there is time to approve a trust prompt in Orca.
|
|
34
|
+
'--timeout-ms', String(WAIT_FIRST_MS + WAIT_RETRY_MS),
|
|
35
|
+
]));
|
|
36
|
+
return { status: 'sent', worktree, handle: dispatch.handle, dispatch };
|
|
37
|
+
}
|
|
38
|
+
let handle = parseHandle(await p.orca(['terminal', 'create', '--worktree', selector, '--title', p.name, '--command', p.plan.command]));
|
|
39
|
+
// After an Orca restart a handle goes stale: re-list once and continue with the replacement only.
|
|
40
|
+
const withHandle = async (args) => {
|
|
41
|
+
try {
|
|
42
|
+
return await p.orca(args(handle));
|
|
43
|
+
}
|
|
44
|
+
catch (error) {
|
|
45
|
+
if (!(error instanceof OrcaError) || error.code !== 'terminal_handle_stale')
|
|
46
|
+
throw error;
|
|
47
|
+
const replacement = findAgentHandle(await p.orca(['terminal', 'list', '--worktree', selector]), worktree.id, p.harness, p.name);
|
|
48
|
+
if (!replacement)
|
|
49
|
+
throw error;
|
|
50
|
+
handle = replacement;
|
|
51
|
+
p.log(`handle refreshed: ${handle}`);
|
|
52
|
+
return p.orca(args(handle));
|
|
53
|
+
}
|
|
54
|
+
};
|
|
55
|
+
const wait = (ms) => withHandle((h) => ['terminal', 'wait', '--terminal', h, '--for', 'tui-idle', '--timeout-ms', String(ms)]);
|
|
56
|
+
const first = await wait(WAIT_FIRST_MS);
|
|
57
|
+
let ready = parseSatisfied(first);
|
|
58
|
+
if (!ready) {
|
|
59
|
+
// Never accept the trust prompt on the user's behalf.
|
|
60
|
+
if (parseBlockedReason(first) === 'agent-trust-workspace') {
|
|
61
|
+
p.log(`Claude is asking whether to trust ${worktree.path}; approve it in Orca — waiting up to 120s`);
|
|
62
|
+
}
|
|
63
|
+
ready = parseSatisfied(await wait(WAIT_RETRY_MS));
|
|
64
|
+
}
|
|
65
|
+
// A prompt typed into a TUI that is still starting is lost, so never send blind.
|
|
66
|
+
if (!ready)
|
|
67
|
+
return { status: 'not-started', worktree, handle, dispatch: null };
|
|
68
|
+
await withHandle((h) => ['terminal', 'send', '--terminal', h, '--text', p.brief, '--enter']);
|
|
69
|
+
return { status: 'sent', worktree, handle, dispatch: null };
|
|
70
|
+
}
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
export function summarize(entries, sinceMs) {
|
|
2
|
+
const recent = entries.filter((e) => e.at >= sinceMs);
|
|
3
|
+
if (recent.length === 0)
|
|
4
|
+
return 'no decisions yet';
|
|
5
|
+
const count = (change) => recent.filter((e) => e.change === change).length;
|
|
6
|
+
const lines = [
|
|
7
|
+
`decisions: ${recent.length}`,
|
|
8
|
+
`set ${count('set')} · up ${count('up')} · down ${count('down')} · hold ${count('hold')}`,
|
|
9
|
+
];
|
|
10
|
+
const judges = [...new Set(recent.map((e) => e.judge))];
|
|
11
|
+
for (const judge of judges) {
|
|
12
|
+
const asked = recent.filter((e) => e.judge === judge && e.outcome !== 'skipped');
|
|
13
|
+
const answered = asked.filter((e) => e.outcome === 'ok');
|
|
14
|
+
const latencies = answered.map((e) => e.ms).filter((ms) => typeof ms === 'number');
|
|
15
|
+
const avg = latencies.length > 0 ? Math.round(latencies.reduce((a, b) => a + b, 0) / latencies.length) : null;
|
|
16
|
+
lines.push(`${judge} answered ${answered.length}/${asked.length}${avg !== null ? ` · avg ${avg}ms` : ''}`);
|
|
17
|
+
}
|
|
18
|
+
return lines.join('\n');
|
|
19
|
+
}
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
import { readFile, readdir } from 'node:fs/promises';
|
|
2
|
+
import { join } from 'node:path';
|
|
3
|
+
import { claudeAlias } from '../core/tiers.js';
|
|
4
|
+
import { formatStatus, parseStatus, statusDir, statusField, statusPath } from '../core/status.js';
|
|
5
|
+
// A status line command gets Claude Code's status JSON on stdin: the session, its model and effort.
|
|
6
|
+
function sessionFromStdin(text) {
|
|
7
|
+
if (!text)
|
|
8
|
+
return null;
|
|
9
|
+
let parsed;
|
|
10
|
+
try {
|
|
11
|
+
parsed = JSON.parse(text);
|
|
12
|
+
}
|
|
13
|
+
catch {
|
|
14
|
+
return null;
|
|
15
|
+
}
|
|
16
|
+
if (!parsed || typeof parsed !== 'object')
|
|
17
|
+
return null;
|
|
18
|
+
const id = parsed.session_id;
|
|
19
|
+
const model = parsed.model?.id;
|
|
20
|
+
const effort = parsed.effort?.level;
|
|
21
|
+
return {
|
|
22
|
+
id: typeof id === 'string' && id ? id : undefined,
|
|
23
|
+
model: typeof model === 'string' && model ? claudeAlias(model) : null,
|
|
24
|
+
effort: typeof effort === 'string' && effort ? effort : null,
|
|
25
|
+
};
|
|
26
|
+
}
|
|
27
|
+
// Until tiergear has a value, the session's own stands in; there is no tier then.
|
|
28
|
+
function withSession(status, info) {
|
|
29
|
+
if (!info || (info.model === null && info.effort === null))
|
|
30
|
+
return status;
|
|
31
|
+
if (status)
|
|
32
|
+
return { ...status, model: status.model ?? info.model, effort: status.effort ?? info.effort };
|
|
33
|
+
return { session: info.id ?? '', tier: null, model: info.model, effort: info.effort, paused: false, reason: 'session', line: '', updatedAt: 0 };
|
|
34
|
+
}
|
|
35
|
+
// Status files and stdin are read from disk and another program; a status line is a terminal, so no escape gets through.
|
|
36
|
+
const MAX_FIELD_CHARS = 200;
|
|
37
|
+
function printable(text) {
|
|
38
|
+
return text.replace(/[\u0000-\u001f\u007f-\u009f]/g, '').slice(0, MAX_FIELD_CHARS);
|
|
39
|
+
}
|
|
40
|
+
function printableStatus(status) {
|
|
41
|
+
if (!status)
|
|
42
|
+
return null;
|
|
43
|
+
return {
|
|
44
|
+
...status,
|
|
45
|
+
session: printable(status.session),
|
|
46
|
+
model: status.model === null ? null : printable(status.model),
|
|
47
|
+
effort: typeof status.effort === 'string' ? printable(status.effort) : status.effort,
|
|
48
|
+
reason: printable(status.reason),
|
|
49
|
+
line: printable(status.line),
|
|
50
|
+
};
|
|
51
|
+
}
|
|
52
|
+
async function readStatus(path) {
|
|
53
|
+
const text = await readFile(path, 'utf8').catch(() => null);
|
|
54
|
+
return text === null ? null : parseStatus(text);
|
|
55
|
+
}
|
|
56
|
+
async function newestStatus(home) {
|
|
57
|
+
let newest = null;
|
|
58
|
+
for (const name of await readdir(statusDir(home)).catch(() => [])) {
|
|
59
|
+
if (!name.endsWith('.json'))
|
|
60
|
+
continue;
|
|
61
|
+
const status = await readStatus(join(statusDir(home), name));
|
|
62
|
+
if (status && (newest === null || status.updatedAt > newest.updatedAt))
|
|
63
|
+
newest = status;
|
|
64
|
+
}
|
|
65
|
+
return newest;
|
|
66
|
+
}
|
|
67
|
+
/** What `tiergear status` prints: the session named by --session, else by stdin, else the one updated last. */
|
|
68
|
+
export async function statusCommand(p) {
|
|
69
|
+
const info = p.session === undefined ? sessionFromStdin(await p.stdin()) : null;
|
|
70
|
+
const session = p.session ?? info?.id;
|
|
71
|
+
const stored = session === undefined ? await newestStatus(p.home) : await readStatus(statusPath(p.home, session));
|
|
72
|
+
const status = printableStatus(withSession(stored, info));
|
|
73
|
+
if (p.field)
|
|
74
|
+
return status ? statusField(status, p.field) : '';
|
|
75
|
+
if (p.json)
|
|
76
|
+
return JSON.stringify(status);
|
|
77
|
+
return status ? formatStatus(status, p.format) : '';
|
|
78
|
+
}
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
import { JUDGE_PRESETS, isJudgeName } from './judges/presets.js';
|
|
2
|
+
// A hook has 10s; the judge gets at most 8s of it so the rest of the hook still fits.
|
|
3
|
+
export const MAX_JUDGE_TIMEOUT_MS = 8000;
|
|
4
|
+
const NUMBER_KEYS = [
|
|
5
|
+
'minUpgradeConfidence',
|
|
6
|
+
'minDowngradeConfidence',
|
|
7
|
+
'downgradeStreak',
|
|
8
|
+
'stuckConfidence',
|
|
9
|
+
'stuckFailures',
|
|
10
|
+
'firstTurnTimeoutMs',
|
|
11
|
+
'turnTimeoutMs',
|
|
12
|
+
];
|
|
13
|
+
function text(options, key) {
|
|
14
|
+
const value = options[key];
|
|
15
|
+
return typeof value === 'string' && value ? value : undefined;
|
|
16
|
+
}
|
|
17
|
+
export function resolveConfig(options) {
|
|
18
|
+
const judge = isJudgeName(options['judge']) ? options['judge'] : 'jev';
|
|
19
|
+
const preset = JUDGE_PRESETS[judge];
|
|
20
|
+
const config = {
|
|
21
|
+
judge,
|
|
22
|
+
judgeBaseUrl: text(options, 'judgeBaseUrl') ?? preset.baseUrl,
|
|
23
|
+
judgeModel: text(options, 'judgeModel') ?? preset.model,
|
|
24
|
+
switchModelMidSession: options['switchModelMidSession'] === true,
|
|
25
|
+
minUpgradeConfidence: 0.5,
|
|
26
|
+
minDowngradeConfidence: 0.85,
|
|
27
|
+
downgradeStreak: 2,
|
|
28
|
+
stuckConfidence: 0.6,
|
|
29
|
+
stuckFailures: 3,
|
|
30
|
+
firstTurnTimeoutMs: preset.firstTurnTimeoutMs,
|
|
31
|
+
turnTimeoutMs: preset.turnTimeoutMs,
|
|
32
|
+
showRecentButton: options['showRecentButton'] !== false,
|
|
33
|
+
showStatusText: options['showStatusText'] !== false,
|
|
34
|
+
showTierButtons: options['showTierButtons'] !== false,
|
|
35
|
+
showPrefix: options['showPrefix'] !== false,
|
|
36
|
+
showTier: options['showTier'] !== false,
|
|
37
|
+
showConfidence: options['showConfidence'] !== false,
|
|
38
|
+
showModelEffort: options['showModelEffort'] !== false,
|
|
39
|
+
showReason: options['showReason'] !== false,
|
|
40
|
+
instantSwitchConfidence: null,
|
|
41
|
+
showFloor: options['showFloor'] !== false,
|
|
42
|
+
};
|
|
43
|
+
for (const key of NUMBER_KEYS) {
|
|
44
|
+
const value = options[key];
|
|
45
|
+
if (typeof value === 'number' && Number.isFinite(value))
|
|
46
|
+
config[key] = value;
|
|
47
|
+
}
|
|
48
|
+
config.firstTurnTimeoutMs = Math.min(config.firstTurnTimeoutMs, MAX_JUDGE_TIMEOUT_MS);
|
|
49
|
+
config.turnTimeoutMs = Math.min(config.turnTimeoutMs, MAX_JUDGE_TIMEOUT_MS);
|
|
50
|
+
const instant = options['instantSwitchConfidence'];
|
|
51
|
+
if (typeof instant === 'number' && Number.isFinite(instant) && instant > 0)
|
|
52
|
+
config.instantSwitchConfidence = instant;
|
|
53
|
+
const apiKey = text(options, 'judgeApiKey');
|
|
54
|
+
if (apiKey)
|
|
55
|
+
config.judgeApiKey = apiKey;
|
|
56
|
+
return config;
|
|
57
|
+
}
|
|
58
|
+
export const DEFAULT_CONFIG = resolveConfig({});
|
|
@@ -0,0 +1,253 @@
|
|
|
1
|
+
import { abridge } from './state.js';
|
|
2
|
+
import { effortFor, firstTarget } from './tables.js';
|
|
3
|
+
import { isEffort, isTier, maxTier, stepDown, stepUp, tierRank } from './tiers.js';
|
|
4
|
+
export const JUDGE_FAILURE_LIMIT = 3;
|
|
5
|
+
export const JUDGE_PAUSE_MS = 5 * 60_000;
|
|
6
|
+
export const RECORD_TTL_MS = 7 * 24 * 60 * 60_000;
|
|
7
|
+
// $.store is one JSON file that rejects writes past 4 MiB, so records are capped in count and size.
|
|
8
|
+
export const MAX_RECORDS = 200;
|
|
9
|
+
export const FIRST_PROMPT_CHARS = 2000;
|
|
10
|
+
export function newRecord(firstPrompt, now) {
|
|
11
|
+
return {
|
|
12
|
+
// The judge only ever sees this many characters of it.
|
|
13
|
+
firstPrompt: abridge(firstPrompt, FIRST_PROMPT_CHARS),
|
|
14
|
+
tier: null,
|
|
15
|
+
floor: null,
|
|
16
|
+
ceiling: null,
|
|
17
|
+
model: null,
|
|
18
|
+
applied: null,
|
|
19
|
+
downStreak: 0,
|
|
20
|
+
pinned: false,
|
|
21
|
+
started: true,
|
|
22
|
+
judgeFailures: 0,
|
|
23
|
+
judgePausedUntil: 0,
|
|
24
|
+
updatedAt: now,
|
|
25
|
+
};
|
|
26
|
+
}
|
|
27
|
+
function count(value) {
|
|
28
|
+
return typeof value === 'number' && Number.isFinite(value) ? value : 0;
|
|
29
|
+
}
|
|
30
|
+
function readApplied(value) {
|
|
31
|
+
if (!value || typeof value !== 'object')
|
|
32
|
+
return null;
|
|
33
|
+
const { model, effort } = value;
|
|
34
|
+
if (effort !== null && !isEffort(effort))
|
|
35
|
+
return null;
|
|
36
|
+
return typeof model === 'string' ? { model, effort } : { effort };
|
|
37
|
+
}
|
|
38
|
+
export function parseRecord(value) {
|
|
39
|
+
if (!value || typeof value !== 'object')
|
|
40
|
+
return null;
|
|
41
|
+
const r = value;
|
|
42
|
+
if (typeof r['firstPrompt'] !== 'string' || typeof r['updatedAt'] !== 'number')
|
|
43
|
+
return null;
|
|
44
|
+
const tier = r['tier'];
|
|
45
|
+
const floor = r['floor'];
|
|
46
|
+
if (tier !== null && !isTier(tier))
|
|
47
|
+
return null;
|
|
48
|
+
if (floor !== null && !isTier(floor))
|
|
49
|
+
return null;
|
|
50
|
+
return {
|
|
51
|
+
firstPrompt: r['firstPrompt'],
|
|
52
|
+
tier,
|
|
53
|
+
floor,
|
|
54
|
+
// Records stored before ceilings have none.
|
|
55
|
+
ceiling: isTier(r['ceiling']) ? r['ceiling'] : null,
|
|
56
|
+
model: typeof r['model'] === 'string' ? r['model'] : null,
|
|
57
|
+
applied: readApplied(r['applied']),
|
|
58
|
+
downStreak: count(r['downStreak']),
|
|
59
|
+
pinned: r['pinned'] === true,
|
|
60
|
+
// Records stored before this mark were all made by a prompt.
|
|
61
|
+
started: r['started'] !== false,
|
|
62
|
+
judgeFailures: count(r['judgeFailures']),
|
|
63
|
+
judgePausedUntil: count(r['judgePausedUntil']),
|
|
64
|
+
updatedAt: r['updatedAt'],
|
|
65
|
+
};
|
|
66
|
+
}
|
|
67
|
+
function appliedFor(model, tier, config, tables) {
|
|
68
|
+
if (config.switchModelMidSession) {
|
|
69
|
+
const target = firstTarget(tables, 'claude', tier);
|
|
70
|
+
return { model: target.model, effort: target.effort };
|
|
71
|
+
}
|
|
72
|
+
// Model held: only the effort follows the tier, from the session model's column in table B.
|
|
73
|
+
const effort = effortFor(tables, 'claude', model ?? '', tier);
|
|
74
|
+
if (model && effort === null) {
|
|
75
|
+
// A model without effort cannot be raised by effort alone, so move to the tier's own model.
|
|
76
|
+
const target = firstTarget(tables, 'claude', tier);
|
|
77
|
+
if (target.model !== model)
|
|
78
|
+
return { model: target.model, effort: target.effort };
|
|
79
|
+
}
|
|
80
|
+
return model ? { model, effort } : { effort };
|
|
81
|
+
}
|
|
82
|
+
// A paused session runs on its own model and effort: nothing is applied.
|
|
83
|
+
function pausedHold(record, confidence) {
|
|
84
|
+
return { record: { ...record, pinned: true, applied: null, downStreak: 0 }, change: 'hold', confidence, reason: 'paused' };
|
|
85
|
+
}
|
|
86
|
+
// Paused, tiergear's held model is not in effect: the session's own is.
|
|
87
|
+
function heldModel(record, sessionModel) {
|
|
88
|
+
return record.pinned ? (sessionModel ?? record.model) : (record.model ?? sessionModel);
|
|
89
|
+
}
|
|
90
|
+
// Before the first prompt there is no cache to lose, so the model follows the tier as on a first turn.
|
|
91
|
+
function appliedForTier(record, model, tier, config, tables) {
|
|
92
|
+
if (record.started)
|
|
93
|
+
return appliedFor(model, tier, config, tables);
|
|
94
|
+
const target = firstTarget(tables, 'claude', tier);
|
|
95
|
+
return { model: target.model, effort: target.effort };
|
|
96
|
+
}
|
|
97
|
+
/** A floor set by hand: a tier under it rises to it; a launch ceiling still bounds it. */
|
|
98
|
+
export function decideSetFloor(input) {
|
|
99
|
+
const { record, config, tables } = input;
|
|
100
|
+
const floor = record.ceiling !== null && tierRank(input.floor) > tierRank(record.ceiling) ? record.ceiling : input.floor;
|
|
101
|
+
const floored = { ...record, floor };
|
|
102
|
+
if (record.pinned || record.tier === null || tierRank(record.tier) >= tierRank(floor)) {
|
|
103
|
+
return { record: floored, change: 'hold', confidence: null, reason: 'floor set' };
|
|
104
|
+
}
|
|
105
|
+
const model = heldModel(record, input.sessionModel);
|
|
106
|
+
const applied = appliedForTier(record, model, floor, config, tables);
|
|
107
|
+
return { record: { ...floored, tier: floor, applied, model: applied.model ?? model, downStreak: 0 }, change: 'up', confidence: null, reason: 'floor set' };
|
|
108
|
+
}
|
|
109
|
+
export function decidePause(record) {
|
|
110
|
+
return pausedHold(record, null);
|
|
111
|
+
}
|
|
112
|
+
/** A tier picked by hand: a starting point the judge keeps moving by the usual rules. */
|
|
113
|
+
export function decidePick(input) {
|
|
114
|
+
const { record, tier, config, tables } = input;
|
|
115
|
+
const model = heldModel(record, input.sessionModel);
|
|
116
|
+
const applied = appliedForTier(record, model, tier, config, tables);
|
|
117
|
+
const floor = record.floor === null ? stepDown(tier) : tierRank(tier) < tierRank(record.floor) ? tier : record.floor;
|
|
118
|
+
return {
|
|
119
|
+
record: { ...record, tier, floor, pinned: false, applied, model: applied.model ?? model, downStreak: 0 },
|
|
120
|
+
change: 'set',
|
|
121
|
+
confidence: null,
|
|
122
|
+
reason: 'manual tier',
|
|
123
|
+
};
|
|
124
|
+
}
|
|
125
|
+
export function decideFirstTurn(input) {
|
|
126
|
+
const { record, floor, verdict, config, tables } = input;
|
|
127
|
+
if (record.pinned)
|
|
128
|
+
return pausedHold(record, null);
|
|
129
|
+
if (!record.started && record.tier !== null) {
|
|
130
|
+
// Picked before the first prompt: the pick stands for this turn, and a launch still bounds the session.
|
|
131
|
+
const launched = floor ? { floor: tierRank(floor.tier) < tierRank(record.tier) ? floor.tier : record.tier, ceiling: floor.ceiling ?? null } : {};
|
|
132
|
+
return { record: { ...record, ...launched }, change: 'hold', confidence: null, reason: 'manual tier' };
|
|
133
|
+
}
|
|
134
|
+
if (floor) {
|
|
135
|
+
const target = firstTarget(tables, 'claude', floor.tier);
|
|
136
|
+
return {
|
|
137
|
+
record: {
|
|
138
|
+
...record,
|
|
139
|
+
tier: floor.tier,
|
|
140
|
+
floor: floor.tier,
|
|
141
|
+
ceiling: floor.ceiling ?? null,
|
|
142
|
+
model: target.model,
|
|
143
|
+
applied: { model: target.model, effort: target.effort },
|
|
144
|
+
},
|
|
145
|
+
change: 'set',
|
|
146
|
+
confidence: null,
|
|
147
|
+
reason: `launched at ${floor.tier}`,
|
|
148
|
+
};
|
|
149
|
+
}
|
|
150
|
+
const answer = verdict?.tier ?? null;
|
|
151
|
+
if (!answer || answer.confidence < config.minUpgradeConfidence) {
|
|
152
|
+
return { record, change: 'hold', confidence: answer?.confidence ?? null, reason: answer ? 'low confidence' : 'no answer' };
|
|
153
|
+
}
|
|
154
|
+
// The first turn has no cache to lose, so the model changes with the tier; a floor set by hand still holds.
|
|
155
|
+
const tier = maxTier(answer.tier, record.floor);
|
|
156
|
+
const target = firstTarget(tables, 'claude', tier);
|
|
157
|
+
return {
|
|
158
|
+
record: { ...record, tier, floor: record.floor ?? stepDown(answer.tier), model: target.model, applied: { model: target.model, effort: target.effort } },
|
|
159
|
+
change: 'set',
|
|
160
|
+
confidence: answer.confidence,
|
|
161
|
+
reason: 'first turn',
|
|
162
|
+
};
|
|
163
|
+
}
|
|
164
|
+
export function decideNextTurn(input) {
|
|
165
|
+
const { record, verdict, config, tables } = input;
|
|
166
|
+
const model = record.model ?? input.sessionModel;
|
|
167
|
+
const answer = verdict?.tier ?? null;
|
|
168
|
+
const confidence = answer?.confidence ?? null;
|
|
169
|
+
const hold = (reason, downStreak = 0) => ({ record: { ...record, downStreak }, change: 'hold', confidence, reason });
|
|
170
|
+
const move = (tier, change, reason) => {
|
|
171
|
+
const applied = appliedFor(model, tier, config, tables);
|
|
172
|
+
return { record: { ...record, tier, applied, model: applied.model ?? model, downStreak: 0 }, change, confidence, reason };
|
|
173
|
+
};
|
|
174
|
+
if (record.pinned)
|
|
175
|
+
return pausedHold(record, confidence);
|
|
176
|
+
if (record.tier === null) {
|
|
177
|
+
if (!answer || answer.confidence < config.minUpgradeConfidence)
|
|
178
|
+
return hold(answer ? 'low confidence' : 'no answer');
|
|
179
|
+
const decided = move(maxTier(answer.tier, record.floor), 'set', 'tier decided');
|
|
180
|
+
return { ...decided, record: { ...decided.record, floor: record.floor ?? stepDown(answer.tier) } };
|
|
181
|
+
}
|
|
182
|
+
const current = record.tier;
|
|
183
|
+
const raise = (tier, reason) => {
|
|
184
|
+
const capped = record.ceiling !== null && tierRank(tier) > tierRank(record.ceiling) ? record.ceiling : tier;
|
|
185
|
+
return tierRank(capped) > tierRank(current) ? move(capped, 'up', reason) : hold('at ceiling');
|
|
186
|
+
};
|
|
187
|
+
if (answer && tierRank(answer.tier) > tierRank(current) && answer.confidence >= config.minUpgradeConfidence) {
|
|
188
|
+
return raise(answer.tier, 'harder step');
|
|
189
|
+
}
|
|
190
|
+
const stuck = (verdict?.stuck ?? 0) >= config.stuckConfidence || input.repeatedFailures >= config.stuckFailures;
|
|
191
|
+
if (stuck)
|
|
192
|
+
return current === 'max' ? hold('stuck at max') : raise(stepUp(current), 'stuck');
|
|
193
|
+
// A judge sure enough for the instant switch lowers at once, straight to its tier, never under the floor.
|
|
194
|
+
if (answer && tierRank(answer.tier) < tierRank(current) && config.instantSwitchConfidence !== null && answer.confidence >= config.instantSwitchConfidence) {
|
|
195
|
+
const lower = maxTier(answer.tier, record.floor);
|
|
196
|
+
return lower === current ? hold('at floor') : move(lower, 'down', 'instant switch');
|
|
197
|
+
}
|
|
198
|
+
// Lowering needs a confident judge on consecutive turns, so one terse follow-up cannot drop the tier.
|
|
199
|
+
if (answer && tierRank(answer.tier) < tierRank(current) && answer.confidence >= config.minDowngradeConfidence) {
|
|
200
|
+
const streak = record.downStreak + 1;
|
|
201
|
+
if (streak < config.downgradeStreak)
|
|
202
|
+
return hold(`easier step ${streak}/${config.downgradeStreak}`, streak);
|
|
203
|
+
const lower = maxTier(stepDown(current), record.floor);
|
|
204
|
+
return lower === current ? hold('at floor') : move(lower, 'down', 'easier steps');
|
|
205
|
+
}
|
|
206
|
+
if (!answer)
|
|
207
|
+
return hold('no answer');
|
|
208
|
+
const bar = tierRank(answer.tier) > tierRank(current) ? config.minUpgradeConfidence : tierRank(answer.tier) < tierRank(current) ? config.minDowngradeConfidence : 0;
|
|
209
|
+
return hold(answer.confidence < bar ? 'low confidence' : 'same tier');
|
|
210
|
+
}
|
|
211
|
+
export function canAskJudge(record, now) {
|
|
212
|
+
return now >= record.judgePausedUntil;
|
|
213
|
+
}
|
|
214
|
+
export function noteJudgeOutcome(record, ok, now) {
|
|
215
|
+
if (ok)
|
|
216
|
+
return { ...record, judgeFailures: 0 };
|
|
217
|
+
const failures = record.judgeFailures + 1;
|
|
218
|
+
return failures >= JUDGE_FAILURE_LIMIT
|
|
219
|
+
? { ...record, judgeFailures: 0, judgePausedUntil: now + JUDGE_PAUSE_MS }
|
|
220
|
+
: { ...record, judgeFailures: failures };
|
|
221
|
+
}
|
|
222
|
+
export function appliedText(applied) {
|
|
223
|
+
const effort = applied.effort ?? '-';
|
|
224
|
+
return applied.model ? `${applied.model}/${effort}` : effort;
|
|
225
|
+
}
|
|
226
|
+
// tiergear's override wins over the engine's values; a missing piece is the engine's.
|
|
227
|
+
export function inEffect(applied, engine) {
|
|
228
|
+
if (!applied)
|
|
229
|
+
return engine;
|
|
230
|
+
return { model: applied.model ?? engine?.model ?? null, effort: applied.effort };
|
|
231
|
+
}
|
|
232
|
+
function inEffectText(current) {
|
|
233
|
+
const effort = current.effort === null ? '-' : String(current.effort);
|
|
234
|
+
return current.model ? `${current.model}/${effort}` : effort;
|
|
235
|
+
}
|
|
236
|
+
const SHOW_ALL = { tier: true, confidence: true, modelEffort: true, reason: true };
|
|
237
|
+
// The line without its `tiergear ·` prefix: `deep 0.91 → opus/xhigh`, or `deep 0.62 · sonnet/medium · unchanged (same tier)`.
|
|
238
|
+
export function statusParts(decision, current, show) {
|
|
239
|
+
const { record, change, confidence, reason } = decision;
|
|
240
|
+
const head = [show.tier ? (record.tier ?? 'unset') : null, show.confidence ? (confidence === null ? 'n/d' : confidence.toFixed(2)) : null]
|
|
241
|
+
.filter((part) => part !== null)
|
|
242
|
+
.join(' ');
|
|
243
|
+
if (change === 'hold' || !record.applied) {
|
|
244
|
+
const now = show.modelEffort && current ? inEffectText(current) : '';
|
|
245
|
+
return [head, now, show.reason ? `unchanged (${reason})` : ''].filter((part) => part).join(' · ');
|
|
246
|
+
}
|
|
247
|
+
if (!show.modelEffort)
|
|
248
|
+
return head;
|
|
249
|
+
return [head, `→ ${current ? inEffectText(current) : appliedText(record.applied)}`].filter((part) => part).join(' ');
|
|
250
|
+
}
|
|
251
|
+
export function statusText(decision, current = null) {
|
|
252
|
+
return `tiergear · ${statusParts(decision, current, SHOW_ALL)}`;
|
|
253
|
+
}
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
export const EMPTY_TRACKER = { signature: null, count: 0 };
|
|
2
|
+
export function failureSignature(tool, errorText) {
|
|
3
|
+
return `${tool}:${errorText.trim().slice(0, 120)}`;
|
|
4
|
+
}
|
|
5
|
+
export function recordFailure(tracker, signature) {
|
|
6
|
+
return tracker.signature === signature ? { signature, count: tracker.count + 1 } : { signature, count: 1 };
|
|
7
|
+
}
|
|
8
|
+
export function recordSuccess(tracker, tool) {
|
|
9
|
+
return tracker.signature?.startsWith(`${tool}:`) ? EMPTY_TRACKER : tracker;
|
|
10
|
+
}
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
import { isTier, tierRank } from './tiers.js';
|
|
2
|
+
export const FLOOR_TTL_MS = 24 * 60 * 60 * 1000;
|
|
3
|
+
export function normalizePath(path) {
|
|
4
|
+
return path.length > 1 ? path.replace(/\/+$/, '') : path;
|
|
5
|
+
}
|
|
6
|
+
// FNV-1a: stable in the hook runtime and in Node without a crypto module.
|
|
7
|
+
export function fnv1a(text) {
|
|
8
|
+
let hash = 0x811c9dc5;
|
|
9
|
+
for (let i = 0; i < text.length; i++) {
|
|
10
|
+
hash ^= text.charCodeAt(i);
|
|
11
|
+
hash = Math.imul(hash, 0x01000193) >>> 0;
|
|
12
|
+
}
|
|
13
|
+
return hash.toString(16).padStart(8, '0');
|
|
14
|
+
}
|
|
15
|
+
export function floorPath(home, worktree) {
|
|
16
|
+
return `${home}/.local/state/tiergear/floors/${fnv1a(normalizePath(worktree))}.json`;
|
|
17
|
+
}
|
|
18
|
+
export function serializeFloor(record) {
|
|
19
|
+
return JSON.stringify(record);
|
|
20
|
+
}
|
|
21
|
+
export function parseFloor(text, worktree, now) {
|
|
22
|
+
let parsed;
|
|
23
|
+
try {
|
|
24
|
+
parsed = JSON.parse(text);
|
|
25
|
+
}
|
|
26
|
+
catch {
|
|
27
|
+
return null;
|
|
28
|
+
}
|
|
29
|
+
if (!parsed || typeof parsed !== 'object')
|
|
30
|
+
return null;
|
|
31
|
+
const record = parsed;
|
|
32
|
+
if (typeof record.worktree !== 'string')
|
|
33
|
+
return null;
|
|
34
|
+
if (normalizePath(record.worktree) !== normalizePath(worktree))
|
|
35
|
+
return null;
|
|
36
|
+
if (!isTier(record.tier) || typeof record.createdAt !== 'number')
|
|
37
|
+
return null;
|
|
38
|
+
if (record.ceiling !== undefined && (!isTier(record.ceiling) || tierRank(record.ceiling) < tierRank(record.tier)))
|
|
39
|
+
return null;
|
|
40
|
+
if (now - record.createdAt > FLOOR_TTL_MS)
|
|
41
|
+
return null;
|
|
42
|
+
const floor = { worktree: record.worktree, tier: record.tier, createdAt: record.createdAt };
|
|
43
|
+
return record.ceiling === undefined ? floor : { ...floor, ceiling: record.ceiling };
|
|
44
|
+
}
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
const TIMED_OUT = Symbol('timed out');
|
|
2
|
+
/** Runs a judge call under a deadline and turns every failure into a result, so a turn never waits or throws. */
|
|
3
|
+
export async function guarded(work, sleep, timeoutMs) {
|
|
4
|
+
try {
|
|
5
|
+
const result = await Promise.race([work(), sleep(timeoutMs).then(() => TIMED_OUT)]);
|
|
6
|
+
if (result === TIMED_OUT)
|
|
7
|
+
return { ok: false, reason: `timed out after ${timeoutMs}ms` };
|
|
8
|
+
return { ok: true, verdict: result };
|
|
9
|
+
}
|
|
10
|
+
catch (error) {
|
|
11
|
+
return { ok: false, reason: error instanceof Error ? error.message : String(error) };
|
|
12
|
+
}
|
|
13
|
+
}
|