@intentic/sandbox-contract 1.298.0 → 1.300.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/contracts/agent.contract.d.ts +6 -7
- package/dist/contracts/agent.contract.d.ts.map +1 -1
- package/dist/contracts/agents.contract.d.ts +0 -81
- package/dist/contracts/agents.contract.d.ts.map +1 -1
- package/dist/contracts/capabilities.contract.d.ts +1 -0
- package/dist/contracts/capabilities.contract.d.ts.map +1 -1
- package/dist/contracts/runner.contract.d.ts +0 -16
- package/dist/contracts/runner.contract.d.ts.map +1 -1
- package/dist/contracts/sessions.contract.d.ts +0 -1
- package/dist/contracts/sessions.contract.d.ts.map +1 -1
- package/dist/contracts/settings.contract.d.ts +0 -33
- package/dist/contracts/settings.contract.d.ts.map +1 -1
- package/dist/contracts/system.contract.d.ts +0 -5
- package/dist/contracts/system.contract.d.ts.map +1 -1
- package/dist/contracts/workspace.contract.d.ts +1 -0
- package/dist/contracts/workspace.contract.d.ts.map +1 -1
- package/dist/events/agent-events.d.ts +1 -15
- package/dist/events/agent-events.d.ts.map +1 -1
- package/dist/events/agent-events.js +0 -10
- package/dist/events/agent-events.js.map +1 -1
- package/dist/events/system-events.d.ts +0 -10
- package/dist/events/system-events.d.ts.map +1 -1
- package/dist/events/transcript.d.ts +0 -6
- package/dist/events/transcript.d.ts.map +1 -1
- package/dist/events/transcript.js +1 -1
- package/dist/events/transcript.js.map +1 -1
- package/dist/index.d.ts +38 -159
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +0 -2
- package/dist/index.js.map +1 -1
- package/dist/models/model-order.d.ts +0 -2
- package/dist/models/model-order.d.ts.map +1 -1
- package/dist/models/model-order.js +0 -13
- package/dist/models/model-order.js.map +1 -1
- package/dist/schemas/agent.d.ts +0 -1
- package/dist/schemas/agent.d.ts.map +1 -1
- package/dist/schemas/agent.js +0 -4
- package/dist/schemas/agent.js.map +1 -1
- package/dist/schemas/agents.d.ts +0 -15
- package/dist/schemas/agents.d.ts.map +1 -1
- package/dist/schemas/agents.js +0 -8
- package/dist/schemas/agents.js.map +1 -1
- package/dist/schemas/automations.d.ts +0 -5
- package/dist/schemas/automations.d.ts.map +1 -1
- package/dist/schemas/capabilities.d.ts +1 -0
- package/dist/schemas/capabilities.d.ts.map +1 -1
- package/dist/schemas/capabilities.js +1 -0
- package/dist/schemas/capabilities.js.map +1 -1
- package/dist/schemas/providers/usage.d.ts +0 -6
- package/dist/schemas/providers/usage.d.ts.map +1 -1
- package/dist/schemas/providers/usage.js +0 -6
- package/dist/schemas/providers/usage.js.map +1 -1
- package/dist/schemas/settings.d.ts +0 -31
- package/dist/schemas/settings.d.ts.map +1 -1
- package/dist/schemas/settings.js +0 -23
- package/dist/schemas/settings.js.map +1 -1
- package/dist/schemas/workspace/workspace-search.d.ts +1 -0
- package/dist/schemas/workspace/workspace-search.d.ts.map +1 -1
- package/dist/schemas/workspace/workspace-search.js +5 -0
- package/dist/schemas/workspace/workspace-search.js.map +1 -1
- package/dist/state/definition.d.ts +0 -12
- package/dist/state/definition.d.ts.map +1 -1
- package/dist/state/workspace-state.d.ts +6 -0
- package/dist/state/workspace-state.d.ts.map +1 -1
- package/dist/state/workspace-state.js +7 -0
- package/dist/state/workspace-state.js.map +1 -1
- package/dist/text/transcript-fold.d.ts.map +1 -1
- package/dist/text/transcript-fold.js +0 -4
- package/dist/text/transcript-fold.js.map +1 -1
- package/package.json +5 -5
- package/src/events/agent-events.ts +0 -14
- package/src/events/transcript.ts +1 -1
- package/src/index.ts +0 -2
- package/src/models/model-order.test.ts +1 -51
- package/src/models/model-order.ts +1 -23
- package/src/models/model-pins.ts +2 -2
- package/src/schemas/agent.ts +0 -8
- package/src/schemas/agents.ts +0 -11
- package/src/schemas/capabilities.ts +3 -0
- package/src/schemas/model-route.ts +1 -1
- package/src/schemas/providers/usage.ts +0 -11
- package/src/schemas/settings.ts +0 -52
- package/src/schemas/workspace/workspace-search.ts +7 -0
- package/src/state/workspace-state.test.ts +13 -5
- package/src/state/workspace-state.ts +11 -0
- package/src/text/transcript-fold.test.ts +0 -17
- package/src/text/transcript-fold.ts +0 -5
- package/dist/models/fast-tier.d.ts +0 -9
- package/dist/models/fast-tier.d.ts.map +0 -1
- package/dist/models/fast-tier.js +0 -19
- package/dist/models/fast-tier.js.map +0 -1
- package/dist/models/prompt-complexity.d.ts +0 -27
- package/dist/models/prompt-complexity.d.ts.map +0 -1
- package/dist/models/prompt-complexity.js +0 -91
- package/dist/models/prompt-complexity.js.map +0 -1
- package/src/models/fast-tier.test.ts +0 -77
- package/src/models/fast-tier.ts +0 -36
- package/src/models/prompt-complexity.test.ts +0 -177
- package/src/models/prompt-complexity.ts +0 -215
|
@@ -1,177 +0,0 @@
|
|
|
1
|
-
import { expect, test } from "vitest";
|
|
2
|
-
import { type ComplexityInput, FAST_CEILING, judgeComplexity } from "./prompt-complexity.js";
|
|
3
|
-
|
|
4
|
-
// Pins the asymmetry: the judge only ever routes down, so silence resolves to standard, any escalating rule or gate
|
|
5
|
-
// wins regardless of order, and the weights (settings.autoTier "shadow") stay unpinned.
|
|
6
|
-
|
|
7
|
-
const turn = (prompt: string, over: Partial<ComplexityInput> = {}): ComplexityInput => ({
|
|
8
|
-
prompt,
|
|
9
|
-
attachments: 0,
|
|
10
|
-
hasImages: false,
|
|
11
|
-
editorContext: false,
|
|
12
|
-
unattended: false,
|
|
13
|
-
planMode: false,
|
|
14
|
-
afterHardTurn: false,
|
|
15
|
-
...over,
|
|
16
|
-
});
|
|
17
|
-
|
|
18
|
-
const tierOf = (prompt: string, over: Partial<ComplexityInput> = {}) => judgeComplexity(turn(prompt, over)).tier;
|
|
19
|
-
|
|
20
|
-
test("a prompt matching nothing at all stays on the model the user picked", () => {
|
|
21
|
-
const verdict = judgeComplexity(turn(`Have another go at the thing we were discussing yesterday afternoon`));
|
|
22
|
-
|
|
23
|
-
expect(verdict.tier).toBe(`standard`);
|
|
24
|
-
expect(verdict.score).toBeGreaterThan(FAST_CEILING);
|
|
25
|
-
// Only the two absence features fire on this input; together they cannot reach the ceiling.
|
|
26
|
-
expect(verdict.rules).toEqual([`short-prompt`, `no-workspace-reference`]);
|
|
27
|
-
});
|
|
28
|
-
|
|
29
|
-
test("a short vague request is not read as an easy one", () => {
|
|
30
|
-
expect(tierOf(`fix the bug`)).toBe(`standard`);
|
|
31
|
-
expect(tierOf(`have a look at the thing we discussed`)).toBe(`standard`);
|
|
32
|
-
});
|
|
33
|
-
|
|
34
|
-
test("an empty-ish prompt is not mistaken for an easy one", () => {
|
|
35
|
-
expect(tierOf(` `)).toBe(`standard`);
|
|
36
|
-
});
|
|
37
|
-
|
|
38
|
-
test("a short question about nothing in the workspace is the case this feature exists for", () => {
|
|
39
|
-
expect(tierOf(`what is a closure?`)).toBe(`fast`);
|
|
40
|
-
});
|
|
41
|
-
|
|
42
|
-
test("naming a file keeps an otherwise easy question on the user's own model", () => {
|
|
43
|
-
// Same easy word; naming a real file, not a concept, tips the verdict to standard.
|
|
44
|
-
expect(tierOf(`explain what is a closure`)).toBe(`fast`);
|
|
45
|
-
expect(tierOf(`explain what src/agent/turn-plan.ts does`)).toBe(`standard`);
|
|
46
|
-
});
|
|
47
|
-
|
|
48
|
-
test("a trivial aside inside a hard conversation still gets through", () => {
|
|
49
|
-
expect(tierOf(`what is a closure?`, { afterHardTurn: true })).toBe(`fast`);
|
|
50
|
-
});
|
|
51
|
-
|
|
52
|
-
test("a workspace-adjacent errand stops being cheap once the conversation has done hard work", () => {
|
|
53
|
-
expect(tierOf(`list the exports`, { afterHardTurn: false })).toBe(`fast`);
|
|
54
|
-
expect(tierOf(`list the exports`, { afterHardTurn: true })).toBe(`standard`);
|
|
55
|
-
});
|
|
56
|
-
|
|
57
|
-
test("a screenshot is never sent to the cheap rung, however simple the question about it", () => {
|
|
58
|
-
expect(tierOf(`what is this?`, { hasImages: true, attachments: 1 })).toBe(`standard`);
|
|
59
|
-
});
|
|
60
|
-
|
|
61
|
-
test("plan mode is a request to think, so it is never answered by the model that thinks least", () => {
|
|
62
|
-
expect(tierOf(`what is a closure?`, { planMode: true })).toBe(`standard`);
|
|
63
|
-
});
|
|
64
|
-
|
|
65
|
-
test("a surface-started run is never downgraded, because nobody is watching it fail", () => {
|
|
66
|
-
expect(tierOf(`what is a closure?`, { unattended: true })).toBe(`standard`);
|
|
67
|
-
});
|
|
68
|
-
|
|
69
|
-
test("a gate reports itself and scores 1, so the ledger can tell a gate from a hard sentence", () => {
|
|
70
|
-
const verdict = judgeComplexity(turn(`hi`, { unattended: true }));
|
|
71
|
-
|
|
72
|
-
expect(verdict.score).toBe(1);
|
|
73
|
-
expect(verdict.rules).toContain(`unattended`);
|
|
74
|
-
});
|
|
75
|
-
|
|
76
|
-
test.each([
|
|
77
|
-
[`pasted code`, "explain this\n```ts\nconst x = 1;\n```"],
|
|
78
|
-
[`a stack trace`, "it broke\n at Object.run (/work/x.ts:12:3)"],
|
|
79
|
-
[`a thrown error`, "help\nTypeError: cannot read properties of undefined"],
|
|
80
|
-
[`a hard word`, `why does the picker reset`],
|
|
81
|
-
[`another hard word`, `refactor this`],
|
|
82
|
-
[`a second job`, `rename it and then update the tests`],
|
|
83
|
-
[`a checklist`, `- rename it\n- update the tests`],
|
|
84
|
-
[`a cross-cutting scope`, `rename it across the codebase`],
|
|
85
|
-
])("%s forces the user's own model even in an otherwise tiny prompt", (_name, prompt) => {
|
|
86
|
-
expect(tierOf(prompt)).toBe(`standard`);
|
|
87
|
-
});
|
|
88
|
-
|
|
89
|
-
test("an escalating rule beats every easing feature at once, so rule order cannot change an answer", () => {
|
|
90
|
-
// This prompt trips every easing weight in the file, plus one escalating rule; escalation still wins.
|
|
91
|
-
const verdict = judgeComplexity(turn(`what is a race condition?`));
|
|
92
|
-
|
|
93
|
-
expect(verdict.tier).toBe(`standard`);
|
|
94
|
-
expect(verdict.score).toBe(1);
|
|
95
|
-
expect(verdict.rules).toEqual([`hard-words`]);
|
|
96
|
-
});
|
|
97
|
-
|
|
98
|
-
test("a long brief is standard on its length alone, whatever words it happens to use", () => {
|
|
99
|
-
expect(tierOf(`explain `.repeat(400))).toBe(`standard`);
|
|
100
|
-
});
|
|
101
|
-
|
|
102
|
-
test("three files in, the job is about a shape rather than about a file", () => {
|
|
103
|
-
expect(tierOf(`have a look`, { attachments: 3 })).toBe(`standard`);
|
|
104
|
-
expect(tierOf(`have a look`, { attachments: 1 })).toBe(`standard`);
|
|
105
|
-
});
|
|
106
|
-
|
|
107
|
-
test("names every rule that fired, because a score alone cannot say which feature did the work", () => {
|
|
108
|
-
const verdict = judgeComplexity(turn(`what is this?`));
|
|
109
|
-
|
|
110
|
-
expect(verdict.rules).toEqual([`short-prompt`, `easy-words`, `bare-question`, `no-workspace-reference`]);
|
|
111
|
-
});
|
|
112
|
-
|
|
113
|
-
test("scores are rounded, so two turns the same rules judged compare equal on the ledger", () => {
|
|
114
|
-
const score = judgeComplexity(turn(`what is this?`)).score;
|
|
115
|
-
|
|
116
|
-
expect(score).toBe(Number(score.toFixed(3)));
|
|
117
|
-
});
|
|
118
|
-
|
|
119
|
-
test("the score never leaves 0..1, so a stored row is always comparable against the ceiling", () => {
|
|
120
|
-
const floor = judgeComplexity(turn(`what is this?`));
|
|
121
|
-
const ceiling = judgeComplexity(turn(`refactor everything`));
|
|
122
|
-
|
|
123
|
-
expect(floor.score).toBeGreaterThanOrEqual(0);
|
|
124
|
-
expect(ceiling.score).toBeLessThanOrEqual(1);
|
|
125
|
-
});
|
|
126
|
-
|
|
127
|
-
// eagerness moves only the cutoff; a downgrade still needs a positive easy signal at any setting.
|
|
128
|
-
|
|
129
|
-
test("the dial widens what counts as simple, in the direction it says", () => {
|
|
130
|
-
// Eased by its words, held back by naming a real file; balanced keeps it standard, eager lets it through.
|
|
131
|
-
const aboutAFile = `explain what src/app.ts does`;
|
|
132
|
-
|
|
133
|
-
expect(tierOf(aboutAFile, { eagerness: `balanced` })).toBe(`standard`);
|
|
134
|
-
expect(tierOf(aboutAFile, { eagerness: `eager` })).toBe(`fast`);
|
|
135
|
-
});
|
|
136
|
-
|
|
137
|
-
test("the cautious stop wants every easing signal at once, not merely an easy word", () => {
|
|
138
|
-
// "explain closures" fails the cautious stop (not a bare question); only a bare question clears every stop.
|
|
139
|
-
expect(tierOf(`explain closures`, { eagerness: `balanced` })).toBe(`fast`);
|
|
140
|
-
expect(tierOf(`explain closures`, { eagerness: `cautious` })).toBe(`standard`);
|
|
141
|
-
expect(tierOf(`what is a closure?`, { eagerness: `cautious` })).toBe(`fast`);
|
|
142
|
-
});
|
|
143
|
-
|
|
144
|
-
test("an absent dial is the balanced stop, so every row recorded before it existed still compares", () => {
|
|
145
|
-
const bare = judgeComplexity(turn(`what is this?`));
|
|
146
|
-
|
|
147
|
-
expect(bare.ceiling).toBe(FAST_CEILING);
|
|
148
|
-
expect(bare.tier).toBe(judgeComplexity(turn(`what is this?`, { eagerness: `balanced` })).tier);
|
|
149
|
-
});
|
|
150
|
-
|
|
151
|
-
test("no setting of the dial can downgrade a short vague request", () => {
|
|
152
|
-
for (const eagerness of [`cautious`, `balanced`, `eager`] as const) {
|
|
153
|
-
expect(tierOf(`fix the bug`, { eagerness })).toBe(`standard`);
|
|
154
|
-
expect(tierOf(`have a look at the thing we discussed`, { eagerness })).toBe(`standard`);
|
|
155
|
-
}
|
|
156
|
-
});
|
|
157
|
-
|
|
158
|
-
test("the deceptive follow-up is standard at every stop, because it never says anything easy", () => {
|
|
159
|
-
for (const eagerness of [`cautious`, `balanced`, `eager`] as const) {
|
|
160
|
-
expect(tierOf(`now do the same for the other file`, { eagerness, afterHardTurn: true })).toBe(`standard`);
|
|
161
|
-
}
|
|
162
|
-
});
|
|
163
|
-
|
|
164
|
-
test("a turn following hard work has to clear a higher bar, and at the default an eased one no longer does", () => {
|
|
165
|
-
expect(tierOf(`explain closures`)).toBe(`fast`);
|
|
166
|
-
expect(tierOf(`explain closures`, { afterHardTurn: true })).toBe(`standard`);
|
|
167
|
-
});
|
|
168
|
-
|
|
169
|
-
test("the verdict carries the cutoff it was judged against, because a score alone stopped being an answer", () => {
|
|
170
|
-
const cautious = judgeComplexity(turn(`explain closures`, { eagerness: `cautious` }));
|
|
171
|
-
const eager = judgeComplexity(turn(`explain closures`, { eagerness: `eager` }));
|
|
172
|
-
|
|
173
|
-
expect(cautious.ceiling).toBeLessThan(FAST_CEILING);
|
|
174
|
-
expect(eager.ceiling).toBeGreaterThan(FAST_CEILING);
|
|
175
|
-
expect(cautious.score).toBe(eager.score);
|
|
176
|
-
expect([cautious.tier, eager.tier]).toEqual([`standard`, `fast`]);
|
|
177
|
-
});
|
|
@@ -1,215 +0,0 @@
|
|
|
1
|
-
// Judges whether a turn could run on the cheap rung of the provider the user is already on: a pure function over the
|
|
2
|
-
// turn's shape, no model or catalog access. Never fails up: every ambiguous case resolves to standard, since
|
|
3
|
-
// substituting a pricier model would just be honoring the original request.
|
|
4
|
-
|
|
5
|
-
// A verdict lists these, the shadow ledger stores them, a screen renders them: renaming one is a breaking change.
|
|
6
|
-
export type ComplexityRule =
|
|
7
|
-
// Gates: any one ends the question, standard with no score computed.
|
|
8
|
-
| "images"
|
|
9
|
-
| "plan-mode"
|
|
10
|
-
| "unattended"
|
|
11
|
-
// Escalating rules: any one forces standard; order between them cannot matter.
|
|
12
|
-
| "code-block"
|
|
13
|
-
| "stack-trace"
|
|
14
|
-
| "hard-words"
|
|
15
|
-
| "multi-step"
|
|
16
|
-
| "cross-cutting"
|
|
17
|
-
| "long-prompt"
|
|
18
|
-
| "many-attachments"
|
|
19
|
-
// Graded features: these only move the score.
|
|
20
|
-
| "medium-prompt"
|
|
21
|
-
| "attachment"
|
|
22
|
-
| "editor-context"
|
|
23
|
-
| "paths"
|
|
24
|
-
| "many-verbs"
|
|
25
|
-
| "after-hard-turn"
|
|
26
|
-
| "short-prompt"
|
|
27
|
-
| "easy-words"
|
|
28
|
-
| "bare-question"
|
|
29
|
-
| "no-workspace-reference";
|
|
30
|
-
|
|
31
|
-
export type ComplexityTier = "fast" | "standard";
|
|
32
|
-
|
|
33
|
-
// Everything the judge may know: primitive counts, flags and the prompt itself, no shared objects, so the daemon's turn
|
|
34
|
-
// and the composer's draft can each build one independently.
|
|
35
|
-
export interface ComplexityInput {
|
|
36
|
-
readonly prompt: string;
|
|
37
|
-
// Uploaded files plus @-mentioned workspace paths, resolved by the daemon into one count.
|
|
38
|
-
readonly attachments: number;
|
|
39
|
-
// Any attachment the model will read as an image; a cheap rung misreading one is the worst cell in the matrix.
|
|
40
|
-
readonly hasImages: boolean;
|
|
41
|
-
// The opt-in editor chip: the user pointed at a file and a selection, so the turn is about real code.
|
|
42
|
-
readonly editorContext: boolean;
|
|
43
|
-
// A surface started this, not a person at a composer (AgentTurn.unattended).
|
|
44
|
-
readonly unattended: boolean;
|
|
45
|
-
// The turn opens in plan mode: it is being asked to think before it acts.
|
|
46
|
-
readonly planMode: boolean;
|
|
47
|
-
// Whether the prior turn's judgement (not what it ran) was standard; a weight that raises the bar, not a lock.
|
|
48
|
-
readonly afterHardTurn: boolean;
|
|
49
|
-
// settings.autoTierEagerness; absent means balanced, what every pre-knob verdict was judged against.
|
|
50
|
-
readonly eagerness?: TierEagerness;
|
|
51
|
-
}
|
|
52
|
-
|
|
53
|
-
export interface ComplexityVerdict {
|
|
54
|
-
readonly tier: ComplexityTier;
|
|
55
|
-
// 0..1, rounded to three places so ledger rows compare exactly; 1 whenever a gate or escalating rule fired.
|
|
56
|
-
readonly score: number;
|
|
57
|
-
// Every rule that fired, in declaration order; empty means the base score, above the fast ceiling.
|
|
58
|
-
readonly rules: readonly ComplexityRule[];
|
|
59
|
-
// The ceiling this score was judged against, carried out since the ceiling is itself a setting that can change.
|
|
60
|
-
readonly ceiling: number;
|
|
61
|
-
}
|
|
62
|
-
|
|
63
|
-
// Above the fast ceiling on purpose: a prompt matching no rule at all is medium, not simple.
|
|
64
|
-
const BASE_SCORE = 0.5;
|
|
65
|
-
|
|
66
|
-
// How eager the judge is (settings.autoTierEagerness), the one knob this feature exposes:
|
|
67
|
-
// - cautious: floor of 0, downgrades only the clearest easy asks
|
|
68
|
-
// - balanced: the shipped default; every verdict before this knob existed was judged against it
|
|
69
|
-
// - eager: also lets an easy-worded question about real code through
|
|
70
|
-
export const FAST_CEILINGS = { cautious: 0, balanced: 0.25, eager: 0.4 } as const;
|
|
71
|
-
export type TierEagerness = keyof typeof FAST_CEILINGS;
|
|
72
|
-
|
|
73
|
-
// The stop used when nobody chose one: balanced, what every pre-knob verdict was judged against.
|
|
74
|
-
export const FAST_CEILING = FAST_CEILINGS.balanced;
|
|
75
|
-
|
|
76
|
-
// Characters, not tokens: thresholds mark a sentence turning into a brief, or carrying unfenced paste.
|
|
77
|
-
const MEDIUM_PROMPT_CHARS = 600;
|
|
78
|
-
const LONG_PROMPT_CHARS = 2400;
|
|
79
|
-
const SHORT_PROMPT_CHARS = 140;
|
|
80
|
-
// Three files in is a job about a shape rather than about a file, whatever the words say.
|
|
81
|
-
const MANY_ATTACHMENTS = 3;
|
|
82
|
-
// Three or more distinct imperatives is a list of jobs wearing the grammar of one.
|
|
83
|
-
const MANY_VERBS = 3;
|
|
84
|
-
|
|
85
|
-
// Short, boring word lists, case-insensitive on word boundaries; easy only lowers score, hard forces standard.
|
|
86
|
-
const EASY_WORDS =
|
|
87
|
-
/\b(?:what(?:'s| is| are)|explain|describe|summari[sz]e|list|show me|where(?:'s| is| are)|rename|typo|reword|reformat|format this|tidy|define|translate|spell|abbreviat)/i;
|
|
88
|
-
|
|
89
|
-
const HARD_WORDS =
|
|
90
|
-
/\b(?:refactor|redesign|architect|architecture|migrat|root cause|debug|investigat|diagnos|optimi[sz]|race condition|deadlock|memory leak|regression|security|threat model|benchmark|profil|audit|design a|plan (?:a|the|out)|why (?:does|is|are|did|would|can't|cannot))/i;
|
|
91
|
-
|
|
92
|
-
// "Do this, and also that": the strongest cheap signal of a job that is several jobs.
|
|
93
|
-
const MULTI_STEP = /\b(?:and then|after that|once (?:that|you)|followed by|as well as|then also)\b/i;
|
|
94
|
-
|
|
95
|
-
// A job whose subject is the shape of the codebase, not a place in it; the cheap rung is worst at exactly this.
|
|
96
|
-
const CROSS_CUTTING =
|
|
97
|
-
/\b(?:across (?:the|all|every)|every(?: single)? (?:file|module|package|component|usage|call ?site)|all (?:the|of the) (?:files|usages|call ?sites|places)|everywhere|throughout the|codebase-wide|repo-wide)\b/i;
|
|
98
|
-
|
|
99
|
-
// Fenced code, a diff, or an inline patch: proof the turn is real code, where the cheap rung fails silently.
|
|
100
|
-
const CODE_BLOCK = /```|^diff --git |^@@ .* @@|^[+-]{3} [ab]\//m;
|
|
101
|
-
|
|
102
|
-
// A thrown error the user pasted in: the canonical case where the expensive tier earns its price.
|
|
103
|
-
const STACK_TRACE =
|
|
104
|
-
/(?:^|\n)\s*(?:at [\w$.<>]+ \(|Traceback \(most recent call last\)|Caused by:|panic:|thread '.*' panicked|Unhandled|[A-Z]\w*(?:Error|Exception): )/;
|
|
105
|
-
|
|
106
|
-
// A workspace path or @-mention, distinguishing "explain closures" from "explain what this file does".
|
|
107
|
-
const PATH_LIKE =
|
|
108
|
-
/(?:^|\s)(?:@[\w./-]+|[\w-]+\/[\w./-]+\.[a-z]{1,5}\b|\b[\w-]+\.(?:ts|tsx|js|jsx|vue|py|go|rs|java|rb|css|scss|json|ya?ml|md|sql|sh)\b)/i;
|
|
109
|
-
|
|
110
|
-
// One sentence, ending in a question mark, with no second clause: knowledge, not work.
|
|
111
|
-
const BARE_QUESTION = /^[^.!?]{0,200}\?\s*$/;
|
|
112
|
-
|
|
113
|
-
// Imperatives that start a request, counted rather than matched: one is a task, three is a list.
|
|
114
|
-
const VERBS =
|
|
115
|
-
/\b(?:add|remove|delete|fix|write|create|make|update|change|move|rename|split|merge|extract|inline|wire|hook|test|check|run|build|deploy|document|implement|replace|convert|handle|support|expose|log|render|validate|parse|sort|filter|cache)\b/gi;
|
|
116
|
-
|
|
117
|
-
// Bullets and numbered steps, a checklist however casually it is written.
|
|
118
|
-
const LIST_LINES = /^\s*(?:[-*+]\s|\d+[.)]\s)/gm;
|
|
119
|
-
|
|
120
|
-
// Every rule that ends the question on its own; split from the graded features below because these are answers, not
|
|
121
|
-
// evidence.
|
|
122
|
-
const forcing = (input: ComplexityInput, text: string): ComplexityRule[] => {
|
|
123
|
-
const rules: ComplexityRule[] = [];
|
|
124
|
-
// Gates first, cheapest and least arguable: about the turn's situation rather than its words.
|
|
125
|
-
if (input.hasImages) {
|
|
126
|
-
rules.push("images");
|
|
127
|
-
}
|
|
128
|
-
if (input.planMode) {
|
|
129
|
-
rules.push("plan-mode");
|
|
130
|
-
}
|
|
131
|
-
// Unattended turns are billed whole with nobody watching a bad guess.
|
|
132
|
-
if (input.unattended) {
|
|
133
|
-
rules.push("unattended");
|
|
134
|
-
}
|
|
135
|
-
// Then the words: each of these claims the turn is about real code doing something real.
|
|
136
|
-
if (CODE_BLOCK.test(text)) {
|
|
137
|
-
rules.push("code-block");
|
|
138
|
-
}
|
|
139
|
-
if (STACK_TRACE.test(text)) {
|
|
140
|
-
rules.push("stack-trace");
|
|
141
|
-
}
|
|
142
|
-
if (HARD_WORDS.test(text)) {
|
|
143
|
-
rules.push("hard-words");
|
|
144
|
-
}
|
|
145
|
-
if (MULTI_STEP.test(text) || (text.match(LIST_LINES)?.length ?? 0) >= 2) {
|
|
146
|
-
rules.push("multi-step");
|
|
147
|
-
}
|
|
148
|
-
if (CROSS_CUTTING.test(text)) {
|
|
149
|
-
rules.push("cross-cutting");
|
|
150
|
-
}
|
|
151
|
-
if (text.length > LONG_PROMPT_CHARS) {
|
|
152
|
-
rules.push("long-prompt");
|
|
153
|
-
}
|
|
154
|
-
if (input.attachments >= MANY_ATTACHMENTS) {
|
|
155
|
-
rules.push("many-attachments");
|
|
156
|
-
}
|
|
157
|
-
return rules;
|
|
158
|
-
};
|
|
159
|
-
|
|
160
|
-
// Weights are a fitted hypothesis with a ledger under them, not a claim (ships in shadow first). A fast verdict always
|
|
161
|
-
// requires a positive `easing` feature to have fired: absence of complexity is not evidence of simplicity.
|
|
162
|
-
interface GradedFeature {
|
|
163
|
-
readonly rule: ComplexityRule;
|
|
164
|
-
readonly weight: number;
|
|
165
|
-
readonly of: (input: ComplexityInput, text: string) => boolean;
|
|
166
|
-
// A positive reason to think this is easy, not merely an absence of reasons to think it is hard.
|
|
167
|
-
readonly easing?: true;
|
|
168
|
-
}
|
|
169
|
-
|
|
170
|
-
const GRADED: readonly GradedFeature[] = [
|
|
171
|
-
{ rule: "medium-prompt", weight: +0.2, of: (_input, text) => text.length > MEDIUM_PROMPT_CHARS },
|
|
172
|
-
{ rule: "attachment", weight: +0.15, of: (input) => input.attachments > 0 },
|
|
173
|
-
{ rule: "editor-context", weight: +0.1, of: (input) => input.editorContext },
|
|
174
|
-
// Naming a file holds an easy-worded question at standard on balanced; eager is the choice to let it through.
|
|
175
|
-
{ rule: "paths", weight: +0.15, of: (_input, text) => PATH_LIKE.test(text) },
|
|
176
|
-
{
|
|
177
|
-
rule: "many-verbs",
|
|
178
|
-
weight: +0.15,
|
|
179
|
-
of: (_input, text) => new Set((text.match(VERBS) ?? []).map((verb) => verb.toLowerCase())).size >= MANY_VERBS,
|
|
180
|
-
},
|
|
181
|
-
// The heaviest weight: the only feature seeing past the words, heavy enough eager cannot downgrade past it.
|
|
182
|
-
{ rule: "after-hard-turn", weight: +0.25, of: (input) => input.afterHardTurn },
|
|
183
|
-
// Declared in ledger order. `easing` marks the two positive signals; the rest are light absence features.
|
|
184
|
-
{ rule: "short-prompt", weight: -0.1, of: (_input, text) => text.length <= SHORT_PROMPT_CHARS },
|
|
185
|
-
{ rule: "easy-words", weight: -0.25, easing: true, of: (_input, text) => EASY_WORDS.test(text) },
|
|
186
|
-
{ rule: "bare-question", weight: -0.15, easing: true, of: (_input, text) => BARE_QUESTION.test(text) },
|
|
187
|
-
{
|
|
188
|
-
rule: "no-workspace-reference",
|
|
189
|
-
weight: -0.1,
|
|
190
|
-
of: (input, text) => input.attachments === 0 && !input.editorContext && !PATH_LIKE.test(text),
|
|
191
|
-
},
|
|
192
|
-
];
|
|
193
|
-
|
|
194
|
-
// Three places, so a stored score is a value not a float artefact, and same-rule rows compare equal.
|
|
195
|
-
const round3 = (value: number): number => Math.round(value * 1000) / 1000;
|
|
196
|
-
|
|
197
|
-
// Gates and escalators force standard at score 1 (monotone: a new rule only moves turns up a tier). Fast otherwise
|
|
198
|
-
// needs both a sub-ceiling score and a positive `easing` signal; silence alone never downgrades.
|
|
199
|
-
export const judgeComplexity = (input: ComplexityInput): ComplexityVerdict => {
|
|
200
|
-
const ceiling = FAST_CEILINGS[input.eagerness ?? "balanced"];
|
|
201
|
-
const text = input.prompt.trim();
|
|
202
|
-
const forced = forcing(input, text);
|
|
203
|
-
if (forced.length > 0) {
|
|
204
|
-
return { tier: "standard", score: 1, rules: forced, ceiling };
|
|
205
|
-
}
|
|
206
|
-
const hits = GRADED.filter((feature) => feature.of(input, text));
|
|
207
|
-
const score = Math.min(1, Math.max(0, BASE_SCORE + hits.reduce((total, feature) => total + feature.weight, 0)));
|
|
208
|
-
const eased = hits.some((feature) => feature.easing === true);
|
|
209
|
-
return {
|
|
210
|
-
tier: eased && score <= ceiling ? "fast" : "standard",
|
|
211
|
-
score: round3(score),
|
|
212
|
-
rules: hits.map((feature) => feature.rule),
|
|
213
|
-
ceiling,
|
|
214
|
-
};
|
|
215
|
-
};
|