@intentic/sandbox-contract 1.246.1 → 1.247.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/batch-runs.d.ts +2 -0
- package/dist/batch-runs.d.ts.map +1 -1
- package/dist/batch-runs.js +1 -0
- package/dist/batch-runs.js.map +1 -1
- package/dist/contracts/agent.contract.d.ts +19 -0
- package/dist/contracts/agent.contract.d.ts.map +1 -1
- package/dist/contracts/settings.contract.d.ts +97 -10
- package/dist/contracts/settings.contract.d.ts.map +1 -1
- package/dist/contracts/usage.contract.d.ts +22 -0
- package/dist/contracts/usage.contract.d.ts.map +1 -1
- package/dist/contracts/usage.contract.js +19 -0
- package/dist/contracts/usage.contract.js.map +1 -1
- package/dist/definition.d.ts +16 -20
- package/dist/definition.d.ts.map +1 -1
- package/dist/fast-tier.js +1 -1
- package/dist/fast-tier.js.map +1 -1
- package/dist/index.d.ts +140 -12
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +2 -2
- package/dist/index.js.map +1 -1
- package/dist/model-pins.d.ts +17 -0
- package/dist/model-pins.d.ts.map +1 -0
- package/dist/{quick-model.js → model-pins.js} +16 -10
- package/dist/model-pins.js.map +1 -0
- package/dist/model-roles.d.ts +144 -0
- package/dist/model-roles.d.ts.map +1 -0
- package/dist/model-roles.js +129 -0
- package/dist/model-roles.js.map +1 -0
- package/dist/schemas/agent.d.ts +21 -2
- package/dist/schemas/agent.d.ts.map +1 -1
- package/dist/schemas/agent.js +4 -2
- package/dist/schemas/agent.js.map +1 -1
- package/dist/schemas/plan-limits.d.ts +20 -0
- package/dist/schemas/plan-limits.d.ts.map +1 -1
- package/dist/schemas/plan-limits.js +21 -0
- package/dist/schemas/plan-limits.js.map +1 -1
- package/dist/schemas/settings.d.ts +82 -5
- package/dist/schemas/settings.d.ts.map +1 -1
- package/dist/schemas/settings.js +15 -18
- package/dist/schemas/settings.js.map +1 -1
- package/dist/schemas/usage.d.ts +5 -0
- package/dist/schemas/usage.d.ts.map +1 -1
- package/dist/schemas/usage.js +5 -0
- package/dist/schemas/usage.js.map +1 -1
- package/package.json +4 -4
- package/src/agent-catalog.ts +1 -1
- package/src/batch-runs.test.ts +10 -5
- package/src/batch-runs.ts +10 -3
- package/src/contracts/usage.contract.ts +31 -0
- package/src/events.ts +1 -1
- package/src/fast-tier.test.ts +1 -1
- package/src/fast-tier.ts +5 -5
- package/src/index.ts +2 -2
- package/src/{quick-model.test.ts → model-pins.test.ts} +73 -29
- package/src/model-pins.ts +183 -0
- package/src/model-roles.ts +224 -0
- package/src/plan-pools.ts +1 -1
- package/src/prompt-complexity.test.ts +1 -1
- package/src/prompt-complexity.ts +2 -2
- package/src/provider-specs.test.ts +1 -1
- package/src/schemas/agent.ts +42 -17
- package/src/schemas/agents.ts +2 -2
- package/src/schemas/plan-limits.ts +50 -0
- package/src/schemas/settings.ts +77 -88
- package/src/schemas/usage.ts +59 -0
- package/dist/agent-run-model.d.ts +0 -4
- package/dist/agent-run-model.d.ts.map +0 -1
- package/dist/agent-run-model.js +0 -13
- package/dist/agent-run-model.js.map +0 -1
- package/dist/quick-model.d.ts +0 -15
- package/dist/quick-model.d.ts.map +0 -1
- package/dist/quick-model.js.map +0 -1
- package/src/agent-run-model.test.ts +0 -76
- package/src/agent-run-model.ts +0 -65
- package/src/quick-model.ts +0 -162
package/src/schemas/settings.ts
CHANGED
|
@@ -1,7 +1,8 @@
|
|
|
1
1
|
// settings: per-sandbox agent settings (.intentic/config/settings.json)
|
|
2
2
|
import { z } from "zod";
|
|
3
3
|
import { CommandJudgeModeSchema } from "../safety-policy.js";
|
|
4
|
-
import {
|
|
4
|
+
import { ModelRoleSchema } from "../model-roles.js";
|
|
5
|
+
import { AdmissionPolicySchema, AdmissionRuleSchema, ModelPinSchema } from "./agent.js";
|
|
5
6
|
// Which prompt the agent is, before this turn composes anything on top. Two built-in bases and an escape
|
|
6
7
|
// hatch: Intentic's own (the default), Claude Code's preset, or the owner's text. Declared out here rather
|
|
7
8
|
// than inline in the settings object because both sides of the wire branch on it, the daemon to build the
|
|
@@ -262,6 +263,8 @@ export const SkillRemoveSchema = z.object({
|
|
|
262
263
|
// workspaceMap , computes an AREA index of the project a run starts in and prepends it to the
|
|
263
264
|
// conversation's opening message, so the turn does not have to buy its own orientation
|
|
264
265
|
// with a directory listing. Generated from the filesystem every time, never stored.
|
|
266
|
+
// workspaceMapHoldout, conversation-level measurement control for workspaceMap (UsageTurn.mapArm), judged on
|
|
267
|
+
// the directory listings the opening turn ran rather than on its searches.
|
|
265
268
|
// sidecars , the background pass converging a markdown shadow of every binary workspace file
|
|
266
269
|
// (docx/pdf/images/audio → .intentic/local/cache/derived/) the moment it lands, via
|
|
267
270
|
// the baked fileq CLI, so reasoning-time reads are pre-derived. The CLI itself is
|
|
@@ -388,6 +391,21 @@ export const SandboxSettingsSchema = z.object({
|
|
|
388
391
|
.describe(
|
|
389
392
|
"Open every conversation with a map of the project it starts in: what is in it, what each part is for, and where the agent is standing. Worked out fresh each time rather than written down anywhere, because a written layout is wrong within a fortnight. Off by default, since it spends tokens on the first message of every conversation.",
|
|
390
393
|
),
|
|
394
|
+
/* Measurement control for the map, at CONVERSATION level for the plainest of reasons: the note is sent on
|
|
395
|
+
* the opening message, so every later turn of a mapped conversation has a map in its transcript and a
|
|
396
|
+
* per-turn flip would call eleven treated turns controls.
|
|
397
|
+
*
|
|
398
|
+
* Judged on `UsageTurn.openingListings` and read on each conversation's opening turn (usage/turn-experiments.ts),
|
|
399
|
+
* because that is the turn the note was sent to and averaging it across a long conversation divides the
|
|
400
|
+
* effect by the conversation's length. 0 ⇒ no measurement and every conversation receives the map. */
|
|
401
|
+
workspaceMapHoldout: z
|
|
402
|
+
.number()
|
|
403
|
+
.min(0)
|
|
404
|
+
.max(1)
|
|
405
|
+
.default(0)
|
|
406
|
+
.describe(
|
|
407
|
+
"What share of conversations to open without the map, so the two can be compared. Whole conversations rather than individual turns, because the map is sent once and stays in the conversation's history afterwards.",
|
|
408
|
+
),
|
|
391
409
|
/* THE MARKDOWN SHADOWS OF BINARY FILES, the eager half of fileq (_sandbox/fileq). The lazy half — the
|
|
392
410
|
* `fileq` CLI an agent runs mid-task — is always on PATH and gated only by its skill; this switch is
|
|
393
411
|
* about the BACKGROUND pass: the daemon watching /work and converging a sidecar under
|
|
@@ -441,26 +459,39 @@ export const SandboxSettingsSchema = z.object({
|
|
|
441
459
|
.max(1)
|
|
442
460
|
.default(0)
|
|
443
461
|
.describe("What share of commands to leave untrimmed, so the saving can be measured against a real comparison rather than estimated."),
|
|
444
|
-
/*
|
|
445
|
-
*
|
|
446
|
-
*
|
|
447
|
-
*
|
|
448
|
-
*
|
|
449
|
-
*
|
|
450
|
-
*
|
|
451
|
-
*
|
|
452
|
-
*
|
|
453
|
-
*
|
|
454
|
-
*
|
|
455
|
-
*
|
|
456
|
-
*
|
|
457
|
-
*
|
|
458
|
-
|
|
459
|
-
|
|
460
|
-
|
|
461
|
-
|
|
462
|
+
/* WHICH MODEL DOES WHICH JOB, one ordered list per ROLE (model-roles.ts declares them all).
|
|
463
|
+
*
|
|
464
|
+
* ONE KEY RATHER THAN SEVENTEEN, and the record is keyed by the role catalog rather than by free strings: a
|
|
465
|
+
* job that starts choosing a model tomorrow becomes configurable by adding a row to that table, and the
|
|
466
|
+
* settings page, the resolver and the daemon's lookup all follow without a schema change. Seventeen named
|
|
467
|
+
* fields here would be the same table written a fourth time, in the one place where getting it out of step
|
|
468
|
+
* spends somebody's money.
|
|
469
|
+
*
|
|
470
|
+
* IT REPLACED THREE BUNDLED KEYS — `quickModel`, `agentRunModels` and `commandJudgeModels` — and the reason
|
|
471
|
+
* is worth keeping: the first two were grouped by assumed INTENSITY, not by job. "Quick" covered commit
|
|
472
|
+
* messages, session titles and loop verdicts at once, so an owner who wanted better commit subjects could
|
|
473
|
+
* not ask for them without also moving every session title onto the same model; "agent runs" covered a
|
|
474
|
+
* production incident and a documentation sweep with one tier. An intensity is a guess about work its owner
|
|
475
|
+
* knows better, and neither the configuration nor the UI was actually saving anyone anything by making it.
|
|
476
|
+
*
|
|
477
|
+
* EACH LIST IS AN ORDERED LADDER of pins, tried top to bottom, because the interesting failure is a model
|
|
478
|
+
* that is connected and will not answer today: the account's allowance went on the chat, and one spent
|
|
479
|
+
* provider takes that job down for hours while the others sit idle.
|
|
480
|
+
*
|
|
481
|
+
* AN ABSENT OR EMPTY LIST IS THE INTERESTING CASE and means the role's declared floor (resolveRoleModels): a
|
|
482
|
+
* one-shot helper derives an Auto ladder from whatever is connected right now — so it can never name a
|
|
483
|
+
* provider this sandbox has no credential for, and it improves by itself as accounts are added — while a
|
|
484
|
+
* whole session falls to the model the owner picked for their own chat, because nothing here can judge what
|
|
485
|
+
* a session is worth and a wrong guess is billed whole. Storing resolved ids instead would go stale exactly
|
|
486
|
+
* as a pinned model does. */
|
|
487
|
+
// `partialRecord`, not `record`: an exhaustive one would make every role a required key, so a settings file
|
|
488
|
+
// that has never been touched would have to spell out seventeen empty arrays to be valid, and adding a role
|
|
489
|
+
// would invalidate every settings file in existence. An absent key IS the answer "this role has no list".
|
|
490
|
+
modelRoles: z
|
|
491
|
+
.partialRecord(ModelRoleSchema, z.array(ModelPinSchema).max(10))
|
|
492
|
+
.default({})
|
|
462
493
|
.describe(
|
|
463
|
-
"Which models do
|
|
494
|
+
"Which models do which job, one ordered list per job: commit messages, session titles, the safety judge, pipeline fixes, and every other place this sandbox picks a model for you. Tried in order, so one spent account does not take a job down. A job with no list falls back to its own default: cheapest connected for the one-shot helpers, your own chat model for whole sessions.",
|
|
464
495
|
),
|
|
465
496
|
/* WHICH REPOS KEEP A CHANGELOG, the repos whose commits carry a `Release-Note:` trailer, written by the
|
|
466
497
|
* same quick model that drafts the subject (git/commit-message.ts) and harvested at release time.
|
|
@@ -482,40 +513,6 @@ export const SandboxSettingsSchema = z.object({
|
|
|
482
513
|
.describe(
|
|
483
514
|
"Which repositories keep a changelog, and so get a user-facing note written alongside each merge. A list rather than a switch, and empty by default, because the commit writer's standing rule is to copy the house style rather than impose one, and a repository that has never written such a note gives it nothing to copy.",
|
|
484
515
|
),
|
|
485
|
-
/* WHAT AN AGENT RUN OPENS ON, the tier above quickModel, and the answer for every turn a SURFACE starts
|
|
486
|
-
* rather than a person at a composer: Fix with agent on a pipeline or a deployment, a Maintenance chore, a
|
|
487
|
-
* Documentation or Acceptance run, the fix a failed pre-push check proposes. An ORDERED list of PINS, each
|
|
488
|
-
* naming a provider and model AND how that one is to be run (AgentRunPinSchema); EMPTY ⇒ whatever the chat
|
|
489
|
-
* composer would have started with, which is the honest floor because it is the model the user already
|
|
490
|
-
* chose to work with.
|
|
491
|
-
*
|
|
492
|
-
* EACH ENTRY CARRIES ITS OWN REASONING AND COST KNOBS, which is why these are objects rather than the
|
|
493
|
-
* `${provider}:${model}` keys the two lists around them still hold. The effort used to be one field beside
|
|
494
|
-
* the list, answering for every model in it, and the entries of this list are the least interchangeable
|
|
495
|
-
* things on the page: the head is the tier the owner wants the work done at and what follows it is the
|
|
496
|
-
* account that catches it when the first is spent. AgentRunPinSchema has the rest of the argument.
|
|
497
|
-
*
|
|
498
|
-
* A LIST, for the reason quickModel is one: the account at the head runs out, and every surface-started run
|
|
499
|
-
* in the sandbox then fails on a credential the user cannot see from the row they pressed. Written in order,
|
|
500
|
-
* the next one down catches it (turn-resume.ts walks it).
|
|
501
|
-
*
|
|
502
|
-
* PINNED, NOT DERIVED, the deliberate difference from quickModel one line above, and the reason these are
|
|
503
|
-
* two settings rather than one. A quick helper exists to stay OFF the frontier tier, so cheapest-connected
|
|
504
|
-
* is the right automatic answer and an empty list resolves to Auto. An agent run has to read a failing
|
|
505
|
-
* suite, or a container log, or a story, and repair the thing: the tier is a judgement about how much the
|
|
506
|
-
* job is worth, nothing here can make it, and a wrong guess is billed in whole sessions rather than in
|
|
507
|
-
* tokens. So an empty list here resolves to NOTHING and the composer's own pick answers instead.
|
|
508
|
-
*
|
|
509
|
-
* The daemon applies this to any turn flagged `unattended` that names no model of its own, one rule, so a
|
|
510
|
-
* surface added tomorrow inherits it by saying what it is instead of re-deriving where models come from. A
|
|
511
|
-
* surface MAY still name one (the shared run button's caret, Acceptance's per-run pick), and that wins. */
|
|
512
|
-
agentRunModels: z
|
|
513
|
-
.array(AgentRunPinSchema)
|
|
514
|
-
.max(10)
|
|
515
|
-
.default([])
|
|
516
|
-
.describe(
|
|
517
|
-
"Which models run the work a screen starts rather than a person: fixing a red pipeline, a maintenance chore, an acceptance run. Tried in order, so one spent account does not take every such run down, and each entry says how hard that model should think as well as which one it is. Empty falls back to whatever the chat would have used, which is the honest floor because it is the model you already chose to work with.",
|
|
518
|
-
),
|
|
519
516
|
/* AUTOMATIC TIER SELECTION: may the daemon run an easy-looking turn on a cheaper rung of the provider the
|
|
520
517
|
* user is already on, instead of on the model they picked?
|
|
521
518
|
*
|
|
@@ -558,9 +555,9 @@ export const SandboxSettingsSchema = z.object({
|
|
|
558
555
|
"How readily a turn counts as simple enough for the cheaper model. It moves only the cutoff: at every setting a turn still has to say something positively easy, so nothing here can downgrade a short vague request.",
|
|
559
556
|
),
|
|
560
557
|
/* WHICH CHEAP MODEL A DOWNGRADED TURN LANDS ON, an ordered list of `${provider}:${model}` keys
|
|
561
|
-
* (
|
|
558
|
+
* (modelPinKey), or EMPTY for Auto.
|
|
562
559
|
*
|
|
563
|
-
* Empty is the default and the interesting case, exactly as
|
|
560
|
+
* Empty is the default and the interesting case, exactly as a one-shot role's is: Auto is the cheapest row the
|
|
564
561
|
* turn's own provider publishes, read through the same cheap-end order (compareCheapestFirst), so the two
|
|
565
562
|
* features can never disagree about which rung is the cheap one, and connecting an account tomorrow
|
|
566
563
|
* improves the answer by itself.
|
|
@@ -724,35 +721,13 @@ export const SandboxSettingsSchema = z.object({
|
|
|
724
721
|
* entitled to say so, and before this they could not: the old rulebook could be set to allow everything, and
|
|
725
722
|
* the redesign quietly made itself the one part of the sandbox you could only opt further into.
|
|
726
723
|
*
|
|
727
|
-
*
|
|
728
|
-
*
|
|
729
|
-
* commit message
|
|
730
|
-
*
|
|
731
|
-
* should have been. */
|
|
724
|
+
* WHICH MODEL judges is not answered here: it is the `safety-judge` role in `modelRoles`, like every other
|
|
725
|
+
* job in this sandbox that picks one. It used to need its own key, on the argument that a verdict is worth a
|
|
726
|
+
* different model than a commit message — true, and the fact that it had to be argued for one job at a time
|
|
727
|
+
* is exactly what the role catalog replaced. */
|
|
732
728
|
commandJudge: CommandJudgeModeSchema.default("on").describe(
|
|
733
729
|
"Whether a model reads your safety policy before a flagged command runs. Off judges nothing and asks about nothing; Watch judges everything and records it without ever interrupting you, which is how you find out what your policy actually does before you let it stop anything; On lets the verdict decide. Wiping a disk or deleting under /history asks at every setting — that rule is typed rather than judged, and cannot be turned off.",
|
|
734
730
|
),
|
|
735
|
-
/* WHICH MODEL READS THE POLICY, an ordered list of `${provider}:${modelId}` keys (quickModelKey) walked top
|
|
736
|
-
* to bottom, or EMPTY for whatever the quick model would be.
|
|
737
|
-
*
|
|
738
|
-
* A LIST, for the reason every other model setting here is one: the head runs out and the whole feature goes
|
|
739
|
-
* with it. Falling back to the quick chain rather than to Auto is deliberate — it is what this did before the
|
|
740
|
-
* setting existed, so an owner who never opens the row keeps exactly the behaviour they had, and one who
|
|
741
|
-
* writes an entry is saying that command verdicts are worth a different model than commit messages.
|
|
742
|
-
*
|
|
743
|
-
* THE ARGUMENT FOR SETTING IT AT ALL, since the cheapest connected rung is the default: this prompt is the
|
|
744
|
-
* one quick job that is genuinely adversarial. Its input includes text the agent is about to run, which may
|
|
745
|
-
* have arrived from a stranger's web page, and a small model can be talked round by it (command-judge.ts is
|
|
746
|
-
* candid about that). It is also the job where being WRONG is expensive in both directions — a needless card
|
|
747
|
-
* teaches the owner to click through the next one. Neither is a reason for us to spend somebody's frontier
|
|
748
|
-
* allowance by default; both are reasons for them to be able to. */
|
|
749
|
-
commandJudgeModels: z
|
|
750
|
-
.array(z.string())
|
|
751
|
-
.max(10)
|
|
752
|
-
.default([])
|
|
753
|
-
.describe(
|
|
754
|
-
"Which models decide whether a flagged command should run, tried in order so one spent account does not take the gate down. Empty uses whatever the quick model is, which is what this did before the setting existed.",
|
|
755
|
-
),
|
|
756
731
|
/* HOW MUCH AN AGENT MAY DELEGATE, the three ceilings the Claude Code harness enforces on its own Agent
|
|
757
732
|
* tool, surfaced here because their defaults are tuned for a laptop and this is a container the owner sized.
|
|
758
733
|
*
|
|
@@ -836,10 +811,15 @@ export const SavingsArmSchema = z.object({ turns: z.number(), mean: z.number() }
|
|
|
836
811
|
* `metric` says what `mean` counts and what `deltaPct` is a delta in, and choosing it is most of the work.
|
|
837
812
|
* searchCalls , the search teaching: the searches a turn ran, which the teaching directly changes.
|
|
838
813
|
* openingSearches, the same, narrower: the searches before the turn first touched a file.
|
|
839
|
-
*
|
|
840
|
-
*
|
|
814
|
+
* openingListings , the project map: the directory listings a turn ran to work out where it was, which is
|
|
815
|
+
* what the map hands over and what its note tells the turn not to go and fetch.
|
|
816
|
+
* callsBeforeTarget, the map again, on value rather than compliance: how far the turn walked before
|
|
817
|
+
* touching a file it went on to edit.
|
|
818
|
+
* Neither mechanism may be judged on COST. Cost is a whole turn's work, both of these move one part of it, and
|
|
819
|
+
* the part sits inside the noise of the rest. The map's whole payload is about 200 tokens, so a cost reading
|
|
820
|
+
* would be measuring a quantity two orders of magnitude under its own error bar. */
|
|
841
821
|
export const TurnMetricReadingSchema = z.object({
|
|
842
|
-
metric: z.enum(["searchCalls", "openingSearches"]),
|
|
822
|
+
metric: z.enum(["searchCalls", "openingSearches", "openingListings", "callsBeforeTarget"]),
|
|
843
823
|
on: SavingsArmSchema,
|
|
844
824
|
off: SavingsArmSchema,
|
|
845
825
|
/* HOW MUCH LONGER, when the margin spans zero and the honest answer is "keep collecting": the additional
|
|
@@ -896,9 +876,14 @@ export const TurnExperimentSchema = z.object({
|
|
|
896
876
|
// counts toward the daemon's real threshold instead of a number the browser guessed. Shared by every
|
|
897
877
|
// reading: they are the same turns counted differently, so they clear it together.
|
|
898
878
|
minTurns: z.number(),
|
|
899
|
-
|
|
900
|
-
|
|
901
|
-
|
|
879
|
+
/* The randomized unit behind the arm counts. Turn mechanisms default to turns. Teaching loaded into a
|
|
880
|
+
* provider session randomizes and analyzes whole conversations, so repeated turns are not false replicas.
|
|
881
|
+
*
|
|
882
|
+
* "opening turns" is the third case and belongs to a treatment sent ONCE: the project map rides the first
|
|
883
|
+
* message and nothing after it, so the sample is one turn per conversation rather than an average over
|
|
884
|
+
* the conversation's turns. The distinction is not cosmetic. Averaging a first-turn treatment across a
|
|
885
|
+
* twelve-turn conversation divides its effect by twelve and reports the remainder as noise. */
|
|
886
|
+
sampleUnit: z.enum(["turns", "conversations", "opening turns"]).optional(),
|
|
902
887
|
// Content-addressed treatment version. Present where mixing rows from two instruction revisions would turn
|
|
903
888
|
// one experiment into two unnamed ones; the reader filters to this (latest) cohort.
|
|
904
889
|
cohort: z.string().optional(),
|
|
@@ -942,6 +927,10 @@ export type TierReport = z.infer<typeof TierReportSchema>;
|
|
|
942
927
|
export const SavingsReportSchema = z.object({
|
|
943
928
|
input: InputSavingsSchema,
|
|
944
929
|
search: TurnExperimentSchema.optional(),
|
|
930
|
+
/* The project map's arms, read on opening turns (see TurnExperimentSchema.sampleUnit). Absent under the
|
|
931
|
+
* same rule as `search`: the switch is off, or no holdout is set, and a section that is not there reads as
|
|
932
|
+
* "not measured", which is the truth, where zeros would read as "measured, worth nothing". */
|
|
933
|
+
map: TurnExperimentSchema.optional(),
|
|
945
934
|
// Automatic tier selection's readout, see TierReportSchema. Absent ⇒ nothing was judged in the window.
|
|
946
935
|
tier: TierReportSchema.optional(),
|
|
947
936
|
});
|
package/src/schemas/usage.ts
CHANGED
|
@@ -119,6 +119,65 @@ export const UsageTurnSchema = z.object({
|
|
|
119
119
|
*
|
|
120
120
|
* Absent ⇒ as for `searchCalls`. */
|
|
121
121
|
openingSearches: z.number().optional(),
|
|
122
|
+
/* THE LISTINGS THIS TURN RAN TO WORK OUT WHERE IT WAS, `ls`, `ls /work`, `tree intentic`, the LS tool
|
|
123
|
+
* aimed at the same places (isRootListing owns the rule). What the project map is judged on, and the
|
|
124
|
+
* reason it could not be judged before.
|
|
125
|
+
*
|
|
126
|
+
* `searchCalls` counts a directory listing and a ripgrep as one event, deliberately, so that a taxonomy
|
|
127
|
+
* cannot report whichever spelling of a search the model happened to reach for. That is right for the
|
|
128
|
+
* search teaching and blind to the map: measured over 468 mapped sessions of this workspace against 497
|
|
129
|
+
* unmapped ones, searches before the first file moved +7.6% with a margin of ±17.6pp, while the share of
|
|
130
|
+
* sessions opening with a directory listing fell from 46.3% to 32.1%. The map does not shorten the
|
|
131
|
+
* orientation burst, it changes what the burst is made of, and only this counts the difference.
|
|
132
|
+
*
|
|
133
|
+
* UP TO THE FIRST FILE, exactly like `openingSearches`, which took the corpus to settle. Counted over the
|
|
134
|
+
* whole turn instead, the same sessions give 66.4% against 73.3%, a gap of 6.9pp where the orientation
|
|
135
|
+
* window gives 14.2pp. The dilution is not noise: a turn already at work lists the directory it has
|
|
136
|
+
* narrowed to, and no map could have answered that. The shape of the listing cannot tell the two apart,
|
|
137
|
+
* because it is the same shape; only when it happened can.
|
|
138
|
+
*
|
|
139
|
+
* Absent ⇒ the turn predates this being measured, never a turn that listed nothing. */
|
|
140
|
+
openingListings: z.number().optional(),
|
|
141
|
+
/* TOOL CALLS BEFORE THE TURN FIRST TOUCHED A FILE IT WENT ON TO EDIT: how far it walked before reaching
|
|
142
|
+
* the thing it turned out to be looking for. The corpus study this map was designed from measured the
|
|
143
|
+
* same quantity by hand (the workspace's docs/agent-exploration-patterns.md §4: median 4, mean 8.2) and called it the one
|
|
144
|
+
* honest reading of whether orientation got better.
|
|
145
|
+
*
|
|
146
|
+
* IT CAN ONLY BE KNOWN AT THE END, which is why it is recorded here and computed nowhere else: whether a
|
|
147
|
+
* file was the target is a fact about the turn's edits, and the turn has to finish before that is
|
|
148
|
+
* decided. The route keeps the first call index per path and intersects it with the edit ledger.
|
|
149
|
+
*
|
|
150
|
+
* Absent on a turn that edited nothing, which is most short turns, and NOT zero: a turn with no target
|
|
151
|
+
* never reached one, and averaging that in as "reached it immediately" would report the turns that did
|
|
152
|
+
* no work as the best targeted. That absence costs the reading three quarters of the population, so it
|
|
153
|
+
* is a metric to accumulate for months rather than a gate to wait on. */
|
|
154
|
+
callsBeforeTarget: z.number().optional(),
|
|
155
|
+
/* WHICH ARM OF THE PROJECT MAP EXPERIMENT this conversation ran (settings.workspaceMapHoldout), and how
|
|
156
|
+
* many characters the note actually cost when it was sent.
|
|
157
|
+
*
|
|
158
|
+
* CONVERSATION-STABLE, and for a plainer reason than the search teaching's: the map is sent once, on the
|
|
159
|
+
* opening message, so every later turn of a mapped conversation is a turn whose transcript holds a map.
|
|
160
|
+
* A per-turn flip would label eleven treated turns as controls.
|
|
161
|
+
*
|
|
162
|
+
* `mapChars` rather than a token estimate, because characters are what the renderer's budget is in
|
|
163
|
+
* (workspace-map.ts caps the note at 2,800) and a tokenizer's answer would vary by model. Present only on
|
|
164
|
+
* the turn that actually sent one, so the ledger says what the feature costs instead of assuming it: the
|
|
165
|
+
* median note over this workspace's own corpus is 795 characters against a ceiling of 2,800.
|
|
166
|
+
*
|
|
167
|
+
* Absent ⇒ no measurement (the holdout is zero, or the row predates it); true/false ⇒ mapped/unmapped. */
|
|
168
|
+
mapArm: z.boolean().optional(),
|
|
169
|
+
mapChars: z.number().optional(),
|
|
170
|
+
/* WHICH TURN OF ITS CONVERSATION THIS WAS, counting from zero, so an opening turn can be recognised from
|
|
171
|
+
* one row instead of inferred from the rows around it.
|
|
172
|
+
*
|
|
173
|
+
* The inference it replaces is wrong at exactly one place and it is the place that matters: a reader
|
|
174
|
+
* windowed to the last seven days would take each conversation's earliest row IN THE WINDOW as its
|
|
175
|
+
* opening turn, so every conversation that started the week before would contribute a mid-conversation
|
|
176
|
+
* turn to a reading about first turns. The project map is sent on turn zero and nowhere else, so that
|
|
177
|
+
* misreading is not an edge case for it, it is the measurement.
|
|
178
|
+
*
|
|
179
|
+
* Absent ⇒ the row predates this, or the turn belonged to no conversation at all. */
|
|
180
|
+
turnIndex: z.number().optional(),
|
|
122
181
|
/* DID THIS TURN FINISH, OR DID IT STOP TALKING. The fields that tell the two apart, and the reason
|
|
123
182
|
* `outcome` alone could never.
|
|
124
183
|
*
|
|
@@ -1,4 +0,0 @@
|
|
|
1
|
-
import { type QuickModelSource } from "./quick-model.js";
|
|
2
|
-
import type { AgentRunPin } from "./schemas/agent.js";
|
|
3
|
-
export declare const resolveAgentRunModels: (sources: readonly QuickModelSource[], pinned: readonly AgentRunPin[]) => readonly AgentRunPin[];
|
|
4
|
-
//# sourceMappingURL=agent-run-model.d.ts.map
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"file":"agent-run-model.d.ts","sourceRoot":"","sources":["../src/agent-run-model.ts"],"names":[],"mappings":"AAAA,OAAO,EAAiB,KAAK,gBAAgB,EAAE,MAAM,kBAAkB,CAAC;AACxE,OAAO,KAAK,EAAE,WAAW,EAAE,MAAM,oBAAoB,CAAC;AA2CtD,eAAO,MAAM,qBAAqB,YAAa,SAAS,gBAAgB,EAAE,UAAU,SAAS,WAAW,EAAE,KAAG,SAAS,WAAW,EAoBhI,CAAC"}
|
package/dist/agent-run-model.js
DELETED
|
@@ -1,13 +0,0 @@
|
|
|
1
|
-
import { quickModelKey } from "./quick-model.js";
|
|
2
|
-
export const resolveAgentRunModels = (sources, pinned) => {
|
|
3
|
-
const ready = new Set(sources.filter((source) => source.ready).map((source) => source.provider));
|
|
4
|
-
const requested = pinned.filter((pin) => ready.has(pin.provider));
|
|
5
|
-
const chain = [];
|
|
6
|
-
for (const pin of requested) {
|
|
7
|
-
if (!chain.some((held) => quickModelKey(held) === quickModelKey(pin))) {
|
|
8
|
-
chain.push(pin);
|
|
9
|
-
}
|
|
10
|
-
}
|
|
11
|
-
return chain;
|
|
12
|
-
};
|
|
13
|
-
//# sourceMappingURL=agent-run-model.js.map
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"file":"agent-run-model.js","sourceRoot":"","sources":["../src/agent-run-model.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,aAAa,EAAyB,MAAM,kBAAkB,CAAC;AA4CxE,MAAM,CAAC,MAAM,qBAAqB,GAAG,CAAC,OAAoC,EAAE,MAA8B,EAA0B,EAAE;IAClI,MAAM,KAAK,GAAG,IAAI,GAAG,CAAC,OAAO,CAAC,MAAM,CAAC,CAAC,MAAM,EAAE,EAAE,CAAC,MAAM,CAAC,KAAK,CAAC,CAAC,GAAG,CAAC,CAAC,MAAM,EAAE,EAAE,CAAC,MAAM,CAAC,QAAQ,CAAC,CAAC,CAAC;IAIjG,MAAM,SAAS,GAAG,MAAM,CAAC,MAAM,CAAC,CAAC,GAAG,EAAE,EAAE,CAAC,KAAK,CAAC,GAAG,CAAC,GAAG,CAAC,QAAQ,CAAC,CAAC,CAAC;IAQlE,MAAM,KAAK,GAAkB,EAAE,CAAC;IAChC,KAAK,MAAM,GAAG,IAAI,SAAS,EAAE,CAAC;QAC1B,IAAI,CAAC,KAAK,CAAC,IAAI,CAAC,CAAC,IAAI,EAAE,EAAE,CAAC,aAAa,CAAC,IAAI,CAAC,KAAK,aAAa,CAAC,GAAG,CAAC,CAAC,EAAE,CAAC;YACpE,KAAK,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC;QACpB,CAAC;IACL,CAAC;IACD,OAAO,KAAK,CAAC;AACjB,CAAC,CAAC"}
|
package/dist/quick-model.d.ts
DELETED
|
@@ -1,15 +0,0 @@
|
|
|
1
|
-
import type { AgentProvider } from "./schemas/agent.js";
|
|
2
|
-
export interface QuickModelSource {
|
|
3
|
-
readonly provider: AgentProvider;
|
|
4
|
-
readonly ready: boolean;
|
|
5
|
-
readonly models: readonly string[];
|
|
6
|
-
}
|
|
7
|
-
export interface QuickModelChoice {
|
|
8
|
-
readonly provider: AgentProvider;
|
|
9
|
-
readonly model: string;
|
|
10
|
-
}
|
|
11
|
-
export declare const quickModelKey: (choice: QuickModelChoice) => string;
|
|
12
|
-
export declare const parsePinned: (pinned: string) => QuickModelChoice | undefined;
|
|
13
|
-
export declare const pinnedModelLabel: (choice: QuickModelChoice) => string;
|
|
14
|
-
export declare const resolveQuickModels: (sources: readonly QuickModelSource[], pinned: readonly string[]) => readonly QuickModelChoice[];
|
|
15
|
-
//# sourceMappingURL=quick-model.d.ts.map
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"file":"quick-model.d.ts","sourceRoot":"","sources":["../src/quick-model.ts"],"names":[],"mappings":"AAGA,OAAO,KAAK,EAAE,aAAa,EAAE,MAAM,oBAAoB,CAAC;AA+BxD,MAAM,WAAW,gBAAgB;IAM7B,QAAQ,CAAC,QAAQ,EAAE,aAAa,CAAC;IAGjC,QAAQ,CAAC,KAAK,EAAE,OAAO,CAAC;IACxB,QAAQ,CAAC,MAAM,EAAE,SAAS,MAAM,EAAE,CAAC;CACtC;AAED,MAAM,WAAW,gBAAgB;IAC7B,QAAQ,CAAC,QAAQ,EAAE,aAAa,CAAC;IACjC,QAAQ,CAAC,KAAK,EAAE,MAAM,CAAC;CAC1B;AAID,eAAO,MAAM,aAAa,WAAY,gBAAgB,KAAG,MAA8C,CAAC;AASxG,eAAO,MAAM,WAAW,WAAY,MAAM,KAAG,gBAAgB,GAAG,SAM/D,CAAC;AAOF,eAAO,MAAM,gBAAgB,WAAY,gBAAgB,KAAG,MACyC,CAAC;AAuEtG,eAAO,MAAM,kBAAkB,YAAa,SAAS,gBAAgB,EAAE,UAAU,SAAS,MAAM,EAAE,KAAG,SAAS,gBAAgB,EAa7H,CAAC"}
|
package/dist/quick-model.js.map
DELETED
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"file":"quick-model.js","sourceRoot":"","sources":["../src/quick-model.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,SAAS,EAAE,SAAS,EAAE,SAAS,EAAE,MAAM,oBAAoB,CAAC;AACrE,OAAO,EAAE,WAAW,EAAE,MAAM,qBAAqB,CAAC;AAClD,OAAO,EAAE,oBAAoB,EAAE,QAAQ,EAAE,UAAU,EAAE,MAAM,kBAAkB,CAAC;AAoD9E,MAAM,CAAC,MAAM,aAAa,GAAG,CAAC,MAAwB,EAAU,EAAE,CAAC,GAAG,MAAM,CAAC,QAAQ,IAAI,MAAM,CAAC,KAAK,EAAE,CAAC;AASxG,MAAM,CAAC,MAAM,WAAW,GAAG,CAAC,MAAc,EAAgC,EAAE;IACxE,MAAM,SAAS,GAAG,MAAM,CAAC,OAAO,CAAC,GAAG,CAAC,CAAC;IACtC,IAAI,SAAS,IAAI,CAAC,IAAI,SAAS,KAAK,MAAM,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;QACpD,OAAO,SAAS,CAAC;IACrB,CAAC;IACD,OAAO,EAAE,QAAQ,EAAE,MAAM,CAAC,KAAK,CAAC,CAAC,EAAE,SAAS,CAAC,EAAE,KAAK,EAAE,MAAM,CAAC,KAAK,CAAC,SAAS,GAAG,CAAC,CAAC,EAAE,CAAC;AACxF,CAAC,CAAC;AAOF,MAAM,CAAC,MAAM,gBAAgB,GAAG,CAAC,MAAwB,EAAU,EAAE,CACjE,SAAS,CAAC,MAAM,CAAC,QAAQ,CAAC,CAAC,IAAI,CAAC,CAAC,MAAM,EAAE,EAAE,CAAC,MAAM,CAAC,KAAK,KAAK,MAAM,CAAC,KAAK,CAAC,EAAE,KAAK,IAAI,MAAM,CAAC,KAAK,CAAC;AAKtG,MAAM,UAAU,GAAG,CAAC,MAAwB,EAAsB,EAAE,CAAC,MAAM,CAAC,MAAM,CAAC,QAAQ,CAAC,oBAAoB,CAAC,CAAC,CAAC,CAAC,CAAC;AAKrH,MAAM,MAAM,GAAG,CAAC,KAAa,EAAU,EAAE,CAAC,UAAU,CAAC,QAAQ,CAAC,KAAK,CAAC,CAAC,CAAC;AAKtE,MAAM,aAAa,GAAG,CAAC,QAAuB,EAAU,EAAE,CAAC,SAAS,CAAC,SAAS,CAAC,CAAC,KAAK,EAAE,EAAE,CAAC,KAAK,CAAC,KAAK,KAAK,QAAQ,CAAC,CAAC;AAQpH,MAAM,YAAY,GAAG,IAAI,CAAC,GAAG,CAAC,GAAG,MAAM,CAAC,MAAM,CAAC,WAAW,CAAC,CAAC,GAAG,CAAC,CAAC;AAIjE,MAAM,MAAM,GAAG,CAAC,QAAuB,EAAU,EAAE;IAC/C,MAAM,MAAM,GAAG,SAAS,CAAC,QAAQ,CAAC,CAAC;IACnC,OAAO,MAAM,KAAK,SAAS,CAAC,CAAC,CAAC,YAAY,CAAC,CAAC,CAAC,WAAW,CAAC,MAAM,CAAC,IAAI,CAAC,CAAC;AAC1E,CAAC,CAAC;AAYF,MAAM,UAAU,GAAG,CAAC,OAAoC,EAA+B,EAAE,CACrF,OAAO;KACF,MAAM,CAAC,CAAC,MAAM,EAAE,EAAE,CAAC,MAAM,CAAC,KAAK,CAAC;KAChC,OAAO,CAAC,CAAC,MAAM,EAAE,EAAE;IAChB,MAAM,KAAK,GAAG,UAAU,CAAC,MAAM,CAAC,CAAC;IACjC,OAAO,KAAK,KAAK,SAAS,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,CAAC,EAAE,QAAQ,EAAE,MAAM,CAAC,QAAQ,EAAE,KAAK,EAAE,CAAC,CAAC;AAC7E,CAAC,CAAC;KACD,QAAQ,CACL,CAAC,IAAI,EAAE,KAAK,EAAE,EAAE,CACZ,MAAM,CAAC,KAAK,CAAC,KAAK,CAAC,GAAG,MAAM,CAAC,IAAI,CAAC,KAAK,CAAC;IACxC,MAAM,CAAC,IAAI,CAAC,QAAQ,CAAC,GAAG,MAAM,CAAC,KAAK,CAAC,QAAQ,CAAC;IAC9C,aAAa,CAAC,IAAI,CAAC,QAAQ,CAAC,GAAG,aAAa,CAAC,KAAK,CAAC,QAAQ,CAAC,CACnE,CAAC;AAiBV,MAAM,CAAC,MAAM,kBAAkB,GAAG,CAAC,OAAoC,EAAE,MAAyB,EAA+B,EAAE;IAC/H,MAAM,KAAK,GAAG,IAAI,GAAG,CAAC,OAAO,CAAC,MAAM,CAAC,CAAC,MAAM,EAAE,EAAE,CAAC,MAAM,CAAC,KAAK,CAAC,CAAC,GAAG,CAAC,CAAC,MAAM,EAAE,EAAE,CAAC,MAAM,CAAC,QAAQ,CAAC,CAAC,CAAC;IACjG,MAAM,SAAS,GAAG,MAAM,CAAC,OAAO,CAAC,CAAC,GAAG,EAAE,EAAE;QAIrC,MAAM,MAAM,GAAG,WAAW,CAAC,GAAG,CAAC,CAAC;QAChC,OAAO,MAAM,KAAK,SAAS,IAAI,CAAC,KAAK,CAAC,GAAG,CAAC,MAAM,CAAC,QAAQ,CAAC,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,CAAC,MAAM,CAAC,CAAC;IAC/E,CAAC,CAAC,CAAC;IAGH,MAAM,KAAK,GAAG,CAAC,GAAG,IAAI,GAAG,CAAC,SAAS,CAAC,GAAG,CAAC,CAAC,MAAM,EAAE,EAAE,CAAC,CAAC,aAAa,CAAC,MAAM,CAAC,EAAE,MAAM,CAAC,CAAC,CAAC,CAAC,MAAM,EAAE,CAAC,CAAC;IAChG,OAAO,KAAK,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC,CAAC,KAAK,CAAC,CAAC,CAAC,UAAU,CAAC,OAAO,CAAC,CAAC;AAC1D,CAAC,CAAC"}
|
|
@@ -1,76 +0,0 @@
|
|
|
1
|
-
import { expect, test } from "vitest";
|
|
2
|
-
import { resolveAgentRunModels } from "./agent-run-model.js";
|
|
3
|
-
import type { QuickModelSource } from "./quick-model.js";
|
|
4
|
-
import type { AgentRunPin } from "./schemas/agent.js";
|
|
5
|
-
|
|
6
|
-
/* Which model a run somebody's BUTTON started opens on. The rule answers the same two surfaces its quick-model
|
|
7
|
-
* sibling does: the daemon walks it, the settings row names it, so what these pin is the pair of properties
|
|
8
|
-
* that separate the two: an account this sandbox cannot reach never sits at the head of the chain, and an empty
|
|
9
|
-
* answer stays empty rather than being filled in with a tier nobody chose.
|
|
10
|
-
*
|
|
11
|
-
* And one property neither of those covers, new with the pins being objects: an entry's own knobs are the
|
|
12
|
-
* entry's, so what survives the walk is the WHOLE pin. A resolver that handed back the pair inside it would run
|
|
13
|
-
* the fallback at the head's effort, which is a tier that appears nowhere on the user's screen. */
|
|
14
|
-
|
|
15
|
-
const CLAUDE: QuickModelSource = { provider: `claude`, ready: true, models: [`claude-opus-5`, `claude-sonnet-5`] };
|
|
16
|
-
const CODEX: QuickModelSource = { provider: `codex`, ready: true, models: [`gpt-5.6`] };
|
|
17
|
-
const GOOGLE: QuickModelSource = { provider: `gemini`, ready: true, models: [`gemini-3-pro`] };
|
|
18
|
-
|
|
19
|
-
const offline = (source: QuickModelSource): QuickModelSource => ({ ...source, ready: false });
|
|
20
|
-
|
|
21
|
-
const pin = (provider: string, model: string, knobs: Partial<AgentRunPin> = {}): AgentRunPin => ({ provider, model, ...knobs });
|
|
22
|
-
|
|
23
|
-
test("keeps the user's own order, this list is read, never ranked", () => {
|
|
24
|
-
// The opposite of the quick chain, which sorts by tier and cost. Here the order IS the setting: someone who
|
|
25
|
-
// put Opus above GPT wants Opus first, and a resolver that knew better would spend the wrong account.
|
|
26
|
-
expect(resolveAgentRunModels([CLAUDE, CODEX], [pin(`codex`, `gpt-5.6`), pin(`claude`, `claude-opus-5`)])).toEqual([
|
|
27
|
-
{ provider: `codex`, model: `gpt-5.6` },
|
|
28
|
-
{ provider: `claude`, model: `claude-opus-5` },
|
|
29
|
-
]);
|
|
30
|
-
});
|
|
31
|
-
|
|
32
|
-
test("steps over a provider this sandbox has no credential for", () => {
|
|
33
|
-
// The whole reason the setting is a list. With Claude disconnected the head would otherwise be an account
|
|
34
|
-
// that fails every Fix with agent, while a perfectly good Codex sits underneath it.
|
|
35
|
-
expect(resolveAgentRunModels([offline(CLAUDE), CODEX], [pin(`claude`, `claude-opus-5`), pin(`codex`, `gpt-5.6`)])).toEqual([
|
|
36
|
-
{ provider: `codex`, model: `gpt-5.6` },
|
|
37
|
-
]);
|
|
38
|
-
});
|
|
39
|
-
|
|
40
|
-
test("each surviving entry keeps its own knobs, not the head's", () => {
|
|
41
|
-
// The effort used to be one setting beside the list, so a fallback ran at whatever the head was set to. It
|
|
42
|
-
// is now a property of the entry that actually answers, which is the only place it was ever true.
|
|
43
|
-
expect(
|
|
44
|
-
resolveAgentRunModels(
|
|
45
|
-
[offline(CODEX), CLAUDE],
|
|
46
|
-
[pin(`codex`, `gpt-5.6`, { effort: `low` }), pin(`claude`, `claude-opus-5`, { effort: `max`, thinking: true })],
|
|
47
|
-
),
|
|
48
|
-
).toEqual([{ provider: `claude`, model: `claude-opus-5`, effort: `max`, thinking: true }]);
|
|
49
|
-
});
|
|
50
|
-
|
|
51
|
-
test("resolves to nothing when no pin is reachable: it does NOT fall back to whatever is connected", () => {
|
|
52
|
-
// The deliberate difference from resolveQuickModels, which lands on its Auto ladder here. An agent run is
|
|
53
|
-
// billed in whole sessions, so an unreachable list hands the choice back to the caller's floor (the user's
|
|
54
|
-
// own composer pick) rather than spending an account they never pointed at.
|
|
55
|
-
expect(resolveAgentRunModels([offline(CLAUDE), GOOGLE], [pin(`claude`, `claude-opus-5`)])).toEqual([]);
|
|
56
|
-
});
|
|
57
|
-
|
|
58
|
-
test("an empty list resolves to nothing even with accounts connected", () => {
|
|
59
|
-
expect(resolveAgentRunModels([CLAUDE, CODEX, GOOGLE], [])).toEqual([]);
|
|
60
|
-
});
|
|
61
|
-
|
|
62
|
-
test("drops a duplicate rather than spending two attempts proving one account is out, and the first one's knobs are the ones kept", () => {
|
|
63
|
-
// Two entries can now name one model and disagree about how hard it thinks, which is what reordering a list
|
|
64
|
-
// by hand produces. The one the user reads first is the one they meant.
|
|
65
|
-
expect(
|
|
66
|
-
resolveAgentRunModels([CLAUDE], [pin(`claude`, `claude-opus-5`, { effort: `max` }), pin(`claude`, `claude-opus-5`, { effort: `low` })]),
|
|
67
|
-
).toEqual([{ provider: `claude`, model: `claude-opus-5`, effort: `max` }]);
|
|
68
|
-
});
|
|
69
|
-
|
|
70
|
-
test("carries a model id the static catalog has never heard of", () => {
|
|
71
|
-
// The picker offers a custom-id escape hatch, so a pin can name a model released after this build. Second-
|
|
72
|
-
// guessing it here would quietly run something other than what the settings row says.
|
|
73
|
-
expect(resolveAgentRunModels([CLAUDE], [pin(`claude`, `claude-opus-9-preview`)])).toEqual([
|
|
74
|
-
{ provider: `claude`, model: `claude-opus-9-preview` },
|
|
75
|
-
]);
|
|
76
|
-
});
|
package/src/agent-run-model.ts
DELETED
|
@@ -1,65 +0,0 @@
|
|
|
1
|
-
import { quickModelKey, type QuickModelSource } from "./quick-model.js";
|
|
2
|
-
import type { AgentRunPin } from "./schemas/agent.js";
|
|
3
|
-
|
|
4
|
-
/* WHAT A SURFACE-STARTED AGENT RUN OPENS ON, the resolver for `agentRunModels`, sibling to resolveQuickModels
|
|
5
|
-
* and deliberately not the same function.
|
|
6
|
-
*
|
|
7
|
-
* BOTH ARE ORDERED LISTS, FOR THE SAME REASON. One connected account whose allowance went on the chat this
|
|
8
|
-
* morning is enough to take every one of these down: the user presses Fix with agent on a red pipeline, an
|
|
9
|
-
* isolated session opens, and it dies on a credential error they cannot see from the row. Written in order, the
|
|
10
|
-
* next entry catches it.
|
|
11
|
-
*
|
|
12
|
-
* THEY DIFFER ON WHAT AN EMPTY LIST MEANS, and that difference is the whole reason this is its own file rather
|
|
13
|
-
* than a flag on the other one. A quick helper exists to stay OFF the frontier tier, so "work it out from what
|
|
14
|
-
* is connected" is a good answer and quickModel's empty list resolves to a derived Auto ladder. An agent run is
|
|
15
|
-
* a full session with a worktree, billed whole: nothing here can judge whether a job is worth the frontier tier,
|
|
16
|
-
* so an empty list resolves to NOTHING and the caller falls back to the model the user picked for their own
|
|
17
|
-
* chat, a choice they made, rather than one this file guessed for them. For the same reason there is no Auto
|
|
18
|
-
* ladder underneath a list that has been emptied by disconnection: it would spend an account the user never
|
|
19
|
-
* pointed at, on the most expensive kind of run this app starts.
|
|
20
|
-
*
|
|
21
|
-
* WHAT "STEPPED OVER" MEANS HERE IS NARROWER than the quick chain's, and worth being exact about. The quick
|
|
22
|
-
* chain re-asks the next rung when a call comes back refused, because a one-shot that failed has cost nothing
|
|
23
|
-
* and can simply be run again. An agent session cannot be replayed that way, by the time a provider refuses
|
|
24
|
-
* mid-turn the agent may have already edited files, so this list is read ONCE, at the moment the turn is
|
|
25
|
-
* composed, and steps over exactly one thing: an account that is not connected. A model that accepts the turn
|
|
26
|
-
* and fails later is a failed run the user reads on the card, like any other. */
|
|
27
|
-
|
|
28
|
-
/* The pins that could actually be started right now, in the user's own order.
|
|
29
|
-
*
|
|
30
|
-
* `sources` is the same readiness view resolveQuickModels takes, so both settings rows agree about which
|
|
31
|
-
* accounts this sandbox can send to, a pin greyed as "Not connected" in one row and silently spent by the
|
|
32
|
-
* other would be the worst of both.
|
|
33
|
-
*
|
|
34
|
-
* A pin whose provider is gone is DROPPED rather than held: it would otherwise sit at the head of the chain
|
|
35
|
-
* failing every run, which is exactly what the list exists to prevent. It stays on SCREEN, greyed, the settings
|
|
36
|
-
* row renders the stored list, not this one, because a setting that vanished from view would look like the app
|
|
37
|
-
* had eaten it.
|
|
38
|
-
*
|
|
39
|
-
* THE WHOLE PIN SURVIVES, not the pair inside it: the entry's own effort, harness and cost knobs are what the
|
|
40
|
-
* turn is composed from (turn-resume.ts), so a resolver that handed back a bare (provider, model) would silently
|
|
41
|
-
* run the head of the list at the provider's defaults. Nothing here reads or judges those fields, which is the
|
|
42
|
-
* point of carrying them whole.
|
|
43
|
-
*
|
|
44
|
-
* Empty out means nobody has pinned anything this sandbox can reach, and the caller's floor takes over. */
|
|
45
|
-
export const resolveAgentRunModels = (sources: readonly QuickModelSource[], pinned: readonly AgentRunPin[]): readonly AgentRunPin[] => {
|
|
46
|
-
const ready = new Set(sources.filter((source) => source.ready).map((source) => source.provider));
|
|
47
|
-
// Taken verbatim, unvalidated against the catalog, the same reading resolveQuickModels gives its keys and
|
|
48
|
-
// for the same reason: the picker offers a custom-id escape hatch for a model the static catalog has not
|
|
49
|
-
// caught up with, and second-guessing the id here would run a different model than the settings row names.
|
|
50
|
-
const requested = pinned.filter((pin) => ready.has(pin.provider));
|
|
51
|
-
/* The same model twice would spend two attempts proving the same account is out. Hand-edited list, so this
|
|
52
|
-
* is a real state rather than a defensive branch.
|
|
53
|
-
*
|
|
54
|
-
* THE FIRST OF A PAIR WINS, WHOLE. Two entries can now name one model and differ in their knobs (the same
|
|
55
|
-
* Sonnet at Max and again at Low, written while reordering the list), and the one the user reads first is
|
|
56
|
-
* the one they meant; keeping the earlier position with the later entry's effort would run a tier that
|
|
57
|
-
* appears nowhere the pin does. */
|
|
58
|
-
const chain: AgentRunPin[] = [];
|
|
59
|
-
for (const pin of requested) {
|
|
60
|
-
if (!chain.some((held) => quickModelKey(held) === quickModelKey(pin))) {
|
|
61
|
-
chain.push(pin);
|
|
62
|
-
}
|
|
63
|
-
}
|
|
64
|
-
return chain;
|
|
65
|
-
};
|