@coinrithm/mcp-trading 0.7.5 → 0.7.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -41,15 +41,28 @@ export function buildSystemPrompt(spec, mergedProse,
41
41
  opts = {}) {
42
42
  const r = spec.risk;
43
43
  const v = spec.venues;
44
- const includeForecast = opts.includeForecast === true;
44
+ const hasFutures = v.includes("futures");
45
+ const hasSpot = v.includes("spot");
46
+ const hasPm = v.includes("pm");
47
+ const hasCoinVenue = hasFutures || hasSpot;
48
+ const coinVenueLabel = [
49
+ ...(hasSpot ? ["spot"] : []),
50
+ ...(hasFutures ? ["futures"] : []),
51
+ ].join(" + ");
52
+ const includeForecast = hasPm && opts.includeForecast === true;
53
+ const sizeKinds = [
54
+ ...(hasFutures ? ["futures margin"] : []),
55
+ ...(hasSpot ? ["spot buy notional"] : []),
56
+ ...(hasPm ? ["PM stake"] : []),
57
+ ];
45
58
  const actions = [];
46
- if (v.includes("futures")) {
59
+ if (hasFutures) {
47
60
  actions.push('{"type":"futures_open","symbol","side":"long"|"short","leverage","marginMusd","stopLossPrice","takeProfitPrice","confidence":0..1}', '{"type":"futures_close","positionId","fraction"}', '{"type":"futures_set_sltp","positionId","stopLossPrice","takeProfitPrice"}', "FUTURES TRIGGER RULES (the server rejects the WHOLE open otherwise): a LONG's takeProfitPrice must be ABOVE the current mark and stopLossPrice BELOW it (and above liquidationPrice); a SHORT is inverted (TP below mark, SL above). Every open position in observation.openPositions shows entryPrice, markPrice, liquidationPrice, stopLossPrice, takeProfitPrice — read them and place triggers on the correct side. NEVER attach stopLossPrice/takeProfitPrice to a futures_open for a symbol you ALREADY hold (the server treats it as an add and rejects it) — adjust that position with futures_set_sltp on its positionId instead.");
48
61
  }
49
- if (v.includes("spot")) {
62
+ if (hasSpot) {
50
63
  actions.push('{"type":"spot_order","symbol","side":"buy"|"sell","orderType":"market"|"limit"|"stop","quantity","limitPrice","stopPrice","confidence":0..1}', '{"type":"spot_cancel","orderId"}');
51
64
  }
52
- if (v.includes("pm")) {
65
+ if (hasPm) {
53
66
  actions.push(`{"type":"pm_open","ref":"pmN","stakeMusd","confidence":0..1${includeForecast ? ',"forecastProbability":1..99' : ""}} (set "ref" to one of the refs listed THIS cycle (pm1..pmN) — the \`ref\` of the ONE observation.pmMarkets entry you are betting, e.g. "pm3", copied EXACTLY; a ref NOT in this cycle's list is rejected as pm_ref_unknown and wastes the cycle; stakeMusd >= 10${includeForecast ? '; set "forecastProbability" to YOUR OWN probability 1-99 that this outcome wins — see the forecast rule below' : ""})`);
54
67
  }
55
68
  return [
@@ -61,17 +74,44 @@ opts = {}) {
61
74
  "",
62
75
  "## Hard caps the runner enforces (do not exceed; proposing over a cap wastes the cycle)",
63
76
  `- venues you may act in: ${v.join(", ")}`,
64
- `- perTradeMarginMusd ${r.perTradeMarginMusd} is the per-trade SIZE cap (futures margin / spot buy notional / PM stake)`,
65
- `- futures: maxLeverage ${r.maxLeverage}, maxConcurrentPositions ${r.maxConcurrentPositions}, requireStopLoss ${r.requireStopLoss} (long stop below entry, short stop above)`,
66
- `- watchlist (spot + futures use ONLY these): ${r.watchlist.join(", ")}`,
67
- ...(r.blocklist && r.blocklist.length > 0
77
+ `- perTradeMarginMusd ${r.perTradeMarginMusd} is the per-trade SIZE cap (${sizeKinds.join(" / ")})`,
78
+ ...(hasFutures
79
+ ? [
80
+ `- futures: maxLeverage ${r.maxLeverage}, maxConcurrentPositions ${r.maxConcurrentPositions}, requireStopLoss ${r.requireStopLoss} (long stop below entry, short stop above)`,
81
+ ]
82
+ : []),
83
+ ...(r.direction
84
+ ? [
85
+ r.direction === "short_only"
86
+ ? '- DIRECTION: SHORT ONLY — every futures_open MUST be side:"short" (and spot buys are forbidden: they are long exposure). A long is REJECTED by the runner no matter how strong the setup looks; a long-bias setup is never yours to take, only to fade when YOUR criteria are met.'
87
+ : '- DIRECTION: LONG ONLY — every futures_open MUST be side:"long". A short is REJECTED by the runner no matter how strong the setup looks.',
88
+ ]
89
+ : []),
90
+ // With universe_scan, the validator's gate is WATCH-membership (manual
91
+ // watchlist ∪ this cycle's discovered entries) — saying "ONLY these" here
92
+ // while the universe-scan section below calls discovered movers tradable
93
+ // made cap-obedient models refuse every discovered candidate (the caps
94
+ // header says proposing outside a cap wastes the cycle). Keep the two
95
+ // sections telling one story.
96
+ ...(hasCoinVenue
97
+ ? [
98
+ spec.capabilities.includes("universe_scan")
99
+ ? `- tradable symbols (${coinVenueLabel}): your watchlist (${r.watchlist.join(", ")}) PLUS this cycle's watch entries marked \`discovered: true\` — nothing outside those`
100
+ : `- watchlist (${coinVenueLabel} use ONLY these): ${r.watchlist.join(", ")}`,
101
+ ]
102
+ : []),
103
+ ...(hasCoinVenue && r.blocklist && r.blocklist.length > 0
68
104
  ? [
69
105
  `- deny-list (NEVER open these, even if on the watchlist): ${r.blocklist.join(", ")}`,
70
106
  ]
71
107
  : []),
72
- "- prediction markets are a FIRST-CLASS venue for you — a pm_open is as real a trade as a futures/spot open, not an afterthought. Each observation.pmMarkets entry carries a short `ref` (pm1, pm2, …), an `outcome` label, and `prob` (0..1, the market's CURRENT odds). BET (pm_open) an outcome when YOUR estimate of its true probability differs MATERIALLY from the market's — that gap is your edge (e.g. prob 0.35 but you think it's really ~0.55 -> buy). Skip only markets pinned near 0 or 1 (no edge left). Every entry in observation.pmMarkets is already filtered to one you CAN open (binary/settlement-grade) — so a listed market will not bounce at quote. Pick ONLY a listed market and identify it by copying its `ref` into the action; min stake 10 mUSD. Do NOT re-bet a market+outcome you ALREADY hold (check observation.pmPositions) — that is churn and will be rejected; bet a DIFFERENT market or skip.",
73
- "- PM stake is a SEPARATE budget from your futures margin: the futures margin cap (maxOpenMarginMusd) does NOT limit pm_open. So when your futures are at the margin/position cap — you hold the max, or a futures_open keeps getting REJECTED with open_margin_exceeds_cap — prediction markets are STILL fully open to you. PIVOT to pm_open on a mispriced market instead of re-proposing a futures_open that will just be rejected: a rejected open wastes the entire cycle, an eligible PM bet does not.",
74
- "- YOUR SHARPEST PM EDGE is the crypto price view you JUST formed: crypto PM markets resolve on the very prices you analyse, so you have a genuine information edge there that you do NOT have on coin futures alone. EVERY cycle you reach a price conviction, it is REQUIRED that you scan observation.pmMarkets for a LISTED crypto market that same view prices wrong and, if one is materially mispriced, open it with pm_open by its `ref` — treat that mispricing exactly like a flagged coin setup (an ACT, not a skip). If you are bearish BTC, a 'BTC above $X by <date>' priced high is a NO; if bullish ETH, an 'ETH above $Y' priced low is a YES. ESCAPE HATCH — only the markets actually listed in observation.pmMarkets THIS cycle (pm1..pmN) are bettable: if NONE of them matches the coin or view you formed, that is a legitimate SKIP for PM (say so in one clause and move on) — do NOT invent, guess, or increment a ref for a market you wish existed, because a made-up ref is rejected (pm_ref_unknown) and wastes the whole cycle exactly like a rejected open. The mistake to avoid is leaving a LISTED, clearly mispriced crypto market untraded — a mispricing that is NOT on this cycle's board is simply not actionable now, not a miss. (For non-crypto events you have no special edge; skip unless the odds are obviously off.)",
108
+ ...(hasPm
109
+ ? [
110
+ "- prediction markets are a FIRST-CLASS venue for you — a pm_open is as real a trade as a futures/spot open, not an afterthought. Each observation.pmMarkets entry carries a short `ref` (pm1, pm2, …), an `outcome` label, and `prob` (0..1, the market's CURRENT odds). BET (pm_open) an outcome when YOUR estimate of its true probability differs MATERIALLY from the market's — that gap is your edge (e.g. prob 0.35 but you think it's really ~0.55 -> buy). Skip only markets pinned near 0 or 1 (no edge left). Every entry in observation.pmMarkets is already filtered to one you CAN open (binary/settlement-grade) — so a listed market will not bounce at quote. Pick ONLY a listed market and identify it by copying its `ref` into the action; min stake 10 mUSD. Do NOT re-bet a market+outcome you ALREADY hold (check observation.pmPositions) — that is churn and will be rejected; bet a DIFFERENT market or skip.",
111
+ "- PM stake is a SEPARATE budget from your futures margin: the futures margin cap (maxOpenMarginMusd) does NOT limit pm_open. So when your futures are at the margin/position cap — you hold the max, or a futures_open keeps getting REJECTED with open_margin_exceeds_cap — prediction markets are STILL fully open to you. PIVOT to pm_open on a mispriced market instead of re-proposing a futures_open that will just be rejected: a rejected open wastes the entire cycle, an eligible PM bet does not.",
112
+ "- YOUR SHARPEST PM EDGE is the crypto price view you JUST formed: crypto PM markets resolve on the very prices you analyse, so you have a genuine information edge there that you do NOT have on coin futures alone. EVERY cycle you reach a price conviction, it is REQUIRED that you scan observation.pmMarkets for a LISTED crypto market that same view prices wrong and, if one is materially mispriced, open it with pm_open by its `ref` — treat that mispricing exactly like a flagged coin setup (an ACT, not a skip). If you are bearish BTC, a 'BTC above $X by <date>' priced high is a NO; if bullish ETH, an 'ETH above $Y' priced low is a YES. ESCAPE HATCH — only the markets actually listed in observation.pmMarkets THIS cycle (pm1..pmN) are bettable: if NONE of them matches the coin or view you formed, that is a legitimate SKIP for PM (say so in one clause and move on) — do NOT invent, guess, or increment a ref for a market you wish existed, because a made-up ref is rejected (pm_ref_unknown) and wastes the whole cycle exactly like a rejected open. The mistake to avoid is leaving a LISTED, clearly mispriced crypto market untraded — a mispricing that is NOT on this cycle's board is simply not actionable now, not a miss. (For non-crypto events you have no special edge; skip unless the odds are obviously off.)",
113
+ ]
114
+ : []),
75
115
  ...(includeForecast
76
116
  ? [
77
117
  "- FORECAST RULE (pm_open forecastProbability): before you look at what the market is pricing, decide YOUR OWN probability the outcome you are backing actually WINS — reason ONLY from the question, its resolution criteria, and the deadline. Put that number (1-99, whole or one decimal) in `forecastProbability`. This is graded against reality as your PUBLIC calibration record, so it must be YOUR judgement, NOT the market's: do NOT copy, round, or anchor it to the observation.pmMarkets `prob`. It is FINE if your honest forecast happens to land on the market's number — but reaching that by echoing the price defeats the point. If you genuinely cannot form an independent view, OMIT the field rather than parroting the market (an absent forecast is better than a fake one, and it never blocks the bet).",
@@ -87,6 +127,15 @@ opts = {}) {
87
127
  "- a null field = not enough data; ignore it. These INFORM your decision; they never widen a cap.",
88
128
  ]
89
129
  : []),
130
+ ...(spec.capabilities.includes("universe_scan")
131
+ ? [
132
+ "",
133
+ "## Universe scan (discovered movers) — candidates beyond your watchlist",
134
+ "Watch entries with `discovered: true` are today's strongest 24h movers across the WHOLE tracked universe, resolved with the same price/sentiment (and indicators) data as your watchlist. observation.universeMovers lists further movers as symbol + 24h change only (context — you cannot trade those directly this cycle).",
135
+ "- Treat a discovered candidate like any other symbol: analyze it for catalysts, exhaustion and reversal BEFORE acting. A big 24h pump is as often a top as a beginning — chasing green candles blind is how discovery loses money.",
136
+ "- All your normal risk rules apply unchanged: caps, stops, blocklist, confidence floor. Discovery widens what you can SEE, never what you may risk.",
137
+ ]
138
+ : []),
90
139
  ...(spec.capabilities.includes("news")
91
140
  ? [
92
141
  "",
@@ -94,7 +143,11 @@ opts = {}) {
94
143
  "Each item has `importance` (0..10; >=8 = genuinely market-moving), `sentiment` (bullish/bearish/neutral), `ageHours`, and the `coins` it concerns. Use it to CONFIRM or VETO the price read, never to trade on alone:",
95
144
  "- A fresh high-importance (>=8) bullish story on a coin you're watching strengthens a long and warns against shorting into it; a bearish >=8 is the reverse. A surprise catalyst can matter more than the chart.",
96
145
  "- Weight by importance AND freshness: a 9 from 30 min ago outweighs a stale 4 from yesterday. Old or low-importance news is noise — don't over-react.",
97
- "- For PM: a high-importance catalyst is exactly the kind of mispricing edge to act on if the market hasn't repriced it yet.",
146
+ ...(hasPm
147
+ ? [
148
+ "- For PM: a high-importance catalyst is exactly the kind of mispricing edge to act on if the market hasn't repriced it yet.",
149
+ ]
150
+ : []),
98
151
  ]
99
152
  : []),
100
153
  "",
@@ -107,12 +160,24 @@ opts = {}) {
107
160
  "## How to act — a decisive trader in character, not a bystander",
108
161
  "You ARE the character in the strategy above; trade like it. When you have a clear read — even a moderate-confidence one — TAKE THE POSITION, sized within your caps and protected with a stop. You wake every cycle and people watch you live: an agent that watches forever and never commits is useless to them and to itself.",
109
162
  "Skip ONLY when the read is genuinely contradictory (signals fight each other), the data is stale, or you truly have no edge this cycle. A quiet tape where your thesis still has a small but REAL edge is an ACT, not a skip — take it, small, with a stop. Do not confuse caution with paralysis.",
110
- 'In "rationale" (shown LIVE in your public terminal) speak in YOUR voice and commit to a view in 1-2 vivid, specific sentences — what you see and what you are DOING about it, like a trader posting their move, not a risk report. Good: "ETH punched through the weekly high on real volume — long here with a stop under the breakout, this is exactly my setup." Weak: "conditions are mixed, waiting for clarity." Keep "reason" a short label.',
163
+ 'In "rationale" (shown LIVE in your public terminal) speak in YOUR voice and commit to a view in 1-2 vivid, specific sentences — what you see and what you are DOING about it, like a trader posting their move, not a risk report. Good: "ETH broke its recent20 high with EMA20 above EMA50 — long here with a stop under the breakout, this is exactly my setup." Weak: "conditions are mixed, waiting for clarity." Keep "reason" a short label.',
111
164
  "",
112
165
  "## Flagged setups this cycle — your wake-up list (observation.setups)",
113
166
  "A deterministic scan already checked every watchlist coin and put the ones with real, tradeable structure RIGHT NOW into observation.setups — each has symbol, kind, bias, strength, and a factual note (trend / RSI / breakout / ATR reads). This is your shortlist; you do NOT need to re-derive whether a setup exists.",
114
167
  '- If observation.setups is NON-EMPTY: act on the strongest one that fits YOUR strategy. The `bias` is the trend-following read; if you are a contrarian / mean-reversion trader, FADE it with the same facts (e.g. a downtrend that is also "RSI oversold" is YOUR long). Skipping a flagged setup needs a SPECIFIC reason tied to your thesis — "no clear setup" is NOT a valid skip when setups are listed.',
115
- "- If observation.setups is EMPTY: no coin has a flagged structure right now — but BEFORE you skip, check observation.pmMarkets for a crypto market your current read prices wrong (a PM mispricing is a valid ACT even with zero coin setups). Only then, if nothing is mispriced, skip new entries and just manage any open positions.",
168
+ // The act-pressure above must never outrank a hard cap: without this
169
+ // release valve a direction-constrained agent, staring at only wrong-way
170
+ // setups, is squeezed between "skipping needs a specific reason" and a
171
+ // constraint the runner enforces — that squeeze is how a short-only agent
172
+ // opened momentum longs on 2026-08-24.
173
+ ...(r.direction
174
+ ? [
175
+ `- Your DIRECTION cap outranks this list: a setup whose only actionable read violates it (${r.direction === "short_only" ? "long" : "short"}-side) is a LEGITIMATE skip — name the constraint in one clause and move on. Never take the wrong side to avoid skipping.`,
176
+ ]
177
+ : []),
178
+ hasPm
179
+ ? "- If observation.setups is EMPTY: no coin has a flagged structure right now — but BEFORE you skip, check observation.pmMarkets for a crypto market your current read prices wrong (a PM mispricing is a valid ACT even with zero coin setups). Only then, if nothing is mispriced, skip new entries and just manage any open positions."
180
+ : "- If observation.setups is EMPTY: no coin has a flagged structure right now — skip new entries and just manage any open positions.",
116
181
  "- A setup tagged `held` (held: long|short) is a position you ALREADY hold. Do NOT propose a new open on it — that only hits the margin cap and wastes the cycle. MANAGE it instead: trail the stop toward your target, ADD only if you have margin room AND fresh conviction, or cut if the thesis broke.",
117
182
  "",
118
183
  "## After you act — hold with conviction, do not churn",
@@ -123,7 +188,14 @@ opts = {}) {
123
188
  "Each cycle, look at your OPEN positions FIRST, not just new entries. A position that is working is your best opportunity: once it moves your way, move the stop to breakeven and then TRAIL it behind the move with futures_set_sltp so a winner keeps running instead of being cut early — and you may ADD to a confirming winner (scale in, never beyond your caps). A position that is clearly wrong — the level broke, the thesis failed — cut it cleanly instead of nursing it. Riding one good trade beats opening ten fresh ones.",
124
189
  ].join("\n");
125
190
  }
126
- export function buildUserPrompt(obs, journal) {
191
+ export function buildUserPrompt(obs, journal, opts = {}) {
192
+ // Default to every venue for backwards-compatible direct callers and probes.
193
+ // The runner always supplies the real spec, so disabled venue instructions and
194
+ // empty observation blocks never consume prompt space or invite invalid acts.
195
+ const venues = opts.venues ?? ["futures", "spot", "pm"];
196
+ const hasFutures = venues.includes("futures");
197
+ const hasSpot = venues.includes("spot");
198
+ const hasPm = venues.includes("pm");
127
199
  const lines = [
128
200
  "Decide for THIS cycle using only the observation below (data available now — no look-ahead).",
129
201
  ];
@@ -133,8 +205,17 @@ export function buildUserPrompt(obs, journal) {
133
205
  // zeroes the cycle). There is nothing to manage when flat, so say so plainly and
134
206
  // point the model at OPENING. (Observed: an 8B agent dead 36/60 cycles this way.)
135
207
  if ((obs.openPositions?.length ?? 0) === 0 &&
136
- (obs.pmPositions?.length ?? 0) === 0) {
137
- lines.push("You currently hold NO open positions and NO resting orders — there is NOTHING to manage or close this cycle. Do NOT emit any futures_close, futures_set_sltp, or spot_cancel action (you have no position/order id to act on; doing so just wastes the cycle). Your ONLY moves are to OPEN the best available setup (futures_open / spot_order / pm_open) or to skip.");
208
+ (!hasPm || (obs.pmPositions?.length ?? 0) === 0)) {
209
+ const openingActions = [
210
+ ...(hasFutures ? ["futures_open"] : []),
211
+ ...(hasSpot ? ["spot_order"] : []),
212
+ ...(hasPm ? ["pm_open"] : []),
213
+ ];
214
+ const forbiddenActions = [
215
+ ...(hasFutures ? ["futures_close", "futures_set_sltp"] : []),
216
+ ...(hasSpot ? ["spot_cancel"] : []),
217
+ ];
218
+ lines.push(`You currently hold NO open positions${hasPm ? " and NO prediction-market positions" : ""} and NO resting orders — there is NOTHING to manage or close this cycle.${forbiddenActions.length > 0 ? ` Do NOT emit any ${forbiddenActions.join(", ")} action (you have no position/order id to act on; doing so just wastes the cycle).` : ""} Your ONLY moves are to OPEN the best available setup (${openingActions.join(" / ")}) or to skip.`);
138
219
  }
139
220
  // Slice-3 memory: the agent's own recent moves, so it manages with continuity —
140
221
  // remembers the thesis behind each open position and does not re-open an idea it
@@ -144,7 +225,8 @@ export function buildUserPrompt(obs, journal) {
144
225
  }
145
226
  // Settlement-feedback loop: surface the agent's recently-RESOLVED PM bets so the
146
227
  // model can reflect and adapt. Reflective context only — never a new action.
147
- lines.push(...formatPmResolutions(obs.pmResolutions ?? []));
228
+ if (hasPm)
229
+ lines.push(...formatPmResolutions(obs.pmResolutions ?? []));
148
230
  lines.push("", "```json",
149
231
  // Compact (no pretty-print indentation — ~40% fewer tokens, still valid JSON)
150
232
  // and the trade ledger is capped so a busy shared book can't bloat the prompt.
@@ -154,21 +236,26 @@ export function buildUserPrompt(obs, journal) {
154
236
  equityMusd: obs.equityMusd,
155
237
  openPositions: obs.openPositions,
156
238
  openOrders: obs.openOrders,
157
- pmPositions: obs.pmPositions,
239
+ ...(hasPm ? { pmPositions: obs.pmPositions } : {}),
158
240
  // Compact display: the model picks a market by its short `ref` and never
159
241
  // sees (or mis-copies) the long source/slug/outcomeExternalMarketId — the
160
242
  // runner resolves the ref back to those. Also ~halves the PM block's tokens.
161
- pmMarkets: obs.pmMarkets.map((m) => ({
162
- ref: m.ref,
163
- source: m.source,
164
- title: m.title,
165
- outcome: m.outcomeName,
166
- prob: m.probability,
167
- freshness: m.freshness?.status,
168
- })),
243
+ ...(hasPm
244
+ ? {
245
+ pmMarkets: obs.pmMarkets.map((m) => ({
246
+ ref: m.ref,
247
+ source: m.source,
248
+ title: m.title,
249
+ outcome: m.outcomeName,
250
+ prob: m.probability,
251
+ freshness: m.freshness?.status,
252
+ })),
253
+ }
254
+ : {}),
169
255
  watch: obs.watch,
170
256
  setups: obs.setups,
171
257
  news: obs.news,
258
+ universeMovers: obs.universeMovers,
172
259
  marketMood: obs.marketMood,
173
260
  newClosedTrades: obs.newClosedTrades.slice(0, 20),
174
261
  polledBeforeWrite: obs.polledBeforeWrite,
@@ -0,0 +1,20 @@
1
+ import { ProviderName } from "./types.js";
2
+ export declare const NVIDIA_BASE_URL = "https://integrate.api.nvidia.com/v1";
3
+ export interface ChatShape {
4
+ family: "openai-reasoning" | "nvidia-nemotron" | "anthropic" | "openai-compat";
5
+ tokenParam: "max_tokens" | "max_completion_tokens";
6
+ allowsTemperature: boolean;
7
+ jsonResponseFormat: boolean;
8
+ extraBody?: Record<string, unknown>;
9
+ systemHint?: string;
10
+ minProbeCompletionTokens: number;
11
+ }
12
+ export declare function chatShapeFor(provider: ProviderName, model: string, baseUrl?: string): ChatShape;
13
+ /** Build the chat-completions body for a route from its capability shape. */
14
+ export declare function buildChatBody(shape: ChatShape, args: {
15
+ model: string;
16
+ system: string;
17
+ user: string;
18
+ maxTokens: number;
19
+ temperature?: number;
20
+ }): Record<string, unknown>;
@@ -0,0 +1,67 @@
1
+ export const NVIDIA_BASE_URL = "https://integrate.api.nvidia.com/v1";
2
+ // gpt-5*, o1/o3/o4* — the OpenAI reasoning-API family, wherever it is served.
3
+ const OPENAI_REASONING_MODEL = /^(gpt-5|o[0-9])/i;
4
+ const NEMOTRON_MODEL = /nemotron/i;
5
+ export function chatShapeFor(provider, model, baseUrl) {
6
+ if (provider === "anthropic") {
7
+ return {
8
+ family: "anthropic",
9
+ tokenParam: "max_tokens",
10
+ allowsTemperature: true,
11
+ jsonResponseFormat: false,
12
+ minProbeCompletionTokens: 1024,
13
+ };
14
+ }
15
+ if (provider === "openai" || OPENAI_REASONING_MODEL.test(model)) {
16
+ return {
17
+ family: "openai-reasoning",
18
+ tokenParam: "max_completion_tokens",
19
+ allowsTemperature: false,
20
+ jsonResponseFormat: true,
21
+ minProbeCompletionTokens: 1024,
22
+ };
23
+ }
24
+ if (NEMOTRON_MODEL.test(model)) {
25
+ return {
26
+ family: "nvidia-nemotron",
27
+ tokenParam: "max_tokens",
28
+ allowsTemperature: true,
29
+ jsonResponseFormat: true,
30
+ // The kwargs switch is only honored (and only safe to send) on the NVIDIA
31
+ // endpoint; the system hint helps on any endpoint serving a Nemotron.
32
+ extraBody: baseUrl === NVIDIA_BASE_URL
33
+ ? { chat_template_kwargs: { enable_thinking: false } }
34
+ : undefined,
35
+ systemHint: "detailed thinking off",
36
+ minProbeCompletionTokens: 1024,
37
+ };
38
+ }
39
+ return {
40
+ family: "openai-compat",
41
+ tokenParam: "max_tokens",
42
+ allowsTemperature: true,
43
+ jsonResponseFormat: true,
44
+ minProbeCompletionTokens: 1024,
45
+ };
46
+ }
47
+ /** Build the chat-completions body for a route from its capability shape. */
48
+ export function buildChatBody(shape, args) {
49
+ const system = shape.systemHint
50
+ ? `${shape.systemHint}\n\n${args.system}`
51
+ : args.system;
52
+ return {
53
+ model: args.model,
54
+ ...(shape.allowsTemperature
55
+ ? { temperature: args.temperature ?? 0.2 }
56
+ : {}),
57
+ [shape.tokenParam]: args.maxTokens,
58
+ ...(shape.jsonResponseFormat
59
+ ? { response_format: { type: "json_object" } }
60
+ : {}),
61
+ ...(shape.extraBody ?? {}),
62
+ messages: [
63
+ { role: "system", content: system },
64
+ { role: "user", content: args.user },
65
+ ],
66
+ };
67
+ }
@@ -1,10 +1,28 @@
1
- import { AgentSpec } from "./types.js";
1
+ import { AgentSpec, ProviderName } from "./types.js";
2
2
  export interface DecideInput {
3
3
  system: string;
4
4
  user: string;
5
5
  maxTokens?: number;
6
6
  timeoutMs?: number;
7
7
  }
8
+ export interface DecideRouteAttempt {
9
+ provider: string;
10
+ model: string;
11
+ outcome: "success" | "failed" | "deferred";
12
+ failureClass?: "capacity" | "permanent" | "transient" | "malformed";
13
+ status?: number;
14
+ retryAfterMs?: number;
15
+ latencyMs: number;
16
+ error?: string;
17
+ }
18
+ export interface DecideRouteMeta {
19
+ policyVersion: string;
20
+ profile: "fast" | "strong" | "configured";
21
+ effectiveProvider?: string;
22
+ effectiveModel?: string;
23
+ reason: "configured" | "circuit_fallback" | "capacity_fallback" | "provider_fallback" | "malformed_fallback" | "byo";
24
+ attempts: DecideRouteAttempt[];
25
+ }
8
26
  export type DecideResult = {
9
27
  ok: true;
10
28
  text: string;
@@ -12,9 +30,14 @@ export type DecideResult = {
12
30
  promptTokens: number;
13
31
  completionTokens: number;
14
32
  };
33
+ route?: DecideRouteMeta;
15
34
  } | {
16
35
  ok: false;
17
36
  error: string;
37
+ status?: number;
38
+ retryAfterMs?: number;
39
+ deferred?: boolean;
40
+ route?: DecideRouteMeta;
18
41
  };
19
42
  export interface Provider {
20
43
  label: string;
@@ -29,3 +52,8 @@ export interface ProviderEnv {
29
52
  MODEL_API_KEY?: string;
30
53
  }
31
54
  export declare function selectProvider(spec: AgentSpec, env: ProviderEnv, fetchFn?: typeof fetch): Provider;
55
+ export declare function providerForRoute(route: {
56
+ provider: ProviderName;
57
+ model: string;
58
+ baseUrl?: string | null;
59
+ }, apiKey: string, fetchFn?: typeof fetch): Provider;
@@ -2,9 +2,10 @@
2
2
  // never from an agent file. One call returns one chunk of text that must be a
3
3
  // single structured-JSON decision (parsed in decision.ts). No free-form tool
4
4
  // execution — the model only proposes; the runner disposes.
5
+ import { chatShapeFor, buildChatBody, NVIDIA_BASE_URL as CAP_NVIDIA_BASE_URL, } from "./providerCapabilities.js";
5
6
  // NVIDIA NIM is OpenAI-compatible; the `nvidia` preset hard-wires the hosted
6
7
  // endpoint so an agent only needs `{ provider: nvidia, name: "<model id>" }`.
7
- const NVIDIA_BASE_URL = "https://integrate.api.nvidia.com/v1";
8
+ const NVIDIA_BASE_URL = CAP_NVIDIA_BASE_URL;
8
9
  // Gemini exposes an OpenAI-compatible surface, so the `gemini` preset hard-wires
9
10
  // its hosted endpoint — an agent only needs `{ provider: gemini, name: "gemini-2.0-flash" }`
10
11
  // plus a GEMINI_API_KEY. The free tier (no credit card, generous Flash quota) makes
@@ -26,16 +27,9 @@ const GEMINI_BASE_URL = "https://generativelanguage.googleapis.com/v1beta/openai
26
27
  // (the recurring Leo/70B timeout). A real hang still aborts -> retried next cadence.
27
28
  // MUST stay below the scheduler's RUN_LOCK_SECONDS and HEARTBEAT_STALE_MS.
28
29
  const DEFAULT_TIMEOUT_MS = 300_000;
29
- // Reasoning models (NVIDIA Nemotron) DEFAULT to emitting a long <think> chain:
30
- // measured ~30-60s/call and a JSON-leak risk. The documented toggle is a
31
- // "detailed thinking off" line in the system prompt, which drops them to
32
- // instruct mode (measured ~3-4s, clean JSON). Apply it automatically for any
33
- // nemotron model so a per-cadence decision never blows the cadence.
34
- function applyReasoningToggle(model, system) {
35
- return /nemotron/i.test(model)
36
- ? `detailed thinking off\n\n${system}`
37
- : system;
38
- }
30
+ // Per-route request quirks (reasoning toggles, token param, temperature) live
31
+ // in the capability table — providerCapabilities.ts is the single source; this
32
+ // module only assembles and sends.
39
33
  // fetch with a hard timeout via AbortController. A custom fetchFn (tests) that
40
34
  // ignores `signal` still works — the timer just never fires for it.
41
35
  async function fetchWithTimeout(fetchFn, url, init, timeoutMs) {
@@ -54,6 +48,23 @@ function callError(err, timeoutMs) {
54
48
  }
55
49
  return err instanceof Error ? err.message : String(err);
56
50
  }
51
+ // Parse a Retry-After header (delta-seconds or HTTP-date) into ms, capped at
52
+ // one hour — a provider asking for more is treated as "an hour, then re-probe".
53
+ const RETRY_AFTER_CAP_MS = 3_600_000;
54
+ function retryAfterMs(res) {
55
+ const raw = res.headers.get("retry-after");
56
+ if (!raw)
57
+ return undefined;
58
+ const secs = Number(raw);
59
+ if (Number.isFinite(secs) && secs >= 0) {
60
+ return Math.min(Math.round(secs * 1000), RETRY_AFTER_CAP_MS);
61
+ }
62
+ const at = Date.parse(raw);
63
+ if (!Number.isFinite(at))
64
+ return undefined;
65
+ const ms = at - Date.now();
66
+ return ms > 0 ? Math.min(ms, RETRY_AFTER_CAP_MS) : 0;
67
+ }
57
68
  function envKey(provider, env) {
58
69
  switch (provider) {
59
70
  case "anthropic":
@@ -141,6 +152,8 @@ class AnthropicProvider {
141
152
  // Cap the upstream body: it lands in agent_cycles.skip_reason, so an
142
153
  // unbounded provider error page must not bloat the ledger row.
143
154
  error: `anthropic HTTP ${res.status}: ${(await res.text()).slice(0, 2000)}`,
155
+ status: res.status,
156
+ retryAfterMs: retryAfterMs(res),
144
157
  };
145
158
  const json = (await res.json());
146
159
  const text = json.content?.map((c) => c.text ?? "").join("") ?? "";
@@ -160,12 +173,14 @@ class AnthropicProvider {
160
173
  }
161
174
  }
162
175
  class OpenAiCompatProvider {
176
+ provider;
163
177
  model;
164
178
  apiKey;
165
179
  baseUrl;
166
180
  fetchFn;
167
181
  label;
168
- constructor(model, apiKey, baseUrl, fetchFn) {
182
+ constructor(provider, model, apiKey, baseUrl, fetchFn) {
183
+ this.provider = provider;
169
184
  this.model = model;
170
185
  this.apiKey = apiKey;
171
186
  this.baseUrl = baseUrl;
@@ -174,6 +189,7 @@ class OpenAiCompatProvider {
174
189
  }
175
190
  async decide(input) {
176
191
  const timeoutMs = input.timeoutMs ?? DEFAULT_TIMEOUT_MS;
192
+ const shape = chatShapeFor(this.provider, this.model, this.baseUrl);
177
193
  try {
178
194
  const res = await fetchWithTimeout(this.fetchFn, `${this.baseUrl}/chat/completions`, {
179
195
  method: "POST",
@@ -181,19 +197,12 @@ class OpenAiCompatProvider {
181
197
  Authorization: `Bearer ${this.apiKey}`,
182
198
  "content-type": "application/json",
183
199
  },
184
- body: JSON.stringify({
200
+ body: JSON.stringify(buildChatBody(shape, {
185
201
  model: this.model,
186
- temperature: 0.2,
187
- max_tokens: input.maxTokens ?? 1024,
188
- response_format: { type: "json_object" },
189
- messages: [
190
- {
191
- role: "system",
192
- content: applyReasoningToggle(this.model, input.system),
193
- },
194
- { role: "user", content: input.user },
195
- ],
196
- }),
202
+ system: input.system,
203
+ user: input.user,
204
+ maxTokens: input.maxTokens ?? 1024,
205
+ })),
197
206
  }, timeoutMs);
198
207
  if (!res.ok)
199
208
  return {
@@ -201,6 +210,8 @@ class OpenAiCompatProvider {
201
210
  // Cap the upstream body: it lands in agent_cycles.skip_reason, so an
202
211
  // unbounded provider error page must not bloat the ledger row.
203
212
  error: `provider HTTP ${res.status}: ${(await res.text()).slice(0, 2000)}`,
213
+ status: res.status,
214
+ retryAfterMs: retryAfterMs(res),
204
215
  };
205
216
  const json = (await res.json());
206
217
  const text = json.choices?.[0]?.message?.content ?? "";
@@ -251,5 +262,19 @@ export function selectProvider(spec, env, fetchFn = fetch) {
251
262
  if (!resolvedBase) {
252
263
  throw new Error("openai-compatible provider needs model.baseUrl");
253
264
  }
254
- return new OpenAiCompatProvider(name, key, resolvedBase, fetchFn);
265
+ return new OpenAiCompatProvider(provider, name, key, resolvedBase, fetchFn);
266
+ }
267
+ // Build a provider for an EXPLICIT route + raw key (no spec, no env) — the
268
+ // decision probe's entry point. Same classes as selectProvider, so a probe
269
+ // exercises byte-identical request shapes to a real cycle.
270
+ export function providerForRoute(route, apiKey, fetchFn = fetch) {
271
+ if (route.provider === "mechanical")
272
+ return new MechanicalProvider(route.model);
273
+ if (route.provider === "anthropic")
274
+ return new AnthropicProvider(route.model, apiKey, fetchFn);
275
+ const resolvedBase = baseUrlFor(route.provider, route.baseUrl ?? undefined);
276
+ if (!resolvedBase) {
277
+ throw new Error("openai-compatible route needs a baseUrl");
278
+ }
279
+ return new OpenAiCompatProvider(route.provider, route.model, apiKey, resolvedBase, fetchFn);
255
280
  }
@@ -3,9 +3,21 @@ export declare class ResolveError extends Error {
3
3
  issues: ResolveIssue[];
4
4
  constructor(issues: ResolveIssue[]);
5
5
  }
6
+ export declare const proseBody: (raw: string) => string;
7
+ export declare const GUARDS_FILE = "character/guards.md";
8
+ export declare const GUARDS_HEADER = "## HARD BEHAVIORAL GUARDS \u2014 never violate these";
9
+ export declare const GUARDS_FOOTER = "(These guards override every other instruction in this strategy. When a guard conflicts with an opportunity, the guard wins and the correct output is a skip that names the guard.)";
10
+ export declare const wrapGuardsProse: (body: string) => string;
6
11
  export declare function isSkillProseSource(source: string): boolean;
7
12
  export declare function mergeProseParts(parts: Array<{
8
13
  source: string;
9
14
  text: string;
10
15
  }>): string;
11
16
  export declare function resolveAgent(inputPath: string): ResolvedAgent;
17
+ export declare const HOSTED_PROSE_MAX_CHARS = 12000;
18
+ /** PURE — exported for tests. Mirrors the backend's trim-then-measure. */
19
+ export declare const hostedProseBudget: (mergedProse: string) => {
20
+ used: number;
21
+ over: number;
22
+ fits: boolean;
23
+ };