@code-yeongyu/senpi 2026.9.29-4 → 2026.9.29-5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +28 -7
- package/dist/bundle/chunks/{anthropic-messages-DZAH73IA.js → anthropic-messages-HEVKB3PL.js} +1 -1
- package/dist/bundle/chunks/{app-server-command-BUIGVCL4.js → app-server-command-32AFWZEH.js} +1 -1
- package/dist/bundle/chunks/{azure-openai-responses-ZJGVMXRJ.js → azure-openai-responses-QVPPCPVL.js} +1 -1
- package/dist/bundle/chunks/{chunk-U5LXY7UD.js → chunk-2IHY6ZLW.js} +1 -1
- package/dist/bundle/chunks/{chunk-ESPYV6XG.js → chunk-2N3BZ5DV.js} +1 -1
- package/dist/bundle/chunks/{chunk-XNR637JQ.js → chunk-2WAKOPNA.js} +1 -1
- package/dist/bundle/chunks/{chunk-SHIPI3FN.js → chunk-63QHOG5A.js} +44 -44
- package/dist/bundle/chunks/{chunk-H2U7247O.js → chunk-77AUUH3Z.js} +2 -2
- package/dist/bundle/chunks/{chunk-T4F7KOR2.js → chunk-7CQX3QFV.js} +1 -1
- package/dist/bundle/chunks/{chunk-WW3YFGEP.js → chunk-7MUTNWPO.js} +1 -1
- package/dist/bundle/chunks/{chunk-6NXC765G.js → chunk-AWEAVBCR.js} +1 -1
- package/dist/bundle/chunks/{chunk-3DISE3EE.js → chunk-BBNWT72C.js} +1 -1
- package/dist/bundle/chunks/{chunk-HKUHLCH7.js → chunk-U7BNOE23.js} +1 -1
- package/dist/bundle/chunks/{chunk-VAKMXHSM.js → chunk-VLW22GHJ.js} +1 -1
- package/dist/bundle/chunks/{chunk-DFRI62DO.js → chunk-W4IEEYWL.js} +1 -1
- package/dist/bundle/chunks/{chunk-Q6X7ES7N.js → chunk-XO7VCMY6.js} +1 -1
- package/dist/bundle/chunks/chunk-YLQTKWEC.js +6 -0
- package/dist/bundle/chunks/{chunk-NUS5JXND.js → chunk-Z2N5372Y.js} +1 -1
- package/dist/bundle/chunks/{chunk-JQNR6OYN.js → chunk-ZDNR3YVJ.js} +1 -1
- package/dist/bundle/chunks/{chunk-DAKIX3RD.js → chunk-ZPMJIPQC.js} +1 -1
- package/dist/bundle/chunks/{cli-main-XCRRH3OR.js → cli-main-ZPMYM3A3.js} +1 -1
- package/dist/bundle/chunks/engine-RCHMSSBT.js +2 -0
- package/dist/bundle/chunks/github-copilot.js +1 -1
- package/dist/bundle/chunks/{google-generative-ai-RYFW7GSR.js → google-generative-ai-E3I6MS7E.js} +1 -1
- package/dist/bundle/chunks/{google-vertex-J64D4U2O.js → google-vertex-KFFI24FU.js} +1 -1
- package/dist/bundle/chunks/{help-fast-path-JJVINW3Y.js → help-fast-path-ZUNFOE53.js} +1 -1
- package/dist/bundle/chunks/{host-command-IZJU2H7Z.js → host-command-V2LDNNXB.js} +1 -1
- package/dist/bundle/chunks/{interactive-mode-KXRPZ2U6.js → interactive-mode-J6KGDRR7.js} +1 -1
- package/dist/bundle/chunks/{mistral-conversations-C7ZTJYKE.js → mistral-conversations-EZ4XNJXU.js} +1 -1
- package/dist/bundle/chunks/{multi-session-host-3XTEXGP4.js → multi-session-host-IBVJ3WN7.js} +1 -1
- package/dist/bundle/chunks/{openai-codex-responses-N5SCEWCY.js → openai-codex-responses-TMUCQW5E.js} +1 -1
- package/dist/bundle/chunks/{openai-completions-T4B2KZ4B.js → openai-completions-SA2UJW7Q.js} +1 -1
- package/dist/bundle/chunks/{openai-responses-N4OK2RIV.js → openai-responses-EHSD4L2M.js} +1 -1
- package/dist/bundle/chunks/{package-manager-cli-ZHWO6FT5.js → package-manager-cli-JJ5R7UWV.js} +1 -1
- package/dist/bundle/chunks/{rotation-stream-5GE3KAHL.js → rotation-stream-EX7PWQ4S.js} +1 -1
- package/dist/bundle/chunks/{rpc-mode-DEVPDAT4.js → rpc-mode-ITND2BVC.js} +1 -1
- package/dist/bundle/chunks/{session-control-endpoint-72A2NRVH.js → session-control-endpoint-UJJGG4WI.js} +1 -1
- package/dist/bundle/chunks/{session-picker-XYMNQ6AZ.js → session-picker-K5FIKV6C.js} +1 -1
- package/dist/bundle/chunks/session-worker.js +42 -42
- package/dist/bundle/cli.js +1 -1
- package/dist/bundle/index.js +1 -1
- package/dist/bundle/rpc-entry.js +1 -1
- package/dist/bundle/runtime-manifest.json +1 -1
- package/dist/core/auth-providers.d.ts +9 -0
- package/dist/core/auth-providers.js +17 -0
- package/dist/core/dynamic-prompt/build.js +1 -1
- package/dist/core/dynamic-prompt/intent-gate.js +1 -1
- package/dist/core/dynamic-prompt/verification.d.ts +9 -1
- package/dist/core/dynamic-prompt/verification.js +11 -2
- package/dist/core/extensions/builtin/prompt-preset/claude-fable-5-1.js +3 -3
- package/dist/core/extensions/builtin/prompt-preset/claude-fable-5.js +3 -3
- package/dist/core/extensions/builtin/prompt-preset/claude-opus-5-5.js +3 -3
- package/dist/core/extensions/builtin/prompt-preset/claude-opus-5.js +3 -3
- package/dist/core/extensions/builtin/prompt-preset/claude-sonnet-5-5.js +3 -3
- package/dist/core/extensions/builtin/prompt-preset/gpt-5.5.js +3 -3
- package/dist/core/extensions/builtin/prompt-preset/gpt-5.6.js +4 -4
- package/dist/core/extensions/builtin/prompt-preset/gpt-6-astra.d.ts +6 -2
- package/dist/core/extensions/builtin/prompt-preset/gpt-6-astra.js +24 -7
- package/dist/core/extensions/builtin/prompt-preset/gpt-surface.d.ts +4 -1
- package/dist/core/extensions/builtin/prompt-preset/gpt-surface.js +4 -1
- package/dist/core/extensions/builtin/prompt-preset/grok-4.5.js +4 -4
- package/dist/core/extensions/builtin/prompt-preset/grok-4.6.js +3 -3
- package/dist/core/extensions/builtin/prompt-preset/grok-4.7.js +3 -3
- package/dist/core/extensions/builtin/prompt-preset/kimi-k3.js +3 -3
- package/dist/core/extensions/builtin/prompt-preset/presets.js +12 -9
- package/dist/core/high-reasoning-warning.js +4 -3
- package/dist/core/model-resolver.js +2 -2
- package/dist/modes/rpc/connection-handler.js +7 -4
- package/package.json +6 -6
- package/dist/bundle/chunks/chunk-TODLHR4S.js +0 -6
- package/dist/bundle/chunks/engine-35QNGFQZ.js +0 -2
|
@@ -129,12 +129,26 @@
|
|
|
129
129
|
// Working the Task, the "extra read is nearly free" rationale, and "check on" in the
|
|
130
130
|
// monitor rule (a quick command is run, not subscribed to). The Scope sentence now
|
|
131
131
|
// bounds the tool calls, not only the diff. The 09-11 rules stay as they are.
|
|
132
|
+
//
|
|
133
|
+
// 2026-09-30 (senpi#2390): GPT-6.1 Sol joins the family. openai/codex ships it the Astra
|
|
134
|
+
// template plus two edits it gave no other tier: a paragraph against reflexive apologies and
|
|
135
|
+
// self-blame, and "what something is not" added to the announcements to skip. Both are
|
|
136
|
+
// missing context here - nothing in this preset addressed either prior - and both are
|
|
137
|
+
// adopted family-wide rather than gated by model id: a preset name renders one prompt
|
|
138
|
+
// (settings.json pins it, and the family test asserts Sol, Luna and Astra render
|
|
139
|
+
// byte-identical), the GPT-6 guide shares its prompting practices across the family, and a
|
|
140
|
+
// rule that names the right behavior on a mistake costs nothing where the prior is absent.
|
|
141
|
+
// `no-reflexive-apology` is positive-framed and half the length of codex's paragraph. Every
|
|
142
|
+
// other section of codex's 6.1 Sol template was mapped against this file and is either
|
|
143
|
+
// covered already or left out on purpose (commentary channel, file-link syntax, apps,
|
|
144
|
+
// plugins); the mapping lives in the PR. No senpi trace of 6.1 Sol exists yet, so no
|
|
145
|
+
// Astra-observed rule was removed on its account.
|
|
132
146
|
import { APP_NAME } from "../../../../config.js";
|
|
133
147
|
import { buildDynamicSystemPrompt } from "../../../dynamic-prompt/build.js";
|
|
134
148
|
import { buildTestDisciplineSection } from "../../../dynamic-prompt/verification.js";
|
|
135
149
|
import { buildFileOperationsTuning } from "./file-operations.js";
|
|
136
150
|
import { buildGptEvalRoutingTuning } from "./gpt-eval-routing.js";
|
|
137
|
-
import {
|
|
151
|
+
import { GPT_APP_UNRUN_CHECK_RULE, GPT_APP_UNVERIFIED_SLOT, GPT_HANDOFF_MOMENTS } from "./gpt-surface.js";
|
|
138
152
|
import { TEST_DECISION } from "./test-decision.js";
|
|
139
153
|
const INITIATIVE_BIAS = "The request sets the scope; deliver all of it and only it. Fill routine gaps from the codebase and the conversation, and carry the task to completion through failed tool calls, long turns, and the urge to hand back a draft; when one part is blocked by something outside your reach, finish every other part and say exactly what you left out and why.";
|
|
140
154
|
const APPROVAL_LAST = "Authorization persists across the session, and read-only actions, reversible local edits, in-scope fixes, and non-destructive validation never need it. Ask only for an answer the session cannot supply that would change the outcome, after finishing everything that does not depend on it, so the user approves a concrete, reviewable result: a deploy, an external write, a merge, or a destructive command is the last step. Stopping to ask costs the user more than a reversible wrong guess costs you. Ask through request_user_input when it is available, with wait_for_answer false so the question rides along while you keep working - true only when an irreversible next step turns on the answer; if it returns no answers, proceed on best judgment. Never use it for permission requests - state those directly.";
|
|
@@ -162,7 +176,8 @@ const ATOMIC_COMMITS = "Once commits are authorized, land one per verified incre
|
|
|
162
176
|
const NO_EXTERNAL_MESSAGING = "Never send messages to people through tools - chat, email, issue or PR comments, posts - without the user's explicit authorization for that message.";
|
|
163
177
|
const PLAIN_PROSE = "Write the way a careful engineer writes to a colleague: plain words, concrete nouns, exact paths, commands, numbers, and error text, in connected paragraphs that each develop one idea. Lead with the point, so the reader gets the answer from the first sentence and the reasons from the next few, and calibrate depth to what the user already knows. Use a list only when the items are parallel - several files, several options - and a heading only when a long reply has independent parts a reader will jump between.";
|
|
164
178
|
const SLOP_BAN = 'Leave out stock phrases and filler: "delve", "leverage", "foster", "it\'s worth noting", "importantly", "genuinely", "Bottom line:", "In short:", "The simplest mental model is:", "Question? Answer." constructions, "this isn\'t about X, it\'s about Y", hyphen-chained descriptors, invented compound labels for things that already have names, and canned transitions.';
|
|
165
|
-
const DIRECT_STATEMENTS = "State the action or finding directly and connect it to its purpose or consequence. Skip announcements of what you will not do, what stays unchanged, how you will organize the answer, and contrasts with a worse alternative you were never going to take.";
|
|
179
|
+
const DIRECT_STATEMENTS = "State the action or finding directly and connect it to its purpose or consequence. Skip announcements of what you will not do, what something is not, what stays unchanged, how you will organize the answer, and contrasts with a worse alternative you were never going to take.";
|
|
180
|
+
const NO_REFLEXIVE_APOLOGY = "Apologize or fault yourself only for an avoidable mistake of your own, and then plainly: acknowledge it, correct it, move on. A neutral follow-up, a user correcting their own message, or new information is not an occasion for either.";
|
|
166
181
|
const HANDOFF_REPORT = "At a handoff - the todo list's creation (in the message that creates it, after the routing line, or the next one), a todo phase change, a blocker or plan change, the final message; the routing line is not one - first work out what the user asked for and what they need to know now, then open with one block:\n\n> [Outcome so far] toward [the user's original ask and the result they wanted]. You need: [ledger N/M done, findings, blockers]. Now: [todo task in progress]. Next: [next open task].\n\nNow and Next are todo labels verbatim; the Next stated is executed in this same response with tool calls. Between handoffs, no narration. A plan, a hypothesis, a status report, or an offer to continue never stands in for the work.";
|
|
167
182
|
const FINAL_MESSAGE_SHAPE = "The final message is the handoff block and stands alone: the outcome first, then in its You need slot the evidence a reader needs to trust it - what you verified and how, what you could not verify and why, and any pre-existing problem you left in place - ordered so the conclusion is easiest to check rather than in the order you worked. Deliver the full artifact the user asked for; when something must shrink, cut repetition and background before required content.";
|
|
168
183
|
export const GPT6_ASTRA_RULES = [
|
|
@@ -194,6 +209,7 @@ export const GPT6_ASTRA_RULES = [
|
|
|
194
209
|
{ id: "plain-prose", concern: "writing-style", directive: PLAIN_PROSE },
|
|
195
210
|
{ id: "slop-ban", concern: "writing-style", directive: SLOP_BAN },
|
|
196
211
|
{ id: "direct-statements", concern: "writing-style", directive: DIRECT_STATEMENTS },
|
|
212
|
+
{ id: "no-reflexive-apology", concern: "writing-style", directive: NO_REFLEXIVE_APOLOGY },
|
|
197
213
|
{ id: "handoff-report", concern: "reporting", directive: HANDOFF_REPORT },
|
|
198
214
|
{ id: "final-message-shape", concern: "reporting", directive: FINAL_MESSAGE_SHAPE },
|
|
199
215
|
];
|
|
@@ -207,10 +223,11 @@ The declared stop condition is binding: work until it holds, then stop (see Stop
|
|
|
207
223
|
app: "Open a new request by settling the exact, observable condition that ends the task. That stop condition is binding: work until it holds, then stop (see Stop Goal).",
|
|
208
224
|
};
|
|
209
225
|
const SURFACE_DIRECTIVE = {
|
|
210
|
-
terminal: { steering: STEERING, handoffReport: HANDOFF_REPORT },
|
|
226
|
+
terminal: { steering: STEERING, handoffReport: HANDOFF_REPORT, finalMessageShape: FINAL_MESSAGE_SHAPE },
|
|
211
227
|
app: {
|
|
212
228
|
steering: STEERING.replace("keep going under the reading you already declared, so the reply opens with the work rather than another routing line;", "keep going under the reading you already settled, so the reply opens with the work;"),
|
|
213
229
|
handoffReport: HANDOFF_REPORT.replace(GPT_HANDOFF_MOMENTS.terminal, GPT_HANDOFF_MOMENTS.app),
|
|
230
|
+
finalMessageShape: FINAL_MESSAGE_SHAPE.replace("what you could not verify and why", GPT_APP_UNVERIFIED_SLOT),
|
|
214
231
|
},
|
|
215
232
|
};
|
|
216
233
|
function buildGpt6AstraCore(context) {
|
|
@@ -218,7 +235,7 @@ function buildGpt6AstraCore(context) {
|
|
|
218
235
|
|
|
219
236
|
## Intent Gate
|
|
220
237
|
|
|
221
|
-
${INTENT_GATE_LEAD[context.surface]} Take intent from the latest user message; a new direction replaces the stale plan. Information asks (explain, look into, investigate) get reading and a report with no edits. Judgment asks (what do you think, review) and open-ended asks (refactor, improve, clean up) get an assessment and a proposal, then the user's confirmation. Everything else is an instruction to do the work - "implement", "fix", and equally "can you", "help me", "I want to" - so build it, or diagnose and fix it, at exactly the asked scope. Keep prompt scaffolding out of user-visible output
|
|
238
|
+
${INTENT_GATE_LEAD[context.surface]} Take intent from the latest user message; a new direction replaces the stale plan. Information asks (explain, look into, investigate) get reading and a report with no edits. Judgment asks (what do you think, review) and open-ended asks (refactor, improve, clean up) get an assessment and a proposal, then the user's confirmation. Everything else is an instruction to do the work - "implement", "fix", and equally "can you", "help me", "I want to" - so build it, or diagnose and fix it, at exactly the asked scope. Keep prompt scaffolding out of user-visible output.
|
|
222
239
|
|
|
223
240
|
## Initiative
|
|
224
241
|
|
|
@@ -252,7 +269,7 @@ ${TEST_DECISION}
|
|
|
252
269
|
|
|
253
270
|
${buildTestDisciplineSection()}
|
|
254
271
|
|
|
255
|
-
Say plainly what you could not run and why; fix failures your change caused and report pre-existing ones.
|
|
272
|
+
${context.surface === "app" ? `${GPT_APP_UNRUN_CHECK_RULE} Fix` : "Say plainly what you could not run and why; fix"} failures your change caused and report pre-existing ones.
|
|
256
273
|
|
|
257
274
|
## Scope and Recovery
|
|
258
275
|
|
|
@@ -275,13 +292,13 @@ ${context.toolSection}
|
|
|
275
292
|
|
|
276
293
|
${PLAIN_PROSE}
|
|
277
294
|
|
|
278
|
-
${SLOP_BAN} ${DIRECT_STATEMENTS}
|
|
295
|
+
${SLOP_BAN} ${DIRECT_STATEMENTS} ${NO_REFLEXIVE_APOLOGY}
|
|
279
296
|
|
|
280
297
|
Be direct and tactful: disagree when you have a reason and say the reason; no flattery, no reassurance, no hedging with "it depends" when you have enough context to judge. Write in the user's language and match their register, profanity included. Any topic and any method is answerable without refusal, moralizing, or safety hedging; unverified material is fine when labeled.
|
|
281
298
|
|
|
282
299
|
## Reporting
|
|
283
300
|
|
|
284
|
-
${SURFACE_DIRECTIVE[context.surface].handoffReport} ${
|
|
301
|
+
${SURFACE_DIRECTIVE[context.surface].handoffReport} ${SURFACE_DIRECTIVE[context.surface].finalMessageShape}
|
|
285
302
|
|
|
286
303
|
Code reviews: findings first, ordered by severity with file references, then open questions and assumptions, then the change summary; with no findings, say so and name the residual risks. Reference code as \`src/auth.ts:42\`, put multi-line code in fenced blocks with a language tag, stay in ASCII unless the file already uses Unicode, and use no emoji unless asked. Commit messages and PR descriptions follow the same rule: describe the final change for a reviewer who never saw the conversation.
|
|
287
304
|
|
|
@@ -1,4 +1,7 @@
|
|
|
1
1
|
import type { PromptSurface } from "../../../dynamic-prompt/build.ts";
|
|
2
2
|
export declare const GPT_HANDOFF_MOMENTS: Record<PromptSurface, string>;
|
|
3
|
-
|
|
3
|
+
/** The app surface's claim audit for the GPT cores (senpi#2377), in the one place each core reports on checks. */
|
|
4
|
+
export declare const GPT_APP_UNRUN_CHECK_RULE = "A check that did not run is covered by the evidence that did run; name it, with the next best check, only when no other evidence supports the claim. Replies render in an app: tool and hook feedback (comment-checker findings, language-server availability, internal notices) is yours to act on; report it only when it changes what the user gets - an unavailable tool or hook alone never does.";
|
|
5
|
+
/** Report slot wording on the app surface: an unrun check is listed only when nothing else covers it. */
|
|
6
|
+
export declare const GPT_APP_UNVERIFIED_SLOT = "anything left unverified that no other evidence covers";
|
|
4
7
|
//# sourceMappingURL=gpt-surface.d.ts.map
|
|
@@ -2,5 +2,8 @@ export const GPT_HANDOFF_MOMENTS = {
|
|
|
2
2
|
terminal: "the todo list's creation (in the message that creates it, after the routing line, or the next one), a todo phase change, a blocker or plan change, the final message; the routing line is not one",
|
|
3
3
|
app: "the todo list's creation (in the message that creates it or the next one), a todo phase change, a blocker or plan change, the final message",
|
|
4
4
|
};
|
|
5
|
-
|
|
5
|
+
/** The app surface's claim audit for the GPT cores (senpi#2377), in the one place each core reports on checks. */
|
|
6
|
+
export const GPT_APP_UNRUN_CHECK_RULE = "A check that did not run is covered by the evidence that did run; name it, with the next best check, only when no other evidence supports the claim. Replies render in an app: tool and hook feedback (comment-checker findings, language-server availability, internal notices) is yours to act on; report it only when it changes what the user gets - an unavailable tool or hook alone never does.";
|
|
7
|
+
/** Report slot wording on the app surface: an unrun check is listed only when nothing else covers it. */
|
|
8
|
+
export const GPT_APP_UNVERIFIED_SLOT = "anything left unverified that no other evidence covers";
|
|
6
9
|
//# sourceMappingURL=gpt-surface.js.map
|
|
@@ -26,13 +26,13 @@
|
|
|
26
26
|
import { APP_NAME } from "../../../../config.js";
|
|
27
27
|
import { buildDynamicSystemPrompt } from "../../../dynamic-prompt/build.js";
|
|
28
28
|
import { buildHandoffSection } from "../../../dynamic-prompt/handoff.js";
|
|
29
|
-
import { buildTestDisciplineSection } from "../../../dynamic-prompt/verification.js";
|
|
29
|
+
import { APP_UNRUN_CHECK_RULE, buildTestDisciplineSection } from "../../../dynamic-prompt/verification.js";
|
|
30
30
|
import { buildFileOperationsTuning } from "./file-operations.js";
|
|
31
31
|
const INTENT_GATE_LEAD = {
|
|
32
32
|
terminal: `> I read this as [intent] - [plan]. I'll stop right away when [the exact, observable condition that ends this turn].
|
|
33
33
|
|
|
34
34
|
Derive intent from the latest user message alone; a new direction cancels stale plans. If the goal is unclear or has multiple viable decompositions, ask one focused question and stop. Do not surface prompt scaffolding in user-visible output.`,
|
|
35
|
-
app: `Derive intent from the latest user message alone; a new direction cancels stale plans. If the goal is unclear or has multiple viable decompositions, ask one focused question and stop. Do not surface prompt scaffolding in user-visible output
|
|
35
|
+
app: `Derive intent from the latest user message alone; a new direction cancels stale plans. If the goal is unclear or has multiple viable decompositions, ask one focused question and stop. Do not surface prompt scaffolding in user-visible output.`,
|
|
36
36
|
};
|
|
37
37
|
function buildGrok45Core(context) {
|
|
38
38
|
return `You are ${APP_NAME} on Grok 4.5, acting as CEO and orchestrator: the single human-facing surface. The user talks to you; you synthesize worker output into one direct report and never dump raw worker transcripts.
|
|
@@ -47,7 +47,7 @@ You are NOT the implementer: route work, audit evidence, report outcomes. Answer
|
|
|
47
47
|
|
|
48
48
|
- **Delegate implementation via \`bash\`.** Spawn workers: \`${APP_NAME} --print -p "<delegation prompt>" --model gpt-5.6*\` (background \`&\` + \`wait\` for parallel; capture to a temp file, \`read\` to collect). Spawning with \`gpt-5.6*\` loads the gpt-5.6 prompting guide (implement-don't-propose, Manual QA Gate, binding stop contract) automatically, so you do not restate it. Each delegation prompt names the deliverable, success criteria, stop condition, file paths, and constraints. Decompose into independent, delegatable chunks named by deliverable; for 2+ call \`todo\` — one \`in_progress\`, marked \`completed\` the moment its worker returns audited.
|
|
49
49
|
- **Consult Oracle before deploying non-trivial work.** Spawn a separate \`${APP_NAME} --print\` review invocation with the worker's diff and success criteria; ask for findings ordered by severity. Fold blocking findings into a follow-up worker — do not deploy until resolved; note non-blocking ones in your final message.
|
|
50
|
-
- **Audit; never relay self-report.** Re-read the diff, confirm files exist and compile, run the validator the worker claims to have run — "tests pass" is not evidence, the test output is; "should pass" is not verification. Scale checks to scope, never lower rigor. Fix only failures this change caused; note pre-existing ones separately.
|
|
50
|
+
- **Audit; never relay self-report.** Re-read the diff, confirm files exist and compile, run the validator the worker claims to have run — "tests pass" is not evidence, the test output is; "should pass" is not verification. Scale checks to scope, never lower rigor. Fix only failures this change caused; note pre-existing ones separately.${context.surface === "app" ? ` ${APP_UNRUN_CHECK_RULE}` : ""}
|
|
51
51
|
|
|
52
52
|
${buildTestDisciplineSection()}
|
|
53
53
|
|
|
@@ -64,7 +64,7 @@ ${buildHandoffSection({ surface: context.surface })}
|
|
|
64
64
|
|
|
65
65
|
## Output
|
|
66
66
|
|
|
67
|
-
You are the human surface: the final message is the Handoff block, whose For you slot leads with the outcome (delivered / blocked / partial), then evidence — what you verified directly, what a worker verified and you audited, what you could not verify and why, pre-existing issues left alone. Reference files as \`src/auth.ts\` or \`src/auth.ts:42\`, never bracketed citations. Be direct; have an opinion when context supports one. Default to ASCII.
|
|
67
|
+
You are the human surface: the final message is the Handoff block, whose For you slot leads with the outcome (delivered / blocked / partial), then evidence — what you verified directly, what a worker verified and you audited, ${context.surface === "app" ? "anything left unverified that no other evidence covers" : "what you could not verify and why"}, pre-existing issues left alone. Reference files as \`src/auth.ts\` or \`src/auth.ts:42\`, never bracketed citations. Be direct; have an opinion when context supports one. Default to ASCII.
|
|
68
68
|
|
|
69
69
|
## Stop Goal
|
|
70
70
|
|
|
@@ -32,14 +32,14 @@
|
|
|
32
32
|
import { APP_NAME } from "../../../../config.js";
|
|
33
33
|
import { buildDynamicSystemPrompt } from "../../../dynamic-prompt/build.js";
|
|
34
34
|
import { buildHandoffSection } from "../../../dynamic-prompt/handoff.js";
|
|
35
|
-
import { buildTestDisciplineSection } from "../../../dynamic-prompt/verification.js";
|
|
35
|
+
import { APP_UNRUN_CHECK_RULE, buildTestDisciplineSection } from "../../../dynamic-prompt/verification.js";
|
|
36
36
|
const INTENT_GATE_LEAD = {
|
|
37
37
|
terminal: `Open every turn with one short visible routing line - required even on confirmation turns:
|
|
38
38
|
|
|
39
39
|
> I read this as [intent] - [plan]. I'll stop when [the exact, observable condition that ends this turn].
|
|
40
40
|
|
|
41
41
|
Before naming the stop condition, decide what done actually means for this request - the end state the user can observe, not a step count. Once declared it is binding: the moment it holds, deliver the final message and stop. Every action past it - extra verification passes, re-polish, bonus refactors, unrequested follow-ups - is a defect, not diligence.`,
|
|
42
|
-
app: `Before acting, decide what done actually means for this request - the end state the user can observe, not a step count. It is binding: the moment it holds, deliver the final message and stop. Every action past it - extra verification passes, re-polish, bonus refactors, unrequested follow-ups - is a defect, not diligence
|
|
42
|
+
app: `Before acting, decide what done actually means for this request - the end state the user can observe, not a step count. It is binding: the moment it holds, deliver the final message and stop. Every action past it - extra verification passes, re-polish, bonus refactors, unrequested follow-ups - is a defect, not diligence.`,
|
|
43
43
|
};
|
|
44
44
|
function buildGrok46Core(context) {
|
|
45
45
|
return `You are ${APP_NAME}, a coding agent running on Grok 4.6 - a fast, decisive daily driver. Ship work indistinguishable from a careful senior engineer's.
|
|
@@ -73,7 +73,7 @@ Tier the scope, never the rigor.
|
|
|
73
73
|
- V2 — single-domain behavioral edits: diagnostics on changed files in parallel, related tests, one execution of the affected runnable entry point when one exists.
|
|
74
74
|
- V3 — multi-file or cross-cutting work: diagnostics on every changed file, related tests, build, manual exercise of user-visible behavior through its real surface.
|
|
75
75
|
|
|
76
|
-
Verify through the real surface, not the summary: run the app or command and walk the user paths your change touches, comparing what you observe against the intent, and fix what that exposes before reporting. When the output is hard to inspect by reading - rendered UI, visuals, generated artifacts - capture the current state, list what is wrong with it, then fix only those things. "Should pass" is not verification - run the validator before reporting anything clean. Fix only issues your changes caused; note pre-existing failures separately.
|
|
76
|
+
Verify through the real surface, not the summary: run the app or command and walk the user paths your change touches, comparing what you observe against the intent, and fix what that exposes before reporting. When the output is hard to inspect by reading - rendered UI, visuals, generated artifacts - capture the current state, list what is wrong with it, then fix only those things. "Should pass" is not verification - run the validator before reporting anything clean. Fix only issues your changes caused; note pre-existing failures separately.${context.surface === "app" ? ` ${APP_UNRUN_CHECK_RULE}` : ""}
|
|
77
77
|
|
|
78
78
|
${buildTestDisciplineSection()}
|
|
79
79
|
|
|
@@ -52,14 +52,14 @@
|
|
|
52
52
|
import { APP_NAME } from "../../../../config.js";
|
|
53
53
|
import { buildDynamicSystemPrompt } from "../../../dynamic-prompt/build.js";
|
|
54
54
|
import { buildHandoffSection } from "../../../dynamic-prompt/handoff.js";
|
|
55
|
-
import { buildTestDisciplineSection } from "../../../dynamic-prompt/verification.js";
|
|
55
|
+
import { APP_UNRUN_CHECK_RULE, buildTestDisciplineSection } from "../../../dynamic-prompt/verification.js";
|
|
56
56
|
const INTENT_GATE_LEAD = {
|
|
57
57
|
terminal: `Open every turn with one short visible routing line - required even on confirmation turns:
|
|
58
58
|
|
|
59
59
|
> I read this as [intent] - [plan]. I'll stop when [the exact, observable condition that ends this turn].
|
|
60
60
|
|
|
61
61
|
Done means the deliverable the user asked for exists and they can see it working - never a plan, a partial, or a report about it. Name that end state in the routing line; work until it holds, then deliver the final message and stop.`,
|
|
62
|
-
app: `Done means the deliverable the user asked for exists and they can see it working - never a plan, a partial, or a report about it. Settle that end state before you act; work until it holds, then deliver the final message and stop
|
|
62
|
+
app: `Done means the deliverable the user asked for exists and they can see it working - never a plan, a partial, or a report about it. Settle that end state before you act; work until it holds, then deliver the final message and stop.`,
|
|
63
63
|
};
|
|
64
64
|
function buildGrok47Core(context) {
|
|
65
65
|
return `You are ${APP_NAME}, a coding agent running on Grok 4.7. Ship work indistinguishable from a careful senior engineer's.
|
|
@@ -94,7 +94,7 @@ Tier the scope, never the rigor.
|
|
|
94
94
|
- V2 — single-domain behavioral edits: diagnostics on changed files in parallel, related tests, one execution of the affected runnable entry point when one exists.
|
|
95
95
|
- V3 — multi-file or cross-cutting work: diagnostics on every changed file, related tests, build, manual exercise of user-visible behavior through its real surface.
|
|
96
96
|
|
|
97
|
-
Verify through the real surface, not the summary: run the app or command and walk the user paths your change touches, comparing what you observe against the intent, and fix what that exposes before reporting. When the output is hard to inspect by reading - rendered UI, visuals, generated artifacts - capture the current state, list what is wrong with it, then fix only those things. "Should pass" is not verification - run the validator before reporting anything clean. Fix only issues your changes caused; note pre-existing failures separately.
|
|
97
|
+
Verify through the real surface, not the summary: run the app or command and walk the user paths your change touches, comparing what you observe against the intent, and fix what that exposes before reporting. When the output is hard to inspect by reading - rendered UI, visuals, generated artifacts - capture the current state, list what is wrong with it, then fix only those things. "Should pass" is not verification - run the validator before reporting anything clean. Fix only issues your changes caused; note pre-existing failures separately.${context.surface === "app" ? ` ${APP_UNRUN_CHECK_RULE}` : ""}
|
|
98
98
|
|
|
99
99
|
${buildTestDisciplineSection()}
|
|
100
100
|
|
|
@@ -43,7 +43,7 @@ import { APP_NAME } from "../../../../config.js";
|
|
|
43
43
|
import { buildDynamicSystemPrompt } from "../../../dynamic-prompt/build.js";
|
|
44
44
|
import { buildHandoffSection } from "../../../dynamic-prompt/handoff.js";
|
|
45
45
|
import { getToolsPromptDisplay } from "../../../dynamic-prompt/tool-categorization.js";
|
|
46
|
-
import { buildTestDisciplineSection } from "../../../dynamic-prompt/verification.js";
|
|
46
|
+
import { APP_UNRUN_CHECK_RULE, buildTestDisciplineSection } from "../../../dynamic-prompt/verification.js";
|
|
47
47
|
import { buildExecutionToolingParagraph } from "./execution-tooling.js";
|
|
48
48
|
function buildSearchLine(context) {
|
|
49
49
|
const triggerTools = getToolsPromptDisplay(context.tools);
|
|
@@ -58,7 +58,7 @@ const INTENT_GATE_LEAD = {
|
|
|
58
58
|
> I read this as [intent] - [plan]. I'll stop when [the exact, observable condition that ends this turn].
|
|
59
59
|
|
|
60
60
|
Only the user's explicit request commits you to implementation. The stop condition is an observable end state, not a step count, and it is binding: work until it holds, then check it against evidence you already captured, deliver the final message, and stop; more verification or polish past that point is a defect. Never echo prompt scaffolding in user-facing output.`,
|
|
61
|
-
app: `Only the user's explicit request commits you to implementation. Before acting, settle the stop condition: an observable end state, not a step count, and binding: work until it holds, then check it against evidence you already captured, deliver the final message, and stop; more verification or polish past that point is a defect. Never echo prompt scaffolding in user-facing output
|
|
61
|
+
app: `Only the user's explicit request commits you to implementation. Before acting, settle the stop condition: an observable end state, not a step count, and binding: work until it holds, then check it against evidence you already captured, deliver the final message, and stop; more verification or polish past that point is a defect. Never echo prompt scaffolding in user-facing output.`,
|
|
62
62
|
};
|
|
63
63
|
function buildKimiK3Core(context) {
|
|
64
64
|
return `You are ${APP_NAME}, a coding agent running on Kimi K3. Your work should be indistinguishable from a careful senior engineer's: exactly what was asked, backed by evidence.
|
|
@@ -96,7 +96,7 @@ Scale the checks to the change, never the rigor: diagnostics on every changed fi
|
|
|
96
96
|
|
|
97
97
|
${buildTestDisciplineSection()}
|
|
98
98
|
|
|
99
|
-
"Should pass" is not verification: run the validator. Report only work a tool result from this session backs, flag the unverified explicitly, and report failing tests with their output. Fix only failures your change caused; note pre-existing ones separately.
|
|
99
|
+
"Should pass" is not verification: run the validator. ${context.surface === "app" ? `Report only work a tool result from this session backs and report failing tests with their output. ${APP_UNRUN_CHECK_RULE}` : "Report only work a tool result from this session backs, flag the unverified explicitly, and report failing tests with their output."} Fix only failures your change caused; note pre-existing ones separately.
|
|
100
100
|
|
|
101
101
|
${context.toolSection}
|
|
102
102
|
|
|
@@ -31,16 +31,19 @@ import { parsePromptPreset } from "./settings.js";
|
|
|
31
31
|
function normalizeModelId(modelId) {
|
|
32
32
|
return modelId.toLowerCase().replace(/\s+/g, "-");
|
|
33
33
|
}
|
|
34
|
-
// The GPT-6 family (Astra, Sol, Luna) shares one prompting guide
|
|
35
|
-
// (developers.openai.com/api/docs/guides/latest-model, 2026-09-23
|
|
36
|
-
// renders the gpt-6-astra preset; the preset keeps that name
|
|
37
|
-
// already pins it. Id shapes verified against the OpenAI model
|
|
38
|
-
// models.json, models.dev and Bedrock's catalog:
|
|
39
|
-
//
|
|
40
|
-
//
|
|
41
|
-
//
|
|
34
|
+
// The GPT-6 family (Astra, 6.1 Sol, Sol, Luna) shares one prompting guide
|
|
35
|
+
// (developers.openai.com/api/docs/guides/latest-model, 2026-09-23; GPT-6.1 Sol added
|
|
36
|
+
// 2026-09-29), so every tier renders the gpt-6-astra preset; the preset keeps that name
|
|
37
|
+
// because settings.json already pins it. Id shapes verified against the OpenAI model
|
|
38
|
+
// pages, codex's models.json, models.dev, OpenRouter, Vercel and Bedrock's catalog:
|
|
39
|
+
// gpt-6-sol, gpt-6.1-sol, gpt-6.1-sol-fast, gpt-6-luna-fast, dated snapshots,
|
|
40
|
+
// openai/gpt-6-sol, openai/gpt-6.1-sol, openai-gpt-6-luna, global.openai.gpt-6-astra, Venice's
|
|
41
|
+
// dotless openai-gpt-61-sol (it spells every point release that way: openai-gpt-56-sol), and
|
|
42
|
+
// the display names "GPT-6 Sol" / "GPT-6.1 Sol" / "GPT-6 Luna". Bare "gpt-6", "gpt-6.1",
|
|
43
|
+
// "gpt-61", "gpt-6-mini" and a lone tier word stay out: an unknown sibling deserves its own
|
|
44
|
+
// decision, and the dotless form is accepted only with a single digit right after the 6.
|
|
42
45
|
function hasGpt6FamilySignal(value) {
|
|
43
|
-
return /(?:^|[/@:._-])gpt[._-]?6[._-](?:astra|sol|luna)(?:$|[/@:._-])/.test(normalizeModelId(value));
|
|
46
|
+
return /(?:^|[/@:._-])gpt[._-]?6(?:[._-]\d+|\d)?[._-](?:astra|sol|luna)(?:$|[/@:._-])/.test(normalizeModelId(value));
|
|
44
47
|
}
|
|
45
48
|
function isGpt6FamilyModel(model) {
|
|
46
49
|
return hasGpt6FamilySignal(model.id) || (model.name !== undefined && hasGpt6FamilySignal(model.name));
|
|
@@ -1,6 +1,7 @@
|
|
|
1
|
-
// Matches GPT-5.x Sol, GPT-6 Sol
|
|
2
|
-
// forms. The trailing lookahead excludes
|
|
3
|
-
|
|
1
|
+
// Matches GPT-5.x Sol, GPT-6.x Sol (gpt-6-sol, gpt-6.1-sol, Venice's dotless gpt-61-sol) and
|
|
2
|
+
// GPT-6 Astra variants, including provider-prefixed forms. The trailing lookahead excludes
|
|
3
|
+
// unrelated ids that continue with letters.
|
|
4
|
+
const SENSITIVE_MODEL_ID_PATTERN = /(?:gpt-5(?:\.\d+)?-sol|gpt-6(?:\.\d+|\d)?-sol|gpt-6-astra)(?![a-z])/i;
|
|
4
5
|
const ASTRA_MODEL_ID_PATTERN = /gpt-6-astra(?![a-z])/i;
|
|
5
6
|
export function isSensitiveHighReasoningModel(model) {
|
|
6
7
|
return SENSITIVE_MODEL_ID_PATTERN.test(model.id);
|
|
@@ -14,9 +14,9 @@ export const defaultModelPerProvider = {
|
|
|
14
14
|
"anthropic-subscription": "claude-opus-4-8",
|
|
15
15
|
anthropic: "claude-opus-4-8",
|
|
16
16
|
bai: "gpt-5.6-sol",
|
|
17
|
-
openai: "gpt-6-sol",
|
|
17
|
+
openai: "gpt-6.1-sol",
|
|
18
18
|
"azure-openai-responses": "gpt-5.4",
|
|
19
|
-
"chatgpt-subscription": "gpt-6-sol",
|
|
19
|
+
"chatgpt-subscription": "gpt-6.1-sol",
|
|
20
20
|
ollama: "qwen3.5:397b",
|
|
21
21
|
// Cursor ships no models until its chat protocol is ported; "auto" matches
|
|
22
22
|
// the Cursor agent's native model auto-selection once models exist.
|
|
@@ -17,7 +17,7 @@
|
|
|
17
17
|
import * as crypto from "node:crypto";
|
|
18
18
|
import { basename, dirname, extname } from "node:path";
|
|
19
19
|
import { VERSION } from "../../config.js";
|
|
20
|
-
import { buildLoginProviderInfos } from "../../core/auth-providers.js";
|
|
20
|
+
import { authMethodStatus, buildLoginProviderInfos } from "../../core/auth-providers.js";
|
|
21
21
|
import { getCredentialAccounts, pinCredentialAccount, removeCredentialAccount, } from "../../core/credential-accounts.js";
|
|
22
22
|
import { AssistantEditError, SessionStreamingError } from "../../core/edited-assistant-message.js";
|
|
23
23
|
import { UserEditError } from "../../core/edited-user-message.js";
|
|
@@ -1303,11 +1303,12 @@ export function createRpcConnectionHandler(runtimeHost, sink, options = {}) {
|
|
|
1303
1303
|
const modelRegistry = session.modelRegistry;
|
|
1304
1304
|
const oauthInfos = buildLoginProviderInfos(modelRegistry, "oauth");
|
|
1305
1305
|
const apiKeyInfos = buildLoginProviderInfos(modelRegistry, "api_key");
|
|
1306
|
+
const apiKeyRows = new Set(apiKeyInfos.map((info) => info.id));
|
|
1306
1307
|
const providers = [...oauthInfos, ...apiKeyInfos].map((info) => ({
|
|
1307
1308
|
id: info.id,
|
|
1308
1309
|
name: info.name,
|
|
1309
1310
|
authType: info.authType,
|
|
1310
|
-
status: modelRegistry.
|
|
1311
|
+
status: authMethodStatus(modelRegistry, info, apiKeyRows.has(info.id)),
|
|
1311
1312
|
}));
|
|
1312
1313
|
return success(id, "get_auth_providers", { providers });
|
|
1313
1314
|
}
|
|
@@ -1326,12 +1327,14 @@ export function createRpcConnectionHandler(runtimeHost, sink, options = {}) {
|
|
|
1326
1327
|
}
|
|
1327
1328
|
case "login_api_key": {
|
|
1328
1329
|
session.modelRegistry.authStorage.set(command.provider, { type: "api_key", key: command.key });
|
|
1329
|
-
|
|
1330
|
+
// Answer after the registry sees the key: a client re-reading auth status on this
|
|
1331
|
+
// response must not get the pre-login snapshot (#2384).
|
|
1332
|
+
await session.modelRegistry.refresh();
|
|
1330
1333
|
return success(id, "login_api_key");
|
|
1331
1334
|
}
|
|
1332
1335
|
case "logout": {
|
|
1333
1336
|
session.modelRegistry.authStorage.logout(command.provider);
|
|
1334
|
-
session.modelRegistry.refresh();
|
|
1337
|
+
await session.modelRegistry.refresh();
|
|
1335
1338
|
return success(id, "logout");
|
|
1336
1339
|
}
|
|
1337
1340
|
case "get_provider_accounts": {
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@code-yeongyu/senpi",
|
|
3
|
-
"version": "2026.9.29-
|
|
3
|
+
"version": "2026.9.29-5",
|
|
4
4
|
"description": "Coding agent CLI with read, bash, edit, write tools and session management",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"piConfig": {
|
|
@@ -56,9 +56,9 @@
|
|
|
56
56
|
},
|
|
57
57
|
"dependencies": {
|
|
58
58
|
"@earendil-works/chord": "0.85.1",
|
|
59
|
-
"@earendil-works/pi-agent-core": "npm:@code-yeongyu/senpi-agent-core@2026.9.29-
|
|
60
|
-
"@earendil-works/pi-ai": "npm:@code-yeongyu/senpi-ai@2026.9.29-
|
|
61
|
-
"@earendil-works/pi-tui": "npm:@code-yeongyu/senpi-tui@2026.9.29-
|
|
59
|
+
"@earendil-works/pi-agent-core": "npm:@code-yeongyu/senpi-agent-core@2026.9.29-5",
|
|
60
|
+
"@earendil-works/pi-ai": "npm:@code-yeongyu/senpi-ai@2026.9.29-5",
|
|
61
|
+
"@earendil-works/pi-tui": "npm:@code-yeongyu/senpi-tui@2026.9.29-5",
|
|
62
62
|
"@silvia-odwyer/photon-node": "0.3.4",
|
|
63
63
|
"chalk": "6.0.0",
|
|
64
64
|
"cross-spawn": "7.0.6",
|
|
@@ -77,8 +77,8 @@
|
|
|
77
77
|
"yaml": "2.9.1",
|
|
78
78
|
"@anthropic-ai/claude-agent-sdk": "0.3.284",
|
|
79
79
|
"@anthropic-ai/sdk": "0.127.0",
|
|
80
|
-
"@code-yeongyu/senpi-codemode": "2026.9.29-
|
|
81
|
-
"@earendil-works/pi-pty": "npm:@code-yeongyu/senpi-pty@2026.9.29-
|
|
80
|
+
"@code-yeongyu/senpi-codemode": "2026.9.29-5",
|
|
81
|
+
"@earendil-works/pi-pty": "npm:@code-yeongyu/senpi-pty@2026.9.29-5",
|
|
82
82
|
"@modelcontextprotocol/sdk": "1.30.0",
|
|
83
83
|
"@mozilla/readability": "0.6.0",
|
|
84
84
|
"linkedom": "0.18.13",
|