@code-yeongyu/senpi 2026.9.29-4 → 2026.9.29-5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. package/CHANGELOG.md +28 -7
  2. package/dist/bundle/chunks/{anthropic-messages-DZAH73IA.js → anthropic-messages-HEVKB3PL.js} +1 -1
  3. package/dist/bundle/chunks/{app-server-command-BUIGVCL4.js → app-server-command-32AFWZEH.js} +1 -1
  4. package/dist/bundle/chunks/{azure-openai-responses-ZJGVMXRJ.js → azure-openai-responses-QVPPCPVL.js} +1 -1
  5. package/dist/bundle/chunks/{chunk-U5LXY7UD.js → chunk-2IHY6ZLW.js} +1 -1
  6. package/dist/bundle/chunks/{chunk-ESPYV6XG.js → chunk-2N3BZ5DV.js} +1 -1
  7. package/dist/bundle/chunks/{chunk-XNR637JQ.js → chunk-2WAKOPNA.js} +1 -1
  8. package/dist/bundle/chunks/{chunk-SHIPI3FN.js → chunk-63QHOG5A.js} +44 -44
  9. package/dist/bundle/chunks/{chunk-H2U7247O.js → chunk-77AUUH3Z.js} +2 -2
  10. package/dist/bundle/chunks/{chunk-T4F7KOR2.js → chunk-7CQX3QFV.js} +1 -1
  11. package/dist/bundle/chunks/{chunk-WW3YFGEP.js → chunk-7MUTNWPO.js} +1 -1
  12. package/dist/bundle/chunks/{chunk-6NXC765G.js → chunk-AWEAVBCR.js} +1 -1
  13. package/dist/bundle/chunks/{chunk-3DISE3EE.js → chunk-BBNWT72C.js} +1 -1
  14. package/dist/bundle/chunks/{chunk-HKUHLCH7.js → chunk-U7BNOE23.js} +1 -1
  15. package/dist/bundle/chunks/{chunk-VAKMXHSM.js → chunk-VLW22GHJ.js} +1 -1
  16. package/dist/bundle/chunks/{chunk-DFRI62DO.js → chunk-W4IEEYWL.js} +1 -1
  17. package/dist/bundle/chunks/{chunk-Q6X7ES7N.js → chunk-XO7VCMY6.js} +1 -1
  18. package/dist/bundle/chunks/chunk-YLQTKWEC.js +6 -0
  19. package/dist/bundle/chunks/{chunk-NUS5JXND.js → chunk-Z2N5372Y.js} +1 -1
  20. package/dist/bundle/chunks/{chunk-JQNR6OYN.js → chunk-ZDNR3YVJ.js} +1 -1
  21. package/dist/bundle/chunks/{chunk-DAKIX3RD.js → chunk-ZPMJIPQC.js} +1 -1
  22. package/dist/bundle/chunks/{cli-main-XCRRH3OR.js → cli-main-ZPMYM3A3.js} +1 -1
  23. package/dist/bundle/chunks/engine-RCHMSSBT.js +2 -0
  24. package/dist/bundle/chunks/github-copilot.js +1 -1
  25. package/dist/bundle/chunks/{google-generative-ai-RYFW7GSR.js → google-generative-ai-E3I6MS7E.js} +1 -1
  26. package/dist/bundle/chunks/{google-vertex-J64D4U2O.js → google-vertex-KFFI24FU.js} +1 -1
  27. package/dist/bundle/chunks/{help-fast-path-JJVINW3Y.js → help-fast-path-ZUNFOE53.js} +1 -1
  28. package/dist/bundle/chunks/{host-command-IZJU2H7Z.js → host-command-V2LDNNXB.js} +1 -1
  29. package/dist/bundle/chunks/{interactive-mode-KXRPZ2U6.js → interactive-mode-J6KGDRR7.js} +1 -1
  30. package/dist/bundle/chunks/{mistral-conversations-C7ZTJYKE.js → mistral-conversations-EZ4XNJXU.js} +1 -1
  31. package/dist/bundle/chunks/{multi-session-host-3XTEXGP4.js → multi-session-host-IBVJ3WN7.js} +1 -1
  32. package/dist/bundle/chunks/{openai-codex-responses-N5SCEWCY.js → openai-codex-responses-TMUCQW5E.js} +1 -1
  33. package/dist/bundle/chunks/{openai-completions-T4B2KZ4B.js → openai-completions-SA2UJW7Q.js} +1 -1
  34. package/dist/bundle/chunks/{openai-responses-N4OK2RIV.js → openai-responses-EHSD4L2M.js} +1 -1
  35. package/dist/bundle/chunks/{package-manager-cli-ZHWO6FT5.js → package-manager-cli-JJ5R7UWV.js} +1 -1
  36. package/dist/bundle/chunks/{rotation-stream-5GE3KAHL.js → rotation-stream-EX7PWQ4S.js} +1 -1
  37. package/dist/bundle/chunks/{rpc-mode-DEVPDAT4.js → rpc-mode-ITND2BVC.js} +1 -1
  38. package/dist/bundle/chunks/{session-control-endpoint-72A2NRVH.js → session-control-endpoint-UJJGG4WI.js} +1 -1
  39. package/dist/bundle/chunks/{session-picker-XYMNQ6AZ.js → session-picker-K5FIKV6C.js} +1 -1
  40. package/dist/bundle/chunks/session-worker.js +42 -42
  41. package/dist/bundle/cli.js +1 -1
  42. package/dist/bundle/index.js +1 -1
  43. package/dist/bundle/rpc-entry.js +1 -1
  44. package/dist/bundle/runtime-manifest.json +1 -1
  45. package/dist/core/auth-providers.d.ts +9 -0
  46. package/dist/core/auth-providers.js +17 -0
  47. package/dist/core/dynamic-prompt/build.js +1 -1
  48. package/dist/core/dynamic-prompt/intent-gate.js +1 -1
  49. package/dist/core/dynamic-prompt/verification.d.ts +9 -1
  50. package/dist/core/dynamic-prompt/verification.js +11 -2
  51. package/dist/core/extensions/builtin/prompt-preset/claude-fable-5-1.js +3 -3
  52. package/dist/core/extensions/builtin/prompt-preset/claude-fable-5.js +3 -3
  53. package/dist/core/extensions/builtin/prompt-preset/claude-opus-5-5.js +3 -3
  54. package/dist/core/extensions/builtin/prompt-preset/claude-opus-5.js +3 -3
  55. package/dist/core/extensions/builtin/prompt-preset/claude-sonnet-5-5.js +3 -3
  56. package/dist/core/extensions/builtin/prompt-preset/gpt-5.5.js +3 -3
  57. package/dist/core/extensions/builtin/prompt-preset/gpt-5.6.js +4 -4
  58. package/dist/core/extensions/builtin/prompt-preset/gpt-6-astra.d.ts +6 -2
  59. package/dist/core/extensions/builtin/prompt-preset/gpt-6-astra.js +24 -7
  60. package/dist/core/extensions/builtin/prompt-preset/gpt-surface.d.ts +4 -1
  61. package/dist/core/extensions/builtin/prompt-preset/gpt-surface.js +4 -1
  62. package/dist/core/extensions/builtin/prompt-preset/grok-4.5.js +4 -4
  63. package/dist/core/extensions/builtin/prompt-preset/grok-4.6.js +3 -3
  64. package/dist/core/extensions/builtin/prompt-preset/grok-4.7.js +3 -3
  65. package/dist/core/extensions/builtin/prompt-preset/kimi-k3.js +3 -3
  66. package/dist/core/extensions/builtin/prompt-preset/presets.js +12 -9
  67. package/dist/core/high-reasoning-warning.js +4 -3
  68. package/dist/core/model-resolver.js +2 -2
  69. package/dist/modes/rpc/connection-handler.js +7 -4
  70. package/package.json +6 -6
  71. package/dist/bundle/chunks/chunk-TODLHR4S.js +0 -6
  72. package/dist/bundle/chunks/engine-35QNGFQZ.js +0 -2
@@ -129,12 +129,26 @@
129
129
  // Working the Task, the "extra read is nearly free" rationale, and "check on" in the
130
130
  // monitor rule (a quick command is run, not subscribed to). The Scope sentence now
131
131
  // bounds the tool calls, not only the diff. The 09-11 rules stay as they are.
132
+ //
133
+ // 2026-09-30 (senpi#2390): GPT-6.1 Sol joins the family. openai/codex ships it the Astra
134
+ // template plus two edits it gave no other tier: a paragraph against reflexive apologies and
135
+ // self-blame, and "what something is not" added to the announcements to skip. Both are
136
+ // missing context here - nothing in this preset addressed either prior - and both are
137
+ // adopted family-wide rather than gated by model id: a preset name renders one prompt
138
+ // (settings.json pins it, and the family test asserts Sol, Luna and Astra render
139
+ // byte-identical), the GPT-6 guide shares its prompting practices across the family, and a
140
+ // rule that names the right behavior on a mistake costs nothing where the prior is absent.
141
+ // `no-reflexive-apology` is positive-framed and half the length of codex's paragraph. Every
142
+ // other section of codex's 6.1 Sol template was mapped against this file and is either
143
+ // covered already or left out on purpose (commentary channel, file-link syntax, apps,
144
+ // plugins); the mapping lives in the PR. No senpi trace of 6.1 Sol exists yet, so no
145
+ // Astra-observed rule was removed on its account.
132
146
  import { APP_NAME } from "../../../../config.js";
133
147
  import { buildDynamicSystemPrompt } from "../../../dynamic-prompt/build.js";
134
148
  import { buildTestDisciplineSection } from "../../../dynamic-prompt/verification.js";
135
149
  import { buildFileOperationsTuning } from "./file-operations.js";
136
150
  import { buildGptEvalRoutingTuning } from "./gpt-eval-routing.js";
137
- import { GPT_APP_FEEDBACK, GPT_HANDOFF_MOMENTS } from "./gpt-surface.js";
151
+ import { GPT_APP_UNRUN_CHECK_RULE, GPT_APP_UNVERIFIED_SLOT, GPT_HANDOFF_MOMENTS } from "./gpt-surface.js";
138
152
  import { TEST_DECISION } from "./test-decision.js";
139
153
  const INITIATIVE_BIAS = "The request sets the scope; deliver all of it and only it. Fill routine gaps from the codebase and the conversation, and carry the task to completion through failed tool calls, long turns, and the urge to hand back a draft; when one part is blocked by something outside your reach, finish every other part and say exactly what you left out and why.";
140
154
  const APPROVAL_LAST = "Authorization persists across the session, and read-only actions, reversible local edits, in-scope fixes, and non-destructive validation never need it. Ask only for an answer the session cannot supply that would change the outcome, after finishing everything that does not depend on it, so the user approves a concrete, reviewable result: a deploy, an external write, a merge, or a destructive command is the last step. Stopping to ask costs the user more than a reversible wrong guess costs you. Ask through request_user_input when it is available, with wait_for_answer false so the question rides along while you keep working - true only when an irreversible next step turns on the answer; if it returns no answers, proceed on best judgment. Never use it for permission requests - state those directly.";
@@ -162,7 +176,8 @@ const ATOMIC_COMMITS = "Once commits are authorized, land one per verified incre
162
176
  const NO_EXTERNAL_MESSAGING = "Never send messages to people through tools - chat, email, issue or PR comments, posts - without the user's explicit authorization for that message.";
163
177
  const PLAIN_PROSE = "Write the way a careful engineer writes to a colleague: plain words, concrete nouns, exact paths, commands, numbers, and error text, in connected paragraphs that each develop one idea. Lead with the point, so the reader gets the answer from the first sentence and the reasons from the next few, and calibrate depth to what the user already knows. Use a list only when the items are parallel - several files, several options - and a heading only when a long reply has independent parts a reader will jump between.";
164
178
  const SLOP_BAN = 'Leave out stock phrases and filler: "delve", "leverage", "foster", "it\'s worth noting", "importantly", "genuinely", "Bottom line:", "In short:", "The simplest mental model is:", "Question? Answer." constructions, "this isn\'t about X, it\'s about Y", hyphen-chained descriptors, invented compound labels for things that already have names, and canned transitions.';
165
- const DIRECT_STATEMENTS = "State the action or finding directly and connect it to its purpose or consequence. Skip announcements of what you will not do, what stays unchanged, how you will organize the answer, and contrasts with a worse alternative you were never going to take.";
179
+ const DIRECT_STATEMENTS = "State the action or finding directly and connect it to its purpose or consequence. Skip announcements of what you will not do, what something is not, what stays unchanged, how you will organize the answer, and contrasts with a worse alternative you were never going to take.";
180
+ const NO_REFLEXIVE_APOLOGY = "Apologize or fault yourself only for an avoidable mistake of your own, and then plainly: acknowledge it, correct it, move on. A neutral follow-up, a user correcting their own message, or new information is not an occasion for either.";
166
181
  const HANDOFF_REPORT = "At a handoff - the todo list's creation (in the message that creates it, after the routing line, or the next one), a todo phase change, a blocker or plan change, the final message; the routing line is not one - first work out what the user asked for and what they need to know now, then open with one block:\n\n> [Outcome so far] toward [the user's original ask and the result they wanted]. You need: [ledger N/M done, findings, blockers]. Now: [todo task in progress]. Next: [next open task].\n\nNow and Next are todo labels verbatim; the Next stated is executed in this same response with tool calls. Between handoffs, no narration. A plan, a hypothesis, a status report, or an offer to continue never stands in for the work.";
167
182
  const FINAL_MESSAGE_SHAPE = "The final message is the handoff block and stands alone: the outcome first, then in its You need slot the evidence a reader needs to trust it - what you verified and how, what you could not verify and why, and any pre-existing problem you left in place - ordered so the conclusion is easiest to check rather than in the order you worked. Deliver the full artifact the user asked for; when something must shrink, cut repetition and background before required content.";
168
183
  export const GPT6_ASTRA_RULES = [
@@ -194,6 +209,7 @@ export const GPT6_ASTRA_RULES = [
194
209
  { id: "plain-prose", concern: "writing-style", directive: PLAIN_PROSE },
195
210
  { id: "slop-ban", concern: "writing-style", directive: SLOP_BAN },
196
211
  { id: "direct-statements", concern: "writing-style", directive: DIRECT_STATEMENTS },
212
+ { id: "no-reflexive-apology", concern: "writing-style", directive: NO_REFLEXIVE_APOLOGY },
197
213
  { id: "handoff-report", concern: "reporting", directive: HANDOFF_REPORT },
198
214
  { id: "final-message-shape", concern: "reporting", directive: FINAL_MESSAGE_SHAPE },
199
215
  ];
@@ -207,10 +223,11 @@ The declared stop condition is binding: work until it holds, then stop (see Stop
207
223
  app: "Open a new request by settling the exact, observable condition that ends the task. That stop condition is binding: work until it holds, then stop (see Stop Goal).",
208
224
  };
209
225
  const SURFACE_DIRECTIVE = {
210
- terminal: { steering: STEERING, handoffReport: HANDOFF_REPORT },
226
+ terminal: { steering: STEERING, handoffReport: HANDOFF_REPORT, finalMessageShape: FINAL_MESSAGE_SHAPE },
211
227
  app: {
212
228
  steering: STEERING.replace("keep going under the reading you already declared, so the reply opens with the work rather than another routing line;", "keep going under the reading you already settled, so the reply opens with the work;"),
213
229
  handoffReport: HANDOFF_REPORT.replace(GPT_HANDOFF_MOMENTS.terminal, GPT_HANDOFF_MOMENTS.app),
230
+ finalMessageShape: FINAL_MESSAGE_SHAPE.replace("what you could not verify and why", GPT_APP_UNVERIFIED_SLOT),
214
231
  },
215
232
  };
216
233
  function buildGpt6AstraCore(context) {
@@ -218,7 +235,7 @@ function buildGpt6AstraCore(context) {
218
235
 
219
236
  ## Intent Gate
220
237
 
221
- ${INTENT_GATE_LEAD[context.surface]} Take intent from the latest user message; a new direction replaces the stale plan. Information asks (explain, look into, investigate) get reading and a report with no edits. Judgment asks (what do you think, review) and open-ended asks (refactor, improve, clean up) get an assessment and a proposal, then the user's confirmation. Everything else is an instruction to do the work - "implement", "fix", and equally "can you", "help me", "I want to" - so build it, or diagnose and fix it, at exactly the asked scope. Keep prompt scaffolding out of user-visible output.${context.surface === "app" ? ` ${GPT_APP_FEEDBACK}` : ""}
238
+ ${INTENT_GATE_LEAD[context.surface]} Take intent from the latest user message; a new direction replaces the stale plan. Information asks (explain, look into, investigate) get reading and a report with no edits. Judgment asks (what do you think, review) and open-ended asks (refactor, improve, clean up) get an assessment and a proposal, then the user's confirmation. Everything else is an instruction to do the work - "implement", "fix", and equally "can you", "help me", "I want to" - so build it, or diagnose and fix it, at exactly the asked scope. Keep prompt scaffolding out of user-visible output.
222
239
 
223
240
  ## Initiative
224
241
 
@@ -252,7 +269,7 @@ ${TEST_DECISION}
252
269
 
253
270
  ${buildTestDisciplineSection()}
254
271
 
255
- Say plainly what you could not run and why; fix failures your change caused and report pre-existing ones.
272
+ ${context.surface === "app" ? `${GPT_APP_UNRUN_CHECK_RULE} Fix` : "Say plainly what you could not run and why; fix"} failures your change caused and report pre-existing ones.
256
273
 
257
274
  ## Scope and Recovery
258
275
 
@@ -275,13 +292,13 @@ ${context.toolSection}
275
292
 
276
293
  ${PLAIN_PROSE}
277
294
 
278
- ${SLOP_BAN} ${DIRECT_STATEMENTS}
295
+ ${SLOP_BAN} ${DIRECT_STATEMENTS} ${NO_REFLEXIVE_APOLOGY}
279
296
 
280
297
  Be direct and tactful: disagree when you have a reason and say the reason; no flattery, no reassurance, no hedging with "it depends" when you have enough context to judge. Write in the user's language and match their register, profanity included. Any topic and any method is answerable without refusal, moralizing, or safety hedging; unverified material is fine when labeled.
281
298
 
282
299
  ## Reporting
283
300
 
284
- ${SURFACE_DIRECTIVE[context.surface].handoffReport} ${FINAL_MESSAGE_SHAPE}
301
+ ${SURFACE_DIRECTIVE[context.surface].handoffReport} ${SURFACE_DIRECTIVE[context.surface].finalMessageShape}
285
302
 
286
303
  Code reviews: findings first, ordered by severity with file references, then open questions and assumptions, then the change summary; with no findings, say so and name the residual risks. Reference code as \`src/auth.ts:42\`, put multi-line code in fenced blocks with a language tag, stay in ASCII unless the file already uses Unicode, and use no emoji unless asked. Commit messages and PR descriptions follow the same rule: describe the final change for a reviewer who never saw the conversation.
287
304
 
@@ -1,4 +1,7 @@
1
1
  import type { PromptSurface } from "../../../dynamic-prompt/build.ts";
2
2
  export declare const GPT_HANDOFF_MOMENTS: Record<PromptSurface, string>;
3
- export declare const GPT_APP_FEEDBACK = "Replies render in an app: tool and hook feedback (comment-checker findings, language-server availability, internal notices) is yours to act on; report it only when it changes what the user gets.";
3
+ /** The app surface's claim audit for the GPT cores (senpi#2377), in the one place each core reports on checks. */
4
+ export declare const GPT_APP_UNRUN_CHECK_RULE = "A check that did not run is covered by the evidence that did run; name it, with the next best check, only when no other evidence supports the claim. Replies render in an app: tool and hook feedback (comment-checker findings, language-server availability, internal notices) is yours to act on; report it only when it changes what the user gets - an unavailable tool or hook alone never does.";
5
+ /** Report slot wording on the app surface: an unrun check is listed only when nothing else covers it. */
6
+ export declare const GPT_APP_UNVERIFIED_SLOT = "anything left unverified that no other evidence covers";
4
7
  //# sourceMappingURL=gpt-surface.d.ts.map
@@ -2,5 +2,8 @@ export const GPT_HANDOFF_MOMENTS = {
2
2
  terminal: "the todo list's creation (in the message that creates it, after the routing line, or the next one), a todo phase change, a blocker or plan change, the final message; the routing line is not one",
3
3
  app: "the todo list's creation (in the message that creates it or the next one), a todo phase change, a blocker or plan change, the final message",
4
4
  };
5
- export const GPT_APP_FEEDBACK = "Replies render in an app: tool and hook feedback (comment-checker findings, language-server availability, internal notices) is yours to act on; report it only when it changes what the user gets.";
5
+ /** The app surface's claim audit for the GPT cores (senpi#2377), in the one place each core reports on checks. */
6
+ export const GPT_APP_UNRUN_CHECK_RULE = "A check that did not run is covered by the evidence that did run; name it, with the next best check, only when no other evidence supports the claim. Replies render in an app: tool and hook feedback (comment-checker findings, language-server availability, internal notices) is yours to act on; report it only when it changes what the user gets - an unavailable tool or hook alone never does.";
7
+ /** Report slot wording on the app surface: an unrun check is listed only when nothing else covers it. */
8
+ export const GPT_APP_UNVERIFIED_SLOT = "anything left unverified that no other evidence covers";
6
9
  //# sourceMappingURL=gpt-surface.js.map
@@ -26,13 +26,13 @@
26
26
  import { APP_NAME } from "../../../../config.js";
27
27
  import { buildDynamicSystemPrompt } from "../../../dynamic-prompt/build.js";
28
28
  import { buildHandoffSection } from "../../../dynamic-prompt/handoff.js";
29
- import { buildTestDisciplineSection } from "../../../dynamic-prompt/verification.js";
29
+ import { APP_UNRUN_CHECK_RULE, buildTestDisciplineSection } from "../../../dynamic-prompt/verification.js";
30
30
  import { buildFileOperationsTuning } from "./file-operations.js";
31
31
  const INTENT_GATE_LEAD = {
32
32
  terminal: `> I read this as [intent] - [plan]. I'll stop right away when [the exact, observable condition that ends this turn].
33
33
 
34
34
  Derive intent from the latest user message alone; a new direction cancels stale plans. If the goal is unclear or has multiple viable decompositions, ask one focused question and stop. Do not surface prompt scaffolding in user-visible output.`,
35
- app: `Derive intent from the latest user message alone; a new direction cancels stale plans. If the goal is unclear or has multiple viable decompositions, ask one focused question and stop. Do not surface prompt scaffolding in user-visible output. Replies render in an app: tool and hook feedback (comment-checker findings, language-server availability, internal notices) is yours to act on; report it only when it changes what the user gets.`,
35
+ app: `Derive intent from the latest user message alone; a new direction cancels stale plans. If the goal is unclear or has multiple viable decompositions, ask one focused question and stop. Do not surface prompt scaffolding in user-visible output.`,
36
36
  };
37
37
  function buildGrok45Core(context) {
38
38
  return `You are ${APP_NAME} on Grok 4.5, acting as CEO and orchestrator: the single human-facing surface. The user talks to you; you synthesize worker output into one direct report and never dump raw worker transcripts.
@@ -47,7 +47,7 @@ You are NOT the implementer: route work, audit evidence, report outcomes. Answer
47
47
 
48
48
  - **Delegate implementation via \`bash\`.** Spawn workers: \`${APP_NAME} --print -p "<delegation prompt>" --model gpt-5.6*\` (background \`&\` + \`wait\` for parallel; capture to a temp file, \`read\` to collect). Spawning with \`gpt-5.6*\` loads the gpt-5.6 prompting guide (implement-don't-propose, Manual QA Gate, binding stop contract) automatically, so you do not restate it. Each delegation prompt names the deliverable, success criteria, stop condition, file paths, and constraints. Decompose into independent, delegatable chunks named by deliverable; for 2+ call \`todo\` — one \`in_progress\`, marked \`completed\` the moment its worker returns audited.
49
49
  - **Consult Oracle before deploying non-trivial work.** Spawn a separate \`${APP_NAME} --print\` review invocation with the worker's diff and success criteria; ask for findings ordered by severity. Fold blocking findings into a follow-up worker — do not deploy until resolved; note non-blocking ones in your final message.
50
- - **Audit; never relay self-report.** Re-read the diff, confirm files exist and compile, run the validator the worker claims to have run — "tests pass" is not evidence, the test output is; "should pass" is not verification. Scale checks to scope, never lower rigor. Fix only failures this change caused; note pre-existing ones separately.
50
+ - **Audit; never relay self-report.** Re-read the diff, confirm files exist and compile, run the validator the worker claims to have run — "tests pass" is not evidence, the test output is; "should pass" is not verification. Scale checks to scope, never lower rigor. Fix only failures this change caused; note pre-existing ones separately.${context.surface === "app" ? ` ${APP_UNRUN_CHECK_RULE}` : ""}
51
51
 
52
52
  ${buildTestDisciplineSection()}
53
53
 
@@ -64,7 +64,7 @@ ${buildHandoffSection({ surface: context.surface })}
64
64
 
65
65
  ## Output
66
66
 
67
- You are the human surface: the final message is the Handoff block, whose For you slot leads with the outcome (delivered / blocked / partial), then evidence — what you verified directly, what a worker verified and you audited, what you could not verify and why, pre-existing issues left alone. Reference files as \`src/auth.ts\` or \`src/auth.ts:42\`, never bracketed citations. Be direct; have an opinion when context supports one. Default to ASCII.
67
+ You are the human surface: the final message is the Handoff block, whose For you slot leads with the outcome (delivered / blocked / partial), then evidence — what you verified directly, what a worker verified and you audited, ${context.surface === "app" ? "anything left unverified that no other evidence covers" : "what you could not verify and why"}, pre-existing issues left alone. Reference files as \`src/auth.ts\` or \`src/auth.ts:42\`, never bracketed citations. Be direct; have an opinion when context supports one. Default to ASCII.
68
68
 
69
69
  ## Stop Goal
70
70
 
@@ -32,14 +32,14 @@
32
32
  import { APP_NAME } from "../../../../config.js";
33
33
  import { buildDynamicSystemPrompt } from "../../../dynamic-prompt/build.js";
34
34
  import { buildHandoffSection } from "../../../dynamic-prompt/handoff.js";
35
- import { buildTestDisciplineSection } from "../../../dynamic-prompt/verification.js";
35
+ import { APP_UNRUN_CHECK_RULE, buildTestDisciplineSection } from "../../../dynamic-prompt/verification.js";
36
36
  const INTENT_GATE_LEAD = {
37
37
  terminal: `Open every turn with one short visible routing line - required even on confirmation turns:
38
38
 
39
39
  > I read this as [intent] - [plan]. I'll stop when [the exact, observable condition that ends this turn].
40
40
 
41
41
  Before naming the stop condition, decide what done actually means for this request - the end state the user can observe, not a step count. Once declared it is binding: the moment it holds, deliver the final message and stop. Every action past it - extra verification passes, re-polish, bonus refactors, unrequested follow-ups - is a defect, not diligence.`,
42
- app: `Before acting, decide what done actually means for this request - the end state the user can observe, not a step count. It is binding: the moment it holds, deliver the final message and stop. Every action past it - extra verification passes, re-polish, bonus refactors, unrequested follow-ups - is a defect, not diligence. Replies render in an app: tool and hook feedback (comment-checker findings, language-server availability, internal notices) is yours to act on; report it only when it changes what the user gets.`,
42
+ app: `Before acting, decide what done actually means for this request - the end state the user can observe, not a step count. It is binding: the moment it holds, deliver the final message and stop. Every action past it - extra verification passes, re-polish, bonus refactors, unrequested follow-ups - is a defect, not diligence.`,
43
43
  };
44
44
  function buildGrok46Core(context) {
45
45
  return `You are ${APP_NAME}, a coding agent running on Grok 4.6 - a fast, decisive daily driver. Ship work indistinguishable from a careful senior engineer's.
@@ -73,7 +73,7 @@ Tier the scope, never the rigor.
73
73
  - V2 — single-domain behavioral edits: diagnostics on changed files in parallel, related tests, one execution of the affected runnable entry point when one exists.
74
74
  - V3 — multi-file or cross-cutting work: diagnostics on every changed file, related tests, build, manual exercise of user-visible behavior through its real surface.
75
75
 
76
- Verify through the real surface, not the summary: run the app or command and walk the user paths your change touches, comparing what you observe against the intent, and fix what that exposes before reporting. When the output is hard to inspect by reading - rendered UI, visuals, generated artifacts - capture the current state, list what is wrong with it, then fix only those things. "Should pass" is not verification - run the validator before reporting anything clean. Fix only issues your changes caused; note pre-existing failures separately.
76
+ Verify through the real surface, not the summary: run the app or command and walk the user paths your change touches, comparing what you observe against the intent, and fix what that exposes before reporting. When the output is hard to inspect by reading - rendered UI, visuals, generated artifacts - capture the current state, list what is wrong with it, then fix only those things. "Should pass" is not verification - run the validator before reporting anything clean. Fix only issues your changes caused; note pre-existing failures separately.${context.surface === "app" ? ` ${APP_UNRUN_CHECK_RULE}` : ""}
77
77
 
78
78
  ${buildTestDisciplineSection()}
79
79
 
@@ -52,14 +52,14 @@
52
52
  import { APP_NAME } from "../../../../config.js";
53
53
  import { buildDynamicSystemPrompt } from "../../../dynamic-prompt/build.js";
54
54
  import { buildHandoffSection } from "../../../dynamic-prompt/handoff.js";
55
- import { buildTestDisciplineSection } from "../../../dynamic-prompt/verification.js";
55
+ import { APP_UNRUN_CHECK_RULE, buildTestDisciplineSection } from "../../../dynamic-prompt/verification.js";
56
56
  const INTENT_GATE_LEAD = {
57
57
  terminal: `Open every turn with one short visible routing line - required even on confirmation turns:
58
58
 
59
59
  > I read this as [intent] - [plan]. I'll stop when [the exact, observable condition that ends this turn].
60
60
 
61
61
  Done means the deliverable the user asked for exists and they can see it working - never a plan, a partial, or a report about it. Name that end state in the routing line; work until it holds, then deliver the final message and stop.`,
62
- app: `Done means the deliverable the user asked for exists and they can see it working - never a plan, a partial, or a report about it. Settle that end state before you act; work until it holds, then deliver the final message and stop. Replies render in an app: tool and hook feedback (comment-checker findings, language-server availability, internal notices) is yours to act on; report it only when it changes what the user gets.`,
62
+ app: `Done means the deliverable the user asked for exists and they can see it working - never a plan, a partial, or a report about it. Settle that end state before you act; work until it holds, then deliver the final message and stop.`,
63
63
  };
64
64
  function buildGrok47Core(context) {
65
65
  return `You are ${APP_NAME}, a coding agent running on Grok 4.7. Ship work indistinguishable from a careful senior engineer's.
@@ -94,7 +94,7 @@ Tier the scope, never the rigor.
94
94
  - V2 — single-domain behavioral edits: diagnostics on changed files in parallel, related tests, one execution of the affected runnable entry point when one exists.
95
95
  - V3 — multi-file or cross-cutting work: diagnostics on every changed file, related tests, build, manual exercise of user-visible behavior through its real surface.
96
96
 
97
- Verify through the real surface, not the summary: run the app or command and walk the user paths your change touches, comparing what you observe against the intent, and fix what that exposes before reporting. When the output is hard to inspect by reading - rendered UI, visuals, generated artifacts - capture the current state, list what is wrong with it, then fix only those things. "Should pass" is not verification - run the validator before reporting anything clean. Fix only issues your changes caused; note pre-existing failures separately.
97
+ Verify through the real surface, not the summary: run the app or command and walk the user paths your change touches, comparing what you observe against the intent, and fix what that exposes before reporting. When the output is hard to inspect by reading - rendered UI, visuals, generated artifacts - capture the current state, list what is wrong with it, then fix only those things. "Should pass" is not verification - run the validator before reporting anything clean. Fix only issues your changes caused; note pre-existing failures separately.${context.surface === "app" ? ` ${APP_UNRUN_CHECK_RULE}` : ""}
98
98
 
99
99
  ${buildTestDisciplineSection()}
100
100
 
@@ -43,7 +43,7 @@ import { APP_NAME } from "../../../../config.js";
43
43
  import { buildDynamicSystemPrompt } from "../../../dynamic-prompt/build.js";
44
44
  import { buildHandoffSection } from "../../../dynamic-prompt/handoff.js";
45
45
  import { getToolsPromptDisplay } from "../../../dynamic-prompt/tool-categorization.js";
46
- import { buildTestDisciplineSection } from "../../../dynamic-prompt/verification.js";
46
+ import { APP_UNRUN_CHECK_RULE, buildTestDisciplineSection } from "../../../dynamic-prompt/verification.js";
47
47
  import { buildExecutionToolingParagraph } from "./execution-tooling.js";
48
48
  function buildSearchLine(context) {
49
49
  const triggerTools = getToolsPromptDisplay(context.tools);
@@ -58,7 +58,7 @@ const INTENT_GATE_LEAD = {
58
58
  > I read this as [intent] - [plan]. I'll stop when [the exact, observable condition that ends this turn].
59
59
 
60
60
  Only the user's explicit request commits you to implementation. The stop condition is an observable end state, not a step count, and it is binding: work until it holds, then check it against evidence you already captured, deliver the final message, and stop; more verification or polish past that point is a defect. Never echo prompt scaffolding in user-facing output.`,
61
- app: `Only the user's explicit request commits you to implementation. Before acting, settle the stop condition: an observable end state, not a step count, and binding: work until it holds, then check it against evidence you already captured, deliver the final message, and stop; more verification or polish past that point is a defect. Never echo prompt scaffolding in user-facing output. Replies render in an app: tool and hook feedback (comment-checker findings, language-server availability, internal notices) is yours to act on, and you mention it only when it changes what the user gets.`,
61
+ app: `Only the user's explicit request commits you to implementation. Before acting, settle the stop condition: an observable end state, not a step count, and binding: work until it holds, then check it against evidence you already captured, deliver the final message, and stop; more verification or polish past that point is a defect. Never echo prompt scaffolding in user-facing output.`,
62
62
  };
63
63
  function buildKimiK3Core(context) {
64
64
  return `You are ${APP_NAME}, a coding agent running on Kimi K3. Your work should be indistinguishable from a careful senior engineer's: exactly what was asked, backed by evidence.
@@ -96,7 +96,7 @@ Scale the checks to the change, never the rigor: diagnostics on every changed fi
96
96
 
97
97
  ${buildTestDisciplineSection()}
98
98
 
99
- "Should pass" is not verification: run the validator. Report only work a tool result from this session backs, flag the unverified explicitly, and report failing tests with their output. Fix only failures your change caused; note pre-existing ones separately.
99
+ "Should pass" is not verification: run the validator. ${context.surface === "app" ? `Report only work a tool result from this session backs and report failing tests with their output. ${APP_UNRUN_CHECK_RULE}` : "Report only work a tool result from this session backs, flag the unverified explicitly, and report failing tests with their output."} Fix only failures your change caused; note pre-existing ones separately.
100
100
 
101
101
  ${context.toolSection}
102
102
 
@@ -31,16 +31,19 @@ import { parsePromptPreset } from "./settings.js";
31
31
  function normalizeModelId(modelId) {
32
32
  return modelId.toLowerCase().replace(/\s+/g, "-");
33
33
  }
34
- // The GPT-6 family (Astra, Sol, Luna) shares one prompting guide
35
- // (developers.openai.com/api/docs/guides/latest-model, 2026-09-23), so every tier
36
- // renders the gpt-6-astra preset; the preset keeps that name because settings.json
37
- // already pins it. Id shapes verified against the OpenAI model pages, codex's
38
- // models.json, models.dev and Bedrock's catalog: gpt-6-sol, gpt-6-luna-fast, dated
39
- // snapshots, openai/gpt-6-sol, openai-gpt-6-luna, global.openai.gpt-6-astra, and
40
- // the display names "GPT-6 Sol" / "GPT-6 Luna". Bare "gpt-6", "gpt-6-mini" and a
41
- // lone tier word stay out: an unknown sibling deserves its own decision.
34
+ // The GPT-6 family (Astra, 6.1 Sol, Sol, Luna) shares one prompting guide
35
+ // (developers.openai.com/api/docs/guides/latest-model, 2026-09-23; GPT-6.1 Sol added
36
+ // 2026-09-29), so every tier renders the gpt-6-astra preset; the preset keeps that name
37
+ // because settings.json already pins it. Id shapes verified against the OpenAI model
38
+ // pages, codex's models.json, models.dev, OpenRouter, Vercel and Bedrock's catalog:
39
+ // gpt-6-sol, gpt-6.1-sol, gpt-6.1-sol-fast, gpt-6-luna-fast, dated snapshots,
40
+ // openai/gpt-6-sol, openai/gpt-6.1-sol, openai-gpt-6-luna, global.openai.gpt-6-astra, Venice's
41
+ // dotless openai-gpt-61-sol (it spells every point release that way: openai-gpt-56-sol), and
42
+ // the display names "GPT-6 Sol" / "GPT-6.1 Sol" / "GPT-6 Luna". Bare "gpt-6", "gpt-6.1",
43
+ // "gpt-61", "gpt-6-mini" and a lone tier word stay out: an unknown sibling deserves its own
44
+ // decision, and the dotless form is accepted only with a single digit right after the 6.
42
45
  function hasGpt6FamilySignal(value) {
43
- return /(?:^|[/@:._-])gpt[._-]?6[._-](?:astra|sol|luna)(?:$|[/@:._-])/.test(normalizeModelId(value));
46
+ return /(?:^|[/@:._-])gpt[._-]?6(?:[._-]\d+|\d)?[._-](?:astra|sol|luna)(?:$|[/@:._-])/.test(normalizeModelId(value));
44
47
  }
45
48
  function isGpt6FamilyModel(model) {
46
49
  return hasGpt6FamilySignal(model.id) || (model.name !== undefined && hasGpt6FamilySignal(model.name));
@@ -1,6 +1,7 @@
1
- // Matches GPT-5.x Sol, GPT-6 Sol and GPT-6 Astra variants, including provider-prefixed
2
- // forms. The trailing lookahead excludes unrelated ids that continue with letters.
3
- const SENSITIVE_MODEL_ID_PATTERN = /(?:gpt-5(?:\.\d+)?-sol|gpt-6-sol|gpt-6-astra)(?![a-z])/i;
1
+ // Matches GPT-5.x Sol, GPT-6.x Sol (gpt-6-sol, gpt-6.1-sol, Venice's dotless gpt-61-sol) and
2
+ // GPT-6 Astra variants, including provider-prefixed forms. The trailing lookahead excludes
3
+ // unrelated ids that continue with letters.
4
+ const SENSITIVE_MODEL_ID_PATTERN = /(?:gpt-5(?:\.\d+)?-sol|gpt-6(?:\.\d+|\d)?-sol|gpt-6-astra)(?![a-z])/i;
4
5
  const ASTRA_MODEL_ID_PATTERN = /gpt-6-astra(?![a-z])/i;
5
6
  export function isSensitiveHighReasoningModel(model) {
6
7
  return SENSITIVE_MODEL_ID_PATTERN.test(model.id);
@@ -14,9 +14,9 @@ export const defaultModelPerProvider = {
14
14
  "anthropic-subscription": "claude-opus-4-8",
15
15
  anthropic: "claude-opus-4-8",
16
16
  bai: "gpt-5.6-sol",
17
- openai: "gpt-6-sol",
17
+ openai: "gpt-6.1-sol",
18
18
  "azure-openai-responses": "gpt-5.4",
19
- "chatgpt-subscription": "gpt-6-sol",
19
+ "chatgpt-subscription": "gpt-6.1-sol",
20
20
  ollama: "qwen3.5:397b",
21
21
  // Cursor ships no models until its chat protocol is ported; "auto" matches
22
22
  // the Cursor agent's native model auto-selection once models exist.
@@ -17,7 +17,7 @@
17
17
  import * as crypto from "node:crypto";
18
18
  import { basename, dirname, extname } from "node:path";
19
19
  import { VERSION } from "../../config.js";
20
- import { buildLoginProviderInfos } from "../../core/auth-providers.js";
20
+ import { authMethodStatus, buildLoginProviderInfos } from "../../core/auth-providers.js";
21
21
  import { getCredentialAccounts, pinCredentialAccount, removeCredentialAccount, } from "../../core/credential-accounts.js";
22
22
  import { AssistantEditError, SessionStreamingError } from "../../core/edited-assistant-message.js";
23
23
  import { UserEditError } from "../../core/edited-user-message.js";
@@ -1303,11 +1303,12 @@ export function createRpcConnectionHandler(runtimeHost, sink, options = {}) {
1303
1303
  const modelRegistry = session.modelRegistry;
1304
1304
  const oauthInfos = buildLoginProviderInfos(modelRegistry, "oauth");
1305
1305
  const apiKeyInfos = buildLoginProviderInfos(modelRegistry, "api_key");
1306
+ const apiKeyRows = new Set(apiKeyInfos.map((info) => info.id));
1306
1307
  const providers = [...oauthInfos, ...apiKeyInfos].map((info) => ({
1307
1308
  id: info.id,
1308
1309
  name: info.name,
1309
1310
  authType: info.authType,
1310
- status: modelRegistry.getProviderAuthStatus(info.id),
1311
+ status: authMethodStatus(modelRegistry, info, apiKeyRows.has(info.id)),
1311
1312
  }));
1312
1313
  return success(id, "get_auth_providers", { providers });
1313
1314
  }
@@ -1326,12 +1327,14 @@ export function createRpcConnectionHandler(runtimeHost, sink, options = {}) {
1326
1327
  }
1327
1328
  case "login_api_key": {
1328
1329
  session.modelRegistry.authStorage.set(command.provider, { type: "api_key", key: command.key });
1329
- session.modelRegistry.refresh();
1330
+ // Answer after the registry sees the key: a client re-reading auth status on this
1331
+ // response must not get the pre-login snapshot (#2384).
1332
+ await session.modelRegistry.refresh();
1330
1333
  return success(id, "login_api_key");
1331
1334
  }
1332
1335
  case "logout": {
1333
1336
  session.modelRegistry.authStorage.logout(command.provider);
1334
- session.modelRegistry.refresh();
1337
+ await session.modelRegistry.refresh();
1335
1338
  return success(id, "logout");
1336
1339
  }
1337
1340
  case "get_provider_accounts": {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@code-yeongyu/senpi",
3
- "version": "2026.9.29-4",
3
+ "version": "2026.9.29-5",
4
4
  "description": "Coding agent CLI with read, bash, edit, write tools and session management",
5
5
  "type": "module",
6
6
  "piConfig": {
@@ -56,9 +56,9 @@
56
56
  },
57
57
  "dependencies": {
58
58
  "@earendil-works/chord": "0.85.1",
59
- "@earendil-works/pi-agent-core": "npm:@code-yeongyu/senpi-agent-core@2026.9.29-4",
60
- "@earendil-works/pi-ai": "npm:@code-yeongyu/senpi-ai@2026.9.29-4",
61
- "@earendil-works/pi-tui": "npm:@code-yeongyu/senpi-tui@2026.9.29-4",
59
+ "@earendil-works/pi-agent-core": "npm:@code-yeongyu/senpi-agent-core@2026.9.29-5",
60
+ "@earendil-works/pi-ai": "npm:@code-yeongyu/senpi-ai@2026.9.29-5",
61
+ "@earendil-works/pi-tui": "npm:@code-yeongyu/senpi-tui@2026.9.29-5",
62
62
  "@silvia-odwyer/photon-node": "0.3.4",
63
63
  "chalk": "6.0.0",
64
64
  "cross-spawn": "7.0.6",
@@ -77,8 +77,8 @@
77
77
  "yaml": "2.9.1",
78
78
  "@anthropic-ai/claude-agent-sdk": "0.3.284",
79
79
  "@anthropic-ai/sdk": "0.127.0",
80
- "@code-yeongyu/senpi-codemode": "2026.9.29-4",
81
- "@earendil-works/pi-pty": "npm:@code-yeongyu/senpi-pty@2026.9.29-4",
80
+ "@code-yeongyu/senpi-codemode": "2026.9.29-5",
81
+ "@earendil-works/pi-pty": "npm:@code-yeongyu/senpi-pty@2026.9.29-5",
82
82
  "@modelcontextprotocol/sdk": "1.30.0",
83
83
  "@mozilla/readability": "0.6.0",
84
84
  "linkedom": "0.18.13",