@bastani/atomic 0.9.18-alpha.3 → 0.9.18-alpha.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. package/CHANGELOG.md +35 -0
  2. package/dist/builtin/intercom/CHANGELOG.md +8 -0
  3. package/dist/builtin/intercom/README.md +1 -1
  4. package/dist/builtin/intercom/index.bundle.mjs +13 -4
  5. package/dist/builtin/intercom/package.json +1 -1
  6. package/dist/builtin/mcp/package.json +1 -1
  7. package/dist/builtin/subagents/package.json +1 -1
  8. package/dist/builtin/web-access/package.json +1 -1
  9. package/dist/builtin/workflows/package.json +1 -1
  10. package/dist/core/agent-session-runtime.js +2 -2
  11. package/dist/core/agent-session-runtime.js.map +1 -1
  12. package/dist/core/anthropic-thinking-guard.d.ts.map +1 -1
  13. package/dist/core/anthropic-thinking-guard.js +71 -5
  14. package/dist/core/anthropic-thinking-guard.js.map +1 -1
  15. package/dist/core/extensions/provider-types.d.ts +2 -2
  16. package/dist/core/extensions/provider-types.d.ts.map +1 -1
  17. package/dist/core/extensions/provider-types.js.map +1 -1
  18. package/dist/core/model-config.d.ts +9 -4
  19. package/dist/core/model-config.d.ts.map +1 -1
  20. package/dist/core/model-config.js +3 -2
  21. package/dist/core/model-config.js.map +1 -1
  22. package/dist/core/provider-composer-internal.d.ts +1 -1
  23. package/dist/core/provider-composer-internal.d.ts.map +1 -1
  24. package/dist/core/provider-composer-internal.js.map +1 -1
  25. package/dist/core/sdk.d.ts.map +1 -1
  26. package/dist/core/sdk.js +3 -0
  27. package/dist/core/sdk.js.map +1 -1
  28. package/dist/core/tools/bash.d.ts.map +1 -1
  29. package/dist/core/tools/bash.js +10 -7
  30. package/dist/core/tools/bash.js.map +1 -1
  31. package/dist/core/tools/edit.d.ts.map +1 -1
  32. package/dist/core/tools/edit.js +25 -5
  33. package/dist/core/tools/edit.js.map +1 -1
  34. package/dist/core/tools/find.d.ts.map +1 -1
  35. package/dist/core/tools/find.js +9 -8
  36. package/dist/core/tools/find.js.map +1 -1
  37. package/dist/core/tools/grep.d.ts.map +1 -1
  38. package/dist/core/tools/grep.js +4 -3
  39. package/dist/core/tools/grep.js.map +1 -1
  40. package/dist/core/tools/ls.d.ts.map +1 -1
  41. package/dist/core/tools/ls.js +2 -2
  42. package/dist/core/tools/ls.js.map +1 -1
  43. package/dist/core/tools/read.d.ts.map +1 -1
  44. package/dist/core/tools/read.js +12 -11
  45. package/dist/core/tools/read.js.map +1 -1
  46. package/dist/core/tools/search.d.ts.map +1 -1
  47. package/dist/core/tools/search.js +20 -19
  48. package/dist/core/tools/search.js.map +1 -1
  49. package/dist/core/tools/write.d.ts.map +1 -1
  50. package/dist/core/tools/write.js +19 -18
  51. package/dist/core/tools/write.js.map +1 -1
  52. package/dist/modes/interactive/components/model-selector.d.ts.map +1 -1
  53. package/dist/modes/interactive/components/model-selector.js +6 -14
  54. package/dist/modes/interactive/components/model-selector.js.map +1 -1
  55. package/dist/modes/interactive/components/scoped-models-selector.d.ts.map +1 -1
  56. package/dist/modes/interactive/components/scoped-models-selector.js +13 -15
  57. package/dist/modes/interactive/components/scoped-models-selector.js.map +1 -1
  58. package/dist/modes/interactive/components/settings-selector-items.d.ts.map +1 -1
  59. package/dist/modes/interactive/components/settings-selector-items.js +9 -4
  60. package/dist/modes/interactive/components/settings-selector-items.js.map +1 -1
  61. package/dist/modes/interactive/components/settings-selector-submenus.d.ts +1 -1
  62. package/dist/modes/interactive/components/settings-selector-submenus.d.ts.map +1 -1
  63. package/dist/modes/interactive/components/settings-selector-submenus.js +10 -7
  64. package/dist/modes/interactive/components/settings-selector-submenus.js.map +1 -1
  65. package/dist/modes/interactive/components/thinking-selector.js +2 -2
  66. package/dist/modes/interactive/components/thinking-selector.js.map +1 -1
  67. package/dist/modes/interactive/components/trust-selector.js +2 -2
  68. package/dist/modes/interactive/components/trust-selector.js.map +1 -1
  69. package/docs/compaction.md +2 -0
  70. package/docs/custom-provider.md +2 -1
  71. package/docs/extensions.md +2 -0
  72. package/docs/intercom.md +1 -1
  73. package/docs/models/artificial-analysis-index.md +2 -1
  74. package/docs/models/model-selection.md +28 -19
  75. package/docs/models/pareto-efficiency.md +37 -29
  76. package/docs/models.md +43 -0
  77. package/docs/terminal-setup.md +15 -0
  78. package/npm-shrinkwrap.json +32 -32
  79. package/package.json +3 -3
@@ -1 +1 @@
1
- {"version":3,"file":"trust-selector.js","sourceRoot":"","sources":["../../../../src/modes/interactive/components/trust-selector.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,SAAS,EAAE,cAAc,EAAE,MAAM,EAAE,IAAI,EAAE,MAAM,wBAAwB,CAAC;AACjF,OAAO,EACN,sBAAsB,EACtB,mBAAmB,GAGnB,MAAM,gCAAgC,CAAC;AACxC,OAAO,EAAE,KAAK,EAAE,MAAM,mBAAmB,CAAC;AAC1C,OAAO,EAAE,aAAa,EAAE,MAAM,qBAAqB,CAAC;AACpD,OAAO,EAAE,OAAO,EAAE,UAAU,EAAE,MAAM,uBAAuB,CAAC;AAY5D,SAAS,cAAc,CAAC,GAAW,EAAE,QAAuC;IAC3E,IAAI,QAAQ,KAAK,IAAI,EAAE,CAAC;QACvB,OAAO,MAAM,CAAC;IACf,CAAC;IACD,MAAM,KAAK,GAAG,QAAQ,CAAC,QAAQ,CAAC,CAAC,CAAC,SAAS,CAAC,CAAC,CAAC,WAAW,CAAC;IAC1D,IAAI,QAAQ,CAAC,IAAI,KAAK,mBAAmB,CAAC,GAAG,CAAC,EAAE,CAAC;QAChD,OAAO,GAAG,KAAK,oBAAoB,QAAQ,CAAC,IAAI,GAAG,CAAC;IACrD,CAAC;IACD,OAAO,GAAG,KAAK,KAAK,QAAQ,CAAC,IAAI,GAAG,CAAC;AACtC,CAAC;AAED,MAAM,OAAO,sBAAuB,SAAQ,SAAS;IAQpD,YAAY,OAA6B;QACxC,KAAK,EAAE,CAAC;QAER,IAAI,CAAC,aAAa,GAAG,OAAO,CAAC,aAAa,CAAC;QAC3C,IAAI,CAAC,YAAY,GAAG,sBAAsB,CAAC,OAAO,CAAC,GAAG,CAAC,CAAC;QACxD,IAAI,CAAC,aAAa,GAAG,IAAI,CAAC,GAAG,CAC5B,CAAC,EACD,IAAI,CAAC,YAAY,CAAC,SAAS,CAAC,CAAC,MAAM,EAAE,EAAE,CAAC,IAAI,CAAC,aAAa,CAAC,MAAM,CAAC,CAAC,CACnE,CAAC;QACF,IAAI,CAAC,gBAAgB,GAAG,OAAO,CAAC,QAAQ,CAAC;QACzC,IAAI,CAAC,gBAAgB,GAAG,OAAO,CAAC,QAAQ,CAAC;QAEzC,IAAI,CAAC,QAAQ,CAAC,IAAI,aAAa,EAAE,CAAC,CAAC;QACnC,IAAI,CAAC,QAAQ,CAAC,IAAI,MAAM,CAAC,CAAC,CAAC,CAAC,CAAC;QAC7B,IAAI,CAAC,QAAQ,CAAC,IAAI,IAAI,CAAC,KAAK,CAAC,EAAE,CAAC,QAAQ,EAAE,KAAK,CAAC,IAAI,CAAC,eAAe,CAAC,CAAC,EAAE,CAAC,EAAE,CAAC,CAAC,CAAC,CAAC;QAC/E,IAAI,CAAC,QAAQ,CAAC,IAAI,IAAI,CAAC,KAAK,CAAC,EAAE,CAAC,OAAO,EAAE,OAAO,CAAC,GAAG,CAAC,EAAE,CAAC,EAAE,CAAC,CAAC,CAAC,CAAC;QAC9D,IAAI,CAAC,QAAQ,CAAC,IAAI,MAAM,CAAC,CAAC,CAAC,CAAC,CAAC;QAC7B,IAAI,CAAC,QAAQ,CACZ,IAAI,IAAI,CAAC,KAAK,CAAC,EAAE,CAAC,OAAO,EAAE,mBAAmB,cAAc,CAAC,OAAO,CAAC,GAAG,EAAE,OAAO,CAAC,aAAa,CAAC,EAAE,CAAC,EAAE,CAAC,EAAE,CAAC,CAAC,CAC1G,CAAC;QACF,IAAI,CAAC,QAAQ,CACZ,IAAI,IAAI,CAAC,KAAK,CAAC,EAAE,CAAC,OAAO,EAAE,oBAAoB,OAAO,CAAC,cAAc,CAAC,CAAC,CAAC,SAAS,CAAC,CAAC,CAAC,WAAW,EAAE,CAAC,EAAE,CAAC,EAAE,CAAC,CAAC,CACzG,CAAC;QACF,IAAI,CAAC,QAAQ,CAAC,IAAI,MAAM,CAAC,CAAC,CAAC,CAAC,CAAC;QAE7B,IAAI,CAAC,aAAa,GAAG,IAAI,SAAS,EAAE,CAAC;QACrC,IAAI,CAAC,QAAQ,CAAC,IAAI,CAAC,aAAa,CAAC,CAAC;QAClC,IAAI,CAAC,QAAQ,CAAC,IAAI,MAAM,CAAC,CAAC,CAAC,CAAC,CAAC;QAC7B,IAAI,CAAC,QAAQ,CACZ,IAAI,IAAI,CACP,UAAU,CAAC,IAAI,EAAE,UAAU,CAAC;YAC3B,IAAI;YACJ,OAAO,CAAC,oBAAoB,EAAE,MAAM,CAAC;YACrC,IAAI;YACJ,OAAO,CAAC,mBAAmB,EAAE,QAAQ,CAAC,EACvC,CAAC,EACD,CAAC,CACD,CACD,CAAC;QACF,IAAI,CAAC,QAAQ,CAAC,IAAI,MAAM,CAAC,CAAC,CAAC,CAAC,CAAC;QAC7B,IAAI,CAAC,QAAQ,CAAC,IAAI,aAAa,EAAE,CAAC,CAAC;QAEnC,IAAI,CAAC,UAAU,EAAE,CAAC;IACnB,CAAC;IAEO,aAAa,CAAC,MAA0B;QAC/C,OAAO,CACN,MAAM,CAAC,SAAS,KAAK,SAAS;YAC9B,IAAI,CAAC,aAAa,EAAE,QAAQ,KAAK,MAAM,CAAC,OAAO;YAC/C,IAAI,CAAC,aAAa,CAAC,IAAI,KAAK,MAAM,CAAC,SAAS,CAC5C,CAAC;IACH,CAAC;IAEO,UAAU;QACjB,IAAI,CAAC,aAAa,CAAC,KAAK,EAAE,CAAC;QAC3B,KAAK,IAAI,CAAC,GAAG,CAAC,EAAE,CAAC,GAAG,IAAI,CAAC,YAAY,CAAC,MAAM,EAAE,CAAC,EAAE,EAAE,CAAC;YACnD,MAAM,MAAM,GAAG,IAAI,CAAC,YAAY,CAAC,CAAC,CAAC,CAAC;YACpC,IAAI,CAAC,MAAM,EAAE,CAAC;gBACb,SAAS;YACV,CAAC;YAED,MAAM,UAAU,GAAG,CAAC,KAAK,IAAI,CAAC,aAAa,CAAC;YAC5C,MAAM,SAAS,GAAG,IAAI,CAAC,aAAa,CAAC,MAAM,CAAC,CAAC;YAC7C,MAAM,SAAS,GAAG,SAAS,CAAC,CAAC,CAAC,KAAK,CAAC,EAAE,CAAC,SAAS,EAAE,IAAI,CAAC,CAAC,CAAC,CAAC,EAAE,CAAC;YAC7D,MAAM,MAAM,GAAG,UAAU,CAAC,CAAC,CAAC,KAAK,CAAC,EAAE,CAAC,QAAQ,EAAE,IAAI,CAAC,CAAC,CAAC,CAAC,IAAI,CAAC;YAC5D,MAAM,KAAK,GAAG,UAAU,CAAC,CAAC,CAAC,KAAK,CAAC,EAAE,CAAC,QAAQ,EAAE,MAAM,CAAC,KAAK,CAAC,CAAC,CAAC,CAAC,KAAK,CAAC,EAAE,CAAC,MAAM,EAAE,MAAM,CAAC,KAAK,CAAC,CAAC;YAC7F,IAAI,CAAC,aAAa,CAAC,QAAQ,CAAC,IAAI,IAAI,CAAC,GAAG,MAAM,GAAG,KAAK,GAAG,SAAS,EAAE,EAAE,CAAC,EAAE,CAAC,CAAC,CAAC,CAAC;QAC9E,CAAC;IACF,CAAC;IAED,WAAW,CAAC,OAAe;QAC1B,MAAM,EAAE,GAAG,cAAc,EAAE,CAAC;QAC5B,IAAI,EAAE,CAAC,OAAO,CAAC,OAAO,EAAE,eAAe,CAAC,IAAI,OAAO,KAAK,GAAG,EAAE,CAAC;YAC7D,IAAI,CAAC,aAAa,GAAG,IAAI,CAAC,GAAG,CAAC,CAAC,EAAE,IAAI,CAAC,aAAa,GAAG,CAAC,CAAC,CAAC;YACzD,IAAI,CAAC,UAAU,EAAE,CAAC;QACnB,CAAC;aAAM,IAAI,EAAE,CAAC,OAAO,CAAC,OAAO,EAAE,iBAAiB,CAAC,IAAI,OAAO,KAAK,GAAG,EAAE,CAAC;YACtE,IAAI,CAAC,aAAa,GAAG,IAAI,CAAC,GAAG,CAAC,IAAI,CAAC,YAAY,CAAC,MAAM,GAAG,CAAC,EAAE,IAAI,CAAC,aAAa,GAAG,CAAC,CAAC,CAAC;YACpF,IAAI,CAAC,UAAU,EAAE,CAAC;QACnB,CAAC;aAAM,IAAI,EAAE,CAAC,OAAO,CAAC,OAAO,EAAE,oBAAoB,CAAC,IAAI,OAAO,KAAK,IAAI,EAAE,CAAC;YAC1E,MAAM,QAAQ,GAAG,IAAI,CAAC,YAAY,CAAC,IAAI,CAAC,aAAa,CAAC,CAAC;YACvD,IAAI,QAAQ,EAAE,CAAC;gBACd,IAAI,CAAC,gBAAgB,CAAC,EAAE,OAAO,EAAE,QAAQ,CAAC,OAAO,EAAE,OAAO,EAAE,QAAQ,CAAC,OAAO,EAAE,CAAC,CAAC;YACjF,CAAC;QACF,CAAC;aAAM,IAAI,EAAE,CAAC,OAAO,CAAC,OAAO,EAAE,mBAAmB,CAAC,EAAE,CAAC;YACrD,IAAI,CAAC,gBAAgB,EAAE,CAAC;QACzB,CAAC;aAAM,CAAC;YACP,OAAO,KAAK,CAAC;QACd,CAAC;QACD,OAAO,IAAI,CAAC;IACb,CAAC;CACD","sourcesContent":["import { Container, getKeybindings, Spacer, Text } from \"@earendil-works/pi-tui\";\nimport {\n\tgetProjectTrustOptions,\n\tgetProjectTrustPath,\n\ttype ProjectTrustOption,\n\ttype ProjectTrustStoreEntry,\n} from \"../../../core/trust-manager.ts\";\nimport { theme } from \"../theme/theme.ts\";\nimport { DynamicBorder } from \"./dynamic-border.ts\";\nimport { keyHint, rawKeyHint } from \"./keybinding-hints.ts\";\n\nexport type TrustSelection = Pick<ProjectTrustOption, \"trusted\" | \"updates\">;\n\nexport interface TrustSelectorOptions {\n\tcwd: string;\n\tsavedDecision: ProjectTrustStoreEntry | null;\n\tprojectTrusted: boolean;\n\tonSelect: (selection: TrustSelection) => void;\n\tonCancel: () => void;\n}\n\nfunction formatDecision(cwd: string, decision: ProjectTrustStoreEntry | null): string {\n\tif (decision === null) {\n\t\treturn \"none\";\n\t}\n\tconst label = decision.decision ? \"trusted\" : \"untrusted\";\n\tif (decision.path !== getProjectTrustPath(cwd)) {\n\t\treturn `${label} (inherited from ${decision.path})`;\n\t}\n\treturn `${label} (${decision.path})`;\n}\n\nexport class TrustSelectorComponent extends Container {\n\tprivate selectedIndex: number;\n\tprivate readonly listContainer: Container;\n\tprivate readonly trustOptions: ProjectTrustOption[];\n\tprivate readonly savedDecision: ProjectTrustStoreEntry | null;\n\tprivate readonly onSelectCallback: (selection: TrustSelection) => void;\n\tprivate readonly onCancelCallback: () => void;\n\n\tconstructor(options: TrustSelectorOptions) {\n\t\tsuper();\n\n\t\tthis.savedDecision = options.savedDecision;\n\t\tthis.trustOptions = getProjectTrustOptions(options.cwd);\n\t\tthis.selectedIndex = Math.max(\n\t\t\t0,\n\t\t\tthis.trustOptions.findIndex((option) => this.isSavedOption(option)),\n\t\t);\n\t\tthis.onSelectCallback = options.onSelect;\n\t\tthis.onCancelCallback = options.onCancel;\n\n\t\tthis.addChild(new DynamicBorder());\n\t\tthis.addChild(new Spacer(1));\n\t\tthis.addChild(new Text(theme.fg(\"accent\", theme.bold(\"Project trust\")), 1, 0));\n\t\tthis.addChild(new Text(theme.fg(\"muted\", options.cwd), 1, 0));\n\t\tthis.addChild(new Spacer(1));\n\t\tthis.addChild(\n\t\t\tnew Text(theme.fg(\"muted\", `Saved decision: ${formatDecision(options.cwd, options.savedDecision)}`), 1, 0),\n\t\t);\n\t\tthis.addChild(\n\t\t\tnew Text(theme.fg(\"muted\", `Current session: ${options.projectTrusted ? \"trusted\" : \"untrusted\"}`), 1, 0),\n\t\t);\n\t\tthis.addChild(new Spacer(1));\n\n\t\tthis.listContainer = new Container();\n\t\tthis.addChild(this.listContainer);\n\t\tthis.addChild(new Spacer(1));\n\t\tthis.addChild(\n\t\t\tnew Text(\n\t\t\t\trawKeyHint(\"↑↓\", \"navigate\") +\n\t\t\t\t\t\" \" +\n\t\t\t\t\tkeyHint(\"tui.select.confirm\", \"save\") +\n\t\t\t\t\t\" \" +\n\t\t\t\t\tkeyHint(\"tui.select.cancel\", \"cancel\"),\n\t\t\t\t1,\n\t\t\t\t0,\n\t\t\t),\n\t\t);\n\t\tthis.addChild(new Spacer(1));\n\t\tthis.addChild(new DynamicBorder());\n\n\t\tthis.updateList();\n\t}\n\n\tprivate isSavedOption(option: ProjectTrustOption): boolean {\n\t\treturn (\n\t\t\toption.savedPath !== undefined &&\n\t\t\tthis.savedDecision?.decision === option.trusted &&\n\t\t\tthis.savedDecision.path === option.savedPath\n\t\t);\n\t}\n\n\tprivate updateList(): void {\n\t\tthis.listContainer.clear();\n\t\tfor (let i = 0; i < this.trustOptions.length; i++) {\n\t\t\tconst option = this.trustOptions[i];\n\t\t\tif (!option) {\n\t\t\t\tcontinue;\n\t\t\t}\n\n\t\t\tconst isSelected = i === this.selectedIndex;\n\t\t\tconst isCurrent = this.isSavedOption(option);\n\t\t\tconst checkmark = isCurrent ? theme.fg(\"success\", \" ✓\") : \"\";\n\t\t\tconst prefix = isSelected ? theme.fg(\"accent\", \"→ \") : \" \";\n\t\t\tconst label = isSelected ? theme.fg(\"accent\", option.label) : theme.fg(\"text\", option.label);\n\t\t\tthis.listContainer.addChild(new Text(`${prefix}${label}${checkmark}`, 1, 0));\n\t\t}\n\t}\n\n\thandleInput(keyData: string): boolean {\n\t\tconst kb = getKeybindings();\n\t\tif (kb.matches(keyData, \"tui.select.up\") || keyData === \"k\") {\n\t\t\tthis.selectedIndex = Math.max(0, this.selectedIndex - 1);\n\t\t\tthis.updateList();\n\t\t} else if (kb.matches(keyData, \"tui.select.down\") || keyData === \"j\") {\n\t\t\tthis.selectedIndex = Math.min(this.trustOptions.length - 1, this.selectedIndex + 1);\n\t\t\tthis.updateList();\n\t\t} else if (kb.matches(keyData, \"tui.select.confirm\") || keyData === \"\\n\") {\n\t\t\tconst selected = this.trustOptions[this.selectedIndex];\n\t\t\tif (selected) {\n\t\t\t\tthis.onSelectCallback({ trusted: selected.trusted, updates: selected.updates });\n\t\t\t}\n\t\t} else if (kb.matches(keyData, \"tui.select.cancel\")) {\n\t\t\tthis.onCancelCallback();\n\t\t} else {\n\t\t\treturn false;\n\t\t}\n\t\treturn true;\n\t}\n}\n"]}
1
+ {"version":3,"file":"trust-selector.js","sourceRoot":"","sources":["../../../../src/modes/interactive/components/trust-selector.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,SAAS,EAAE,cAAc,EAAE,MAAM,EAAE,IAAI,EAAE,MAAM,wBAAwB,CAAC;AACjF,OAAO,EACN,sBAAsB,EACtB,mBAAmB,GAGnB,MAAM,gCAAgC,CAAC;AACxC,OAAO,EAAE,KAAK,EAAE,MAAM,mBAAmB,CAAC;AAC1C,OAAO,EAAE,aAAa,EAAE,MAAM,qBAAqB,CAAC;AACpD,OAAO,EAAE,OAAO,EAAE,UAAU,EAAE,MAAM,uBAAuB,CAAC;AAY5D,SAAS,cAAc,CAAC,GAAW,EAAE,QAAuC;IAC3E,IAAI,QAAQ,KAAK,IAAI,EAAE,CAAC;QACvB,OAAO,MAAM,CAAC;IACf,CAAC;IACD,MAAM,KAAK,GAAG,QAAQ,CAAC,QAAQ,CAAC,CAAC,CAAC,SAAS,CAAC,CAAC,CAAC,WAAW,CAAC;IAC1D,IAAI,QAAQ,CAAC,IAAI,KAAK,mBAAmB,CAAC,GAAG,CAAC,EAAE,CAAC;QAChD,OAAO,GAAG,KAAK,oBAAoB,QAAQ,CAAC,IAAI,GAAG,CAAC;IACrD,CAAC;IACD,OAAO,GAAG,KAAK,KAAK,QAAQ,CAAC,IAAI,GAAG,CAAC;AACtC,CAAC;AAED,MAAM,OAAO,sBAAuB,SAAQ,SAAS;IAQpD,YAAY,OAA6B;QACxC,KAAK,EAAE,CAAC;QAER,IAAI,CAAC,aAAa,GAAG,OAAO,CAAC,aAAa,CAAC;QAC3C,IAAI,CAAC,YAAY,GAAG,sBAAsB,CAAC,OAAO,CAAC,GAAG,CAAC,CAAC;QACxD,IAAI,CAAC,aAAa,GAAG,IAAI,CAAC,GAAG,CAC5B,CAAC,EACD,IAAI,CAAC,YAAY,CAAC,SAAS,CAAC,CAAC,MAAM,EAAE,EAAE,CAAC,IAAI,CAAC,aAAa,CAAC,MAAM,CAAC,CAAC,CACnE,CAAC;QACF,IAAI,CAAC,gBAAgB,GAAG,OAAO,CAAC,QAAQ,CAAC;QACzC,IAAI,CAAC,gBAAgB,GAAG,OAAO,CAAC,QAAQ,CAAC;QAEzC,IAAI,CAAC,QAAQ,CAAC,IAAI,aAAa,EAAE,CAAC,CAAC;QACnC,IAAI,CAAC,QAAQ,CAAC,IAAI,MAAM,CAAC,CAAC,CAAC,CAAC,CAAC;QAC7B,IAAI,CAAC,QAAQ,CAAC,IAAI,IAAI,CAAC,KAAK,CAAC,EAAE,CAAC,QAAQ,EAAE,KAAK,CAAC,IAAI,CAAC,eAAe,CAAC,CAAC,EAAE,CAAC,EAAE,CAAC,CAAC,CAAC,CAAC;QAC/E,IAAI,CAAC,QAAQ,CAAC,IAAI,IAAI,CAAC,KAAK,CAAC,EAAE,CAAC,OAAO,EAAE,OAAO,CAAC,GAAG,CAAC,EAAE,CAAC,EAAE,CAAC,CAAC,CAAC,CAAC;QAC9D,IAAI,CAAC,QAAQ,CAAC,IAAI,MAAM,CAAC,CAAC,CAAC,CAAC,CAAC;QAC7B,IAAI,CAAC,QAAQ,CACZ,IAAI,IAAI,CAAC,KAAK,CAAC,EAAE,CAAC,OAAO,EAAE,mBAAmB,cAAc,CAAC,OAAO,CAAC,GAAG,EAAE,OAAO,CAAC,aAAa,CAAC,EAAE,CAAC,EAAE,CAAC,EAAE,CAAC,CAAC,CAC1G,CAAC;QACF,IAAI,CAAC,QAAQ,CACZ,IAAI,IAAI,CAAC,KAAK,CAAC,EAAE,CAAC,OAAO,EAAE,oBAAoB,OAAO,CAAC,cAAc,CAAC,CAAC,CAAC,SAAS,CAAC,CAAC,CAAC,WAAW,EAAE,CAAC,EAAE,CAAC,EAAE,CAAC,CAAC,CACzG,CAAC;QACF,IAAI,CAAC,QAAQ,CAAC,IAAI,MAAM,CAAC,CAAC,CAAC,CAAC,CAAC;QAE7B,IAAI,CAAC,aAAa,GAAG,IAAI,SAAS,EAAE,CAAC;QACrC,IAAI,CAAC,QAAQ,CAAC,IAAI,CAAC,aAAa,CAAC,CAAC;QAClC,IAAI,CAAC,QAAQ,CAAC,IAAI,MAAM,CAAC,CAAC,CAAC,CAAC,CAAC;QAC7B,IAAI,CAAC,QAAQ,CACZ,IAAI,IAAI,CACP,UAAU,CAAC,IAAI,EAAE,UAAU,CAAC;YAC3B,IAAI;YACJ,OAAO,CAAC,oBAAoB,EAAE,MAAM,CAAC;YACrC,IAAI;YACJ,OAAO,CAAC,mBAAmB,EAAE,QAAQ,CAAC,EACvC,CAAC,EACD,CAAC,CACD,CACD,CAAC;QACF,IAAI,CAAC,QAAQ,CAAC,IAAI,MAAM,CAAC,CAAC,CAAC,CAAC,CAAC;QAC7B,IAAI,CAAC,QAAQ,CAAC,IAAI,aAAa,EAAE,CAAC,CAAC;QAEnC,IAAI,CAAC,UAAU,EAAE,CAAC;IACnB,CAAC;IAEO,aAAa,CAAC,MAA0B;QAC/C,OAAO,CACN,MAAM,CAAC,SAAS,KAAK,SAAS;YAC9B,IAAI,CAAC,aAAa,EAAE,QAAQ,KAAK,MAAM,CAAC,OAAO;YAC/C,IAAI,CAAC,aAAa,CAAC,IAAI,KAAK,MAAM,CAAC,SAAS,CAC5C,CAAC;IACH,CAAC;IAEO,UAAU;QACjB,IAAI,CAAC,aAAa,CAAC,KAAK,EAAE,CAAC;QAC3B,KAAK,IAAI,CAAC,GAAG,CAAC,EAAE,CAAC,GAAG,IAAI,CAAC,YAAY,CAAC,MAAM,EAAE,CAAC,EAAE,EAAE,CAAC;YACnD,MAAM,MAAM,GAAG,IAAI,CAAC,YAAY,CAAC,CAAC,CAAC,CAAC;YACpC,IAAI,CAAC,MAAM,EAAE,CAAC;gBACb,SAAS;YACV,CAAC;YAED,MAAM,UAAU,GAAG,CAAC,KAAK,IAAI,CAAC,aAAa,CAAC;YAC5C,MAAM,SAAS,GAAG,IAAI,CAAC,aAAa,CAAC,MAAM,CAAC,CAAC;YAC7C,MAAM,aAAa,GAAG,SAAS,CAAC,CAAC,CAAC,KAAK,CAAC,EAAE,CAAC,QAAQ,EAAE,IAAI,CAAC,CAAC,CAAC,CAAC,IAAI,CAAC;YAClE,MAAM,MAAM,GAAG,UAAU,CAAC,CAAC,CAAC,KAAK,CAAC,EAAE,CAAC,QAAQ,EAAE,IAAI,CAAC,CAAC,CAAC,CAAC,IAAI,CAAC;YAC5D,MAAM,KAAK,GAAG,UAAU,CAAC,CAAC,CAAC,KAAK,CAAC,EAAE,CAAC,QAAQ,EAAE,MAAM,CAAC,KAAK,CAAC,CAAC,CAAC,CAAC,KAAK,CAAC,EAAE,CAAC,MAAM,EAAE,MAAM,CAAC,KAAK,CAAC,CAAC;YAC7F,IAAI,CAAC,aAAa,CAAC,QAAQ,CAAC,IAAI,IAAI,CAAC,GAAG,MAAM,GAAG,aAAa,GAAG,KAAK,EAAE,EAAE,CAAC,EAAE,CAAC,CAAC,CAAC,CAAC;QAClF,CAAC;IACF,CAAC;IAED,WAAW,CAAC,OAAe;QAC1B,MAAM,EAAE,GAAG,cAAc,EAAE,CAAC;QAC5B,IAAI,EAAE,CAAC,OAAO,CAAC,OAAO,EAAE,eAAe,CAAC,IAAI,OAAO,KAAK,GAAG,EAAE,CAAC;YAC7D,IAAI,CAAC,aAAa,GAAG,IAAI,CAAC,GAAG,CAAC,CAAC,EAAE,IAAI,CAAC,aAAa,GAAG,CAAC,CAAC,CAAC;YACzD,IAAI,CAAC,UAAU,EAAE,CAAC;QACnB,CAAC;aAAM,IAAI,EAAE,CAAC,OAAO,CAAC,OAAO,EAAE,iBAAiB,CAAC,IAAI,OAAO,KAAK,GAAG,EAAE,CAAC;YACtE,IAAI,CAAC,aAAa,GAAG,IAAI,CAAC,GAAG,CAAC,IAAI,CAAC,YAAY,CAAC,MAAM,GAAG,CAAC,EAAE,IAAI,CAAC,aAAa,GAAG,CAAC,CAAC,CAAC;YACpF,IAAI,CAAC,UAAU,EAAE,CAAC;QACnB,CAAC;aAAM,IAAI,EAAE,CAAC,OAAO,CAAC,OAAO,EAAE,oBAAoB,CAAC,IAAI,OAAO,KAAK,IAAI,EAAE,CAAC;YAC1E,MAAM,QAAQ,GAAG,IAAI,CAAC,YAAY,CAAC,IAAI,CAAC,aAAa,CAAC,CAAC;YACvD,IAAI,QAAQ,EAAE,CAAC;gBACd,IAAI,CAAC,gBAAgB,CAAC,EAAE,OAAO,EAAE,QAAQ,CAAC,OAAO,EAAE,OAAO,EAAE,QAAQ,CAAC,OAAO,EAAE,CAAC,CAAC;YACjF,CAAC;QACF,CAAC;aAAM,IAAI,EAAE,CAAC,OAAO,CAAC,OAAO,EAAE,mBAAmB,CAAC,EAAE,CAAC;YACrD,IAAI,CAAC,gBAAgB,EAAE,CAAC;QACzB,CAAC;aAAM,CAAC;YACP,OAAO,KAAK,CAAC;QACd,CAAC;QACD,OAAO,IAAI,CAAC;IACb,CAAC;CACD","sourcesContent":["import { Container, getKeybindings, Spacer, Text } from \"@earendil-works/pi-tui\";\nimport {\n\tgetProjectTrustOptions,\n\tgetProjectTrustPath,\n\ttype ProjectTrustOption,\n\ttype ProjectTrustStoreEntry,\n} from \"../../../core/trust-manager.ts\";\nimport { theme } from \"../theme/theme.ts\";\nimport { DynamicBorder } from \"./dynamic-border.ts\";\nimport { keyHint, rawKeyHint } from \"./keybinding-hints.ts\";\n\nexport type TrustSelection = Pick<ProjectTrustOption, \"trusted\" | \"updates\">;\n\nexport interface TrustSelectorOptions {\n\tcwd: string;\n\tsavedDecision: ProjectTrustStoreEntry | null;\n\tprojectTrusted: boolean;\n\tonSelect: (selection: TrustSelection) => void;\n\tonCancel: () => void;\n}\n\nfunction formatDecision(cwd: string, decision: ProjectTrustStoreEntry | null): string {\n\tif (decision === null) {\n\t\treturn \"none\";\n\t}\n\tconst label = decision.decision ? \"trusted\" : \"untrusted\";\n\tif (decision.path !== getProjectTrustPath(cwd)) {\n\t\treturn `${label} (inherited from ${decision.path})`;\n\t}\n\treturn `${label} (${decision.path})`;\n}\n\nexport class TrustSelectorComponent extends Container {\n\tprivate selectedIndex: number;\n\tprivate readonly listContainer: Container;\n\tprivate readonly trustOptions: ProjectTrustOption[];\n\tprivate readonly savedDecision: ProjectTrustStoreEntry | null;\n\tprivate readonly onSelectCallback: (selection: TrustSelection) => void;\n\tprivate readonly onCancelCallback: () => void;\n\n\tconstructor(options: TrustSelectorOptions) {\n\t\tsuper();\n\n\t\tthis.savedDecision = options.savedDecision;\n\t\tthis.trustOptions = getProjectTrustOptions(options.cwd);\n\t\tthis.selectedIndex = Math.max(\n\t\t\t0,\n\t\t\tthis.trustOptions.findIndex((option) => this.isSavedOption(option)),\n\t\t);\n\t\tthis.onSelectCallback = options.onSelect;\n\t\tthis.onCancelCallback = options.onCancel;\n\n\t\tthis.addChild(new DynamicBorder());\n\t\tthis.addChild(new Spacer(1));\n\t\tthis.addChild(new Text(theme.fg(\"accent\", theme.bold(\"Project trust\")), 1, 0));\n\t\tthis.addChild(new Text(theme.fg(\"muted\", options.cwd), 1, 0));\n\t\tthis.addChild(new Spacer(1));\n\t\tthis.addChild(\n\t\t\tnew Text(theme.fg(\"muted\", `Saved decision: ${formatDecision(options.cwd, options.savedDecision)}`), 1, 0),\n\t\t);\n\t\tthis.addChild(\n\t\t\tnew Text(theme.fg(\"muted\", `Current session: ${options.projectTrusted ? \"trusted\" : \"untrusted\"}`), 1, 0),\n\t\t);\n\t\tthis.addChild(new Spacer(1));\n\n\t\tthis.listContainer = new Container();\n\t\tthis.addChild(this.listContainer);\n\t\tthis.addChild(new Spacer(1));\n\t\tthis.addChild(\n\t\t\tnew Text(\n\t\t\t\trawKeyHint(\"↑↓\", \"navigate\") +\n\t\t\t\t\t\" \" +\n\t\t\t\t\tkeyHint(\"tui.select.confirm\", \"save\") +\n\t\t\t\t\t\" \" +\n\t\t\t\t\tkeyHint(\"tui.select.cancel\", \"cancel\"),\n\t\t\t\t1,\n\t\t\t\t0,\n\t\t\t),\n\t\t);\n\t\tthis.addChild(new Spacer(1));\n\t\tthis.addChild(new DynamicBorder());\n\n\t\tthis.updateList();\n\t}\n\n\tprivate isSavedOption(option: ProjectTrustOption): boolean {\n\t\treturn (\n\t\t\toption.savedPath !== undefined &&\n\t\t\tthis.savedDecision?.decision === option.trusted &&\n\t\t\tthis.savedDecision.path === option.savedPath\n\t\t);\n\t}\n\n\tprivate updateList(): void {\n\t\tthis.listContainer.clear();\n\t\tfor (let i = 0; i < this.trustOptions.length; i++) {\n\t\t\tconst option = this.trustOptions[i];\n\t\t\tif (!option) {\n\t\t\t\tcontinue;\n\t\t\t}\n\n\t\t\tconst isSelected = i === this.selectedIndex;\n\t\t\tconst isCurrent = this.isSavedOption(option);\n\t\t\tconst currentMarker = isCurrent ? theme.fg(\"accent\", \" \") : \" \";\n\t\t\tconst prefix = isSelected ? theme.fg(\"accent\", \"→ \") : \" \";\n\t\t\tconst label = isSelected ? theme.fg(\"accent\", option.label) : theme.fg(\"text\", option.label);\n\t\t\tthis.listContainer.addChild(new Text(`${prefix}${currentMarker}${label}`, 1, 0));\n\t\t}\n\t}\n\n\thandleInput(keyData: string): boolean {\n\t\tconst kb = getKeybindings();\n\t\tif (kb.matches(keyData, \"tui.select.up\") || keyData === \"k\") {\n\t\t\tthis.selectedIndex = Math.max(0, this.selectedIndex - 1);\n\t\t\tthis.updateList();\n\t\t} else if (kb.matches(keyData, \"tui.select.down\") || keyData === \"j\") {\n\t\t\tthis.selectedIndex = Math.min(this.trustOptions.length - 1, this.selectedIndex + 1);\n\t\t\tthis.updateList();\n\t\t} else if (kb.matches(keyData, \"tui.select.confirm\") || keyData === \"\\n\") {\n\t\t\tconst selected = this.trustOptions[this.selectedIndex];\n\t\t\tif (selected) {\n\t\t\t\tthis.onSelectCallback({ trusted: selected.trusted, updates: selected.updates });\n\t\t\t}\n\t\t} else if (kb.matches(keyData, \"tui.select.cancel\")) {\n\t\t\tthis.onCancelCallback();\n\t\t} else {\n\t\t\treturn false;\n\t\t}\n\t\treturn true;\n\t}\n}\n"]}
@@ -88,6 +88,8 @@ The effective parameters appear in extension events and successful results:
88
88
 
89
89
  `preserve_recent` counts context-visible messages without aligning the boundary to a user turn. An assistant message or tool result may therefore begin the kept tail. Because such a tail can start or end mid-turn, the kept messages are not replayed as structured message blocks: they are serialized with the same transcript grammar as the compacted region and appended to the end of the boundary string, so the whole boundary reaches the provider as one message. Serialization of the kept tail is lossless — tool results keep their full text instead of being truncated at 16k characters, and images stay attached as image blocks rather than becoming `[image]` markers — so protected content is preserved, not merely summarized. A value of `0` protects no messages and makes the entire active transcript compactable. If `query` is absent, Atomic derives it from the last visible user message.
90
90
 
91
+ One consequence is worth stating for Claude models that sign their reasoning. Because the kept tail is serialized into the boundary string rather than replayed as structured assistant messages, no `thinking` or `redacted_thinking` block survives a compaction boundary. Compaction therefore **intentionally resets the signed reasoning chain**: reasoning produced before a boundary is not carried across it. This is deliberate, and it is the first of the two remedies Anthropic documents for keep-tail compaction — carry the text and tool calls across, leave the thinking blocks behind — reached structurally rather than by a stripping pass. The tail's own content is unaffected: text, tool calls, and tool results cross the boundary losslessly. See [Preserved thinking and model switches](/models#preserved-thinking-and-model-switches) for how Atomic handles prefix changes *between* boundaries, which is a separate mechanism.
92
+
91
93
  The query is used whole and is never truncated. This matters for structured prompts: a truncated query would make section order the retention policy, because only the leading section could influence what the planner kept, and a constraint stated later in the prompt could not. Long queries are safe — an oversized planner request surfaces as an explicit provider-overflow failure rather than silent truncation — but `keepContext` tags, not query length, are the way to guarantee a span survives.
92
94
 
93
95
  Configure defaults in `~/.atomic/agent/settings.json` or `.atomic/settings.json`:
@@ -721,6 +721,7 @@ interface ProviderModelConfig {
721
721
  sessionAffinityFormat?: "openai" | "openai-nosession" | "openrouter";
722
722
  supportsLongCacheRetention?: boolean;
723
723
  supportsToolSearch?: boolean;
724
+ supportsMaxOutputTokens?: boolean;
724
725
  };
725
726
  }
726
727
  ```
@@ -733,4 +734,4 @@ The `cost` shape is equivalent to `Model<Api>["cost"]`. Base rates and every tie
733
734
 
734
735
  Capability flags are enforcement claims, not preferences. `supportsStrictMode` controls strict JSON-schema tools for OpenAI-compatible APIs; Anthropic/Bedrock use `supportsStrictTools`; `supportsOpenAIGrammarTools` controls OpenAI Lark/regex custom tools. Atomic also accepts `supportsGrammarTools` as a compatibility alias and synchronizes it to the canonical OpenAI name; when both disagree, the canonical field wins. Leave these fields unset/false unless the endpoint and selected model actually preserve and enforce the corresponding request shape. See [Extensions](/extensions#constrained-sampling) for exact `constrainedSampling` modes.
735
736
 
736
- For `openai-responses` providers, set `compat.sessionAffinityFormat` to `"openai"` for `session_id` plus `x-client-request-id`, `"openai-nosession"` to omit `session_id` while retaining `x-client-request-id`, or `"openrouter"` for `x-session-id`. Responses-compatible providers may also set `supportsToolSearch` when they support deferred tool loading.
737
+ For `openai-responses` providers, set `compat.sessionAffinityFormat` to `"openai"` for `session_id` plus `x-client-request-id`, `"openai-nosession"` to omit `session_id` while retaining `x-client-request-id`, or `"openrouter"` for `x-session-id`. Responses-compatible providers may also set `supportsToolSearch` when they support deferred tool loading. `supportsMaxOutputTokens` defaults to `true`; set it to `false` for OpenAI Responses-compatible gateways such as Codex-protocol proxies that reject `max_output_tokens` with a 400, and Atomic omits the parameter from those requests.
@@ -1043,6 +1043,8 @@ UI methods for user interaction. See [Custom UI](#custom-ui) for full details.
1043
1043
 
1044
1044
  Current working directory.
1045
1045
 
1046
+ Built-in cwd-sensitive tools (`read`, `write`, `edit`, `search`, `find`, `ls`, `bash`, `powershell`) resolve relative paths against `ctx.cwd` when an extension invokes them, falling back to the cwd captured when the tool was created. An extension that registers a tool and forwards its own context therefore gets paths resolved against the live session cwd rather than a stale one.
1047
+
1046
1048
  Use `CONFIG_DIR_NAME` instead of hardcoding `.atomic` (or legacy `.pi`) when constructing project-local config paths. Rebranded distributions can use a different config directory name.
1047
1049
 
1048
1050
  ```typescript
package/docs/intercom.md CHANGED
@@ -472,7 +472,7 @@ graph TB
472
472
  B2 <-->|Local Socket/Pipe| B3
473
473
  ```
474
474
 
475
- The broker is a standalone process that manages session registration and message routing. It auto-spawns when the first session that invokes Intercom needs it and exits 5 seconds after it last has no registered sessions, including brokers that never received a connection and sockets that close before register; clients reconnect automatically if the broker restarts. A spawn lock keyed by PID and timestamp prevents duplicate brokers when multiple sessions start at once.
475
+ The broker is a standalone process that manages session registration and message routing. It auto-spawns when the first session that invokes Intercom needs it and exits 5 seconds after it last has no registered sessions, including brokers that never received a connection and sockets that close before register; clients reconnect automatically if the broker restarts. A reconnect that fails schedules the next attempt on a bounded backoff (1s, 2s, 5s, 10s, then 30s) and keeps retrying until the session connects or shuts down, so recovery never waits for an explicit Intercom call. A failed explicit `intercom` or overlay connection surfaces its error to the caller and still leaves that background retry in place. A reconnect that fails after the broker already accepted it closes that connection first, so a session never appears twice in `intercom list`. A spawn lock keyed by PID and timestamp prevents duplicate brokers when multiple sessions start at once.
476
476
 
477
477
  Transport is local IPC only — a Unix domain socket on macOS/Linux or a named pipe on Windows — using length-prefixed JSON (4-byte length + payload) with request correlation for session listing, explicit delivery failures, and validation of malformed or out-of-order messages. `ask` stays client-side: the broker routes plain messages, and the client waits for the matching reply before returning it as the tool result.
478
478
 
@@ -8,7 +8,7 @@ description: "The external benchmarks that inform Atomic model selection — Art
8
8
  Atomic's model-selection docs are keyed to two live external benchmark sources rather than a hand-maintained table of scores. This page lists each benchmark, what it measures, and **when to reference it** for a given workflow role — so the docs stay useful as new models ship without a manual rewrite every time.
9
9
 
10
10
  <Warning>
11
- No single benchmark is the source of truth. Use these as inputs and validate against Atomic's own workflow evals — public suites test different task distributions than real engineering loops. When Atomic's numbers disagree with a public index, Atomic's evals win. **Last reviewed: 2026-08-21.**
11
+ No single benchmark is the source of truth. Use these as inputs and validate against Atomic's own workflow evals — public suites test different task distributions than real engineering loops. When Atomic's numbers disagree with a public index, Atomic's evals win. The DeepSWE snapshot used by the linked model-selection pages was updated August 26, 2026. **Last reviewed: 2026-09-01.**
12
12
  </Warning>
13
13
 
14
14
  ## The two sources at a glance
@@ -22,6 +22,7 @@ No single benchmark is the source of truth. Use these as inputs and validate aga
22
22
 
23
23
  DeepSWE is the closest public proxy for what Atomic actually does. Tasks are written from scratch (not scraped from PRs), so no model has seen the solutions; solutions require substantially more code than SWE-bench-style suites; and verifiers test behavior rather than implementation.
24
24
 
25
+ - **Current snapshot:** DeepSWE v1.1, 113 tasks across 91 repositories and 5 languages, updated August 26, 2026. The site reports 26 measured models and displays 19 leaderboard rows.
25
26
  - **Metric:** `pass@1`, plus average cost per task, output tokens, and agent steps.
26
27
  - **When to reference:** default weighting for debugger, worker, and any code-writing role. This is the table that drives [Model Selection](/models/model-selection) and [Pareto Efficiency](/models/pareto-efficiency).
27
28
  - **Watch:** cost and step count, not just score — a model that passes but takes 268 steps (e.g. sonnet-5) is a poor worker even at a good pass rate.
@@ -14,7 +14,7 @@ This page gives workflow authors and runtime policy code a practical way to answ
14
14
  It is a **static reference**. It does not change runtime model routing — routing is configured elsewhere. Treat these recommendations as a starting point and validate against your own workflow evals.
15
15
 
16
16
  <Note>
17
- The table below is a snapshot of the [DeepSWE](https://deepswe.datacurve.ai/) leaderboard (v1.1, highest published thinking level per model), a long-horizon coding-agent benchmark reporting `pass@1` and average dollars per task. Benchmarks and pricing drift and new models ship constantly, so **treat the live leaderboards as authoritative** and refresh this page from them rather than hand-maintaining scores. See [Benchmark sources & when to reference each](/models/artificial-analysis-index). **Last compiled: 2026-08-21.**
17
+ The table below is a snapshot of the [DeepSWE](https://deepswe.datacurve.ai/) leaderboard (v1.1, highest published thinking level per model), a long-horizon coding-agent benchmark reporting `pass@1` and average dollars per task. The source reports 113 tasks and was updated August 26, 2026. Benchmarks and pricing drift and new models ship constantly, so **treat the live leaderboards as authoritative** and refresh this page from them rather than hand-maintaining scores. See [Benchmark sources & when to reference each](/models/artificial-analysis-index). **Last compiled: 2026-09-01.**
18
18
  </Note>
19
19
 
20
20
  ## Benchmark levels are measurement settings
@@ -30,38 +30,47 @@ reports the ambiguity. Use `--provider <provider> --model <id>` or `--model <pro
30
30
 
31
31
  ## Recommendation chart
32
32
 
33
- The current highest-effort-config Pareto frontier is **claude-opus-5** (accuracy ceiling), **gpt-5.6-sol**, **glm-5.3**, **gpt-5.6-luna**, **deepseek-v4-pro**, and **deepseek-v4-flash**. Everything else is dominated on DeepSWE cost and accuracy and earns a place only through role fit or provider diversity. For the frontier reasoning, see [Pareto Efficiency](/models/pareto-efficiency).
33
+ The current highest-effort-config Pareto frontier is **claude-opus-5** (accuracy ceiling), **gpt-5.6-sol**, **glm-5.3**, **gpt-5.6-luna**, and **glm-5.3-flash** (cheapest point). Everything else displayed on the live DeepSWE leaderboard is dominated on cost and accuracy and earns a place only through role fit or provider diversity. For the frontier reasoning, see [Pareto Efficiency](/models/pareto-efficiency).
34
34
 
35
35
  | Model [benchmark measurement level] | pass@1 | $/task | Verdict | Use it for |
36
36
  | --- | --- | --- | --- | --- |
37
37
  | claude-opus-5 [max] | 74% | $11.84 | Accuracy ceiling / frontier | Final approval and the hardest debugging when one more point can justify the cost |
38
- | gpt-5.6-sol [max] | 73% | $8.39 | Frontier | High-cost judgment gates; nearly the top score at lower cost than Opus 5 |
39
- | gpt-5.6-terra [max] | 70% | $3.96 | Off the live board | Last published measurement (July pricing); no longer displayed on the live leaderboard as of the August 20 refresh re-verify against the source before relying on it |
38
+ | gpt-5.6-sol [max] | 73% | $6.46 | Frontier | High-cost judgment gates; nearly the top score for about half the task cost of Opus 5 |
39
+ | gpt-5.6-terra [max] | 70% | $3.96 | Historical — off the live board | Last published measurement; not displayed on the August 26 leaderboard, so re-verify before relying on it |
40
40
  | claude-fable-5 [max] | 70% | $21.63 | Drop | Sol matches or beats its score for much less |
41
- | glm-5.3 [max] | 69% | $3.99 | Frontier — open-weights value | Best open-weights cost/accuracy point; matches Kimi K3's rounded score for less |
41
+ | glm-5.3 [max] | 69% | $3.99 | Frontier — open-weights value | Best open-weights mid-tier cost/accuracy point; matches Kimi K3's rounded score for less |
42
42
  | kimi-k3 [max] | 69% | $4.65 | Dominated | GLM-5.3 matches its rounded score for $0.66 less; Moonshot-family diversity only |
43
43
  | gpt-5.6-luna [max] | 67% | $0.61 | Frontier — best general value | Research, orchestration, workers, and code simplification |
44
44
  | gpt-5.5 [xhigh] | 67% | $7.23 | Superseded | Luna matches its score for less than one tenth of the task cost |
45
45
  | grok-4.6 [xhigh] | 67% | $5.50 | Provider fallback | xAI diversity; Luna has the same rounded score at lower DeepSWE task cost |
46
46
  | gemini-3.7-flash [high] | 65% | $2.18 | Provider fallback | Strong Google-family result, but Luna is cheaper and more accurate |
47
- | deepseek-v4-pro [max] | 63% | $0.24 | Frontier (budget) | Low-cost work where its 155-step average remains acceptable |
47
+ | glm-5.3-flash [max] | 63% | $0.24 | Frontier cheapest | Budget worker loops that can accept lower accuracy and 123 average steps |
48
+ | deepseek-v4-pro [max] | 63% | $1.67 | Dominated / provider fallback | DeepSeek diversity only; GLM-5.3 Flash has a higher unrounded score, fewer steps, and about one seventh of the cost |
48
49
  | claude-opus-4.8 [max] | 59% | $13.22 | Fallback only | Anthropic diversity and long-context behavior, not cost efficiency |
49
- | qwen3.8-max [xhigh] | 57% | $3.73 | Provider fallback | Qwen diversity only; DeepSeek Pro and Luna dominate it |
50
- | muse-spark-1.2 [xhigh] | 55% | $3.70 | Drop | DeepSeek Pro is cheaper and more accurate |
50
+ | qwen3.8-max [xhigh] | 57% | $3.73 | Provider fallback | Qwen diversity only; GLM-5.3 Flash and Luna dominate it |
51
+ | muse-spark-1.2 [xhigh] | 55% | $3.70 | Drop | GLM-5.3 Flash is cheaper and more accurate |
51
52
  | claude-sonnet-5 [max] | 54% | $26.40 | Drop everywhere | Highest task cost and 268 average steps for a mid-table score |
52
- | grok-4.5 [high] | 54% | $2.42 | Superseded | Grok 4.6 adds 13 points; cheaper frontier models still dominate both generations |
53
- | deepseek-v4-flash [max] | 53% | $0.10 | Frontier (cheapest) | Very cheap mechanical loops that can tolerate lower accuracy and 153 average steps |
54
- | muse-spark-1.1 [xhigh] | 53% | $2.36 | Superseded | Replaced by Muse Spark 1.2 and dominated by DeepSeek Pro |
55
- | gpt-5.4 [xhigh] | 52% | $5.65 | Superseded | Luna is cheaper and 15 points more accurate |
53
+ | grok-4.5 [high] | 54% | $2.42 | Historical — off the live board | Last published measurement; superseded by Grok 4.6 and dominated by current frontier models |
54
+ | deepseek-v4-flash [max] | 53% | $0.46 | Dominated / provider fallback | DeepSeek diversity only; GLM-5.3 Flash is ten points more accurate for about half the cost |
55
+ | muse-spark-1.1 [xhigh] | 53% | $2.36 | Historical — off the live board | Last published measurement; replaced by Muse Spark 1.2 and dominated by current frontier models |
56
+ | gpt-5.4 [xhigh] | 52% | $5.65 | Historical — off the live board | Last published measurement; Luna is cheaper and 15 points more accurate |
56
57
  | gemini-3.6-flash [high] | 47% | $2.21 | Drop from reasoning | Superseded by Gemini 3.7 Flash |
57
58
  | glm-5.2 [max] | 44% | $3.92 | Superseded | Measured predecessor only; do not relabel this as GLM-5.3 |
58
59
  | gemini-3.5-flash [high] | 36% | $3.45 | Drop from reasoning | Retain only where a low-effort retrieval role has separate evidence |
59
- | kimi-k2.7-code [low] | 31% | $2.19 | Superseded | Kimi K3 is the current family fallback |
60
- | claude-sonnet-4.6 [high] | 30% | $5.52 | Drop everywhere | Removed from all chains |
61
- | gemini-3.1-pro [high] | 12% | $2.14 | Drop everywhere | Removed from all chains |
60
+ | kimi-k2.7-code | 31% | $2.82 | Historical — off the live board | Last published measurement had no effort level; Kimi K3 is the current family fallback |
61
+ | claude-sonnet-4.6 [high] | 30% | $5.52 | Historical off the live board | Last published measurement; removed from all chains |
62
+ | gemini-3.1-pro-preview [high] | 12% | $2.14 | Historical off the live board | Last published measurement; removed from all chains |
62
63
 
63
64
  <Note>
64
- DeepSWE values above use the v1.1 results and reporting corrections published through August 20, 2026; `pass@1` is rounded as on the live leaderboard and confidence intervals are omitted here. The highest published thinking level is a measurement choice, not a production default. GPT-5.6 Terra's July measurement no longer appears on the live board, so its row keeps the last published values. See the live page for intervals, output tokens, steps, lower-effort configurations, and later corrections.
65
+ DeepSWE values above use the v1.1 results displayed on the August 26, 2026 leaderboard, including the August 21 pricing corrections for GPT-5.6 Sol and DeepSeek V4. Sol's cost reflects OpenAI's promotional input and output price cut through at least November 21, 2026. DeepSWE uses DeepSeek's peak rates; its off-peak rates are half as much. `pass@1` is rounded as on the live leaderboard and confidence intervals are omitted here. The highest published thinking level is a measurement choice, not a production default. Seven historical configurations are retained with their last published values because they are no longer displayed: GPT-5.6 Terra, Grok 4.5, Muse Spark 1.1, GPT-5.4, Kimi K2.7 Code, Claude Sonnet 4.6, and Gemini 3.1 Pro Preview. See the live page for intervals, output tokens, steps, lower-effort configurations, and later corrections.
66
+ </Note>
67
+
68
+ <Note>
69
+ **Claude Fable 5.1 is in Atomic's catalog and is not in the table above.** It was released September 1, 2026, after the August 26, 2026 DeepSWE snapshot this page is compiled from, so it has no measured `pass@1` or `$/task` here. Do not read the `claude-fable-5` row as a Fable 5.1 result: the two models differ in price and behavior, and an unmeasured model must not inherit its predecessor's score. Benchmark it on your own workflow evals before promoting it into a stage.
70
+
71
+ What is source-backed for `claude-fable-5-1` today, from [Anthropic's model overview](https://platform.claude.com/docs/en/models/fable-5-1/overview): a 1M-token context window and 128K maximum output; adaptive thinking that is always on, with effort `low`, `medium`, `high`, `xhigh`, and `max` and an Anthropic default of `high`; a June 2026 knowledge cutoff; and $10 input, $50 output, $12.50 five-minute cache write, $20 one-hour cache write, and $0.25 cache read per million tokens. The cache read is a quarter of Fable 5's $1.00, which is the main pricing reason to prefer it for long agentic sessions that re-read a cached prefix. Non-default `temperature`, `top_p`, and `top_k` return a 400 on every request, so Atomic omits `temperature` for this model.
72
+
73
+ Atomic generates Fable 5.1 for the providers it has a matching runtime integration for. At the time of writing that is Anthropic, three Amazon Bedrock inference profiles (`anthropic.`, `global.`, and `us.`), OpenRouter, and the Vercel AI Gateway; a provider "latest" alias such as OpenRouter's `~anthropic/claude-fable-latest` may also route to it without naming it. That set genuinely moves — opencode zen published the model and then withdrew it while this page was being written — so run `workflow({ action: "models" })` or `--list-models` for the current list rather than trusting this one. Published catalogs also list the model on Google Vertex, Google Vertex (Anthropic), Azure, and Azure Cognitive Services; Atomic has no Claude runtime integration for those providers and generates no entries for them, which is a current limitation rather than a roadmap commitment. What does *not* vary is the invariant that matters: **Atomic's preserved-thinking handling is scoped to `provider: "anthropic"` on the `anthropic-messages` API and applies to none of the other mirrors** — including the Vercel AI Gateway, which rides `anthropic-messages` but is deliberately excluded. See [Preserved thinking and model switches](/models#preserved-thinking-and-model-switches).
65
74
  </Note>
66
75
 
67
76
  ## Role-based thinking effort
@@ -82,12 +91,12 @@ Reserve `max` for a high-cost-of-error role or an explicit user request. An expl
82
91
  Pick by the cost of being wrong in each role, not by raw accuracy. Match the role to the benchmark that best measures it (see [Benchmark sources](/models/artificial-analysis-index)).
83
92
 
84
93
  - **Reviewer / judgment gates** — use `max` when the reviewer makes a security, identity, adversarial, or final-approval decision whose wrong verdict discards an entire loop. `claude-opus-5` is the DeepSWE accuracy ceiling; `gpt-5.6-sol` is the lower-cost near-peer. Use another family when decorrelated errors matter.
85
- - **Codebase mapping / planner** — start at `high` for repository mapping, lifecycle analysis, compatibility, and plans. `gpt-5.6-sol` is the strongest top-tier value at its measured `max` configuration, and `glm-5.3` carries the open-weights mid tier; raise production effort to `max` only when the plan gates a high-cost loop or the user asks for it.
94
+ - **Codebase mapping / planner** — start at `high` for repository mapping, lifecycle analysis, compatibility, and plans. `gpt-5.6-sol` is the strongest top-tier value at its measured `max` configuration, and `glm-5.3` holds the open-weights mid tier; raise production effort to `max` only when the plan gates a high-cost loop or the user asks for it.
86
95
  - **Debugger / triage / repair** — start at `high`; deep reasoning pays off when root-causing or repairing is costly. Weight DeepSWE and Terminal-Bench together rather than treating either as a complete measure.
87
96
  - **Research / synthesis** — use `high` for demanding research and evidence reconciliation; use `medium` for routine synthesis when the evidence is already strong. `gpt-5.6-luna` remains the workhorse. Benchmark to weight: AA-LCR and AA-Omniscience.
88
- - **Orchestrator / worker / cheap loops** — Luna offers the best broad cost/accuracy balance. DeepSeek V4 Pro and Flash occupy the budget frontier but take 155 and 153 steps on average; use them only when that longer path fits. Use another family when provider diversity matters.
97
+ - **Orchestrator / worker / cheap loops** — Luna offers the best broad cost/accuracy balance. GLM-5.3 Flash is the cheapest live frontier point at 63% for $0.24 with 123 average steps. DeepSeek V4 Pro and Flash are provider-diversity options, not budget-frontier choices.
89
98
  - **User-impact review / final reporting** — use `medium` for impact summaries and reports that preserve the evidence needed by the user. Do not spend `max` here unless the user explicitly requests it or the role has become a high-cost-of-error approval.
90
- - **Design** — a quality-first, unbenchmarked domain; keep a top-tier model (`gpt-5.6-sol` or `claude-fable-5`) when the design decision has high failure cost, and choose effort by the review or approval role rather than by the benchmark row.
99
+ - **Design** — a quality-first, unbenchmarked domain; keep a top-tier model (`gpt-5.6-sol` or `claude-fable-5`) when the design decision has high failure cost, and choose effort by the review or approval role rather than by the benchmark row. `claude-fable-5-1` is the newer Anthropic model in this family and is also unmeasured here; treat it as a candidate to evaluate rather than a drop-in replacement, and do not carry Fable 5's row over to it.
91
100
  - **Interactive coding sessions** — use `high` for complex, multi-step coding and `medium` for routine edits; reserve `max` for a high-cost-of-error judgment or an explicit user request.
92
101
  - **Deterministic checks** — make typechecks, tests, schema validation, runtime probes, and artifact inspection tool nodes with no model call. Model self-report is not verification evidence.
93
102
 
@@ -5,58 +5,66 @@ description: "Cost-vs-accuracy frontier for model selection: which models domina
5
5
 
6
6
  # Pareto Efficiency
7
7
 
8
- A model is **Pareto-efficient** (on the frontier) if no other model is both cheaper and more accurate. Everything not on the frontier is **dominated** some other option matches or beats it on accuracy for less money and should be avoided unless it earns a slot through a specific role fit or provider diversity.
8
+ A model is **Pareto-efficient** (on the frontier) if no other model is both cheaper and more accurate. Everything not on the frontier is **dominated**: some other option matches or beats it on accuracy for less money. Avoid a dominated model unless it earns a slot through a specific role fit or provider diversity.
9
9
 
10
10
  The axes here are `pass@1` (accuracy) and `average dollars per task` (cost), taken from the [DeepSWE](https://deepswe.datacurve.ai/) coding-agent leaderboard. For the full table and role guidance, see [Model Selection](/models/model-selection).
11
11
 
12
12
  <Note>
13
- Figures are a snapshot of DeepSWE v1.1 using the highest published thinking level per model and reporting corrections through August 20, 2026. The frontier moves whenever a model, run, or price changes DeepSWE publishes a live cost-vs-score scatter, so **read the frontier off the live chart** rather than trusting a static list. **Last compiled: 2026-08-21.**
13
+ Figures are a snapshot of DeepSWE v1.1 using the highest published thinking level for each of the 19 models displayed on the August 26, 2026 leaderboard. They include the August 21 pricing corrections for GPT-5.6 Sol and DeepSeek V4. DeepSWE publishes a live cost-vs-score scatter, so **read the frontier off the live chart** rather than trusting a static list. **Last compiled: 2026-09-01.**
14
14
  </Note>
15
15
 
16
16
  ## The frontier
17
17
 
18
- Six highest-effort model configurations currently sit on the frontier, from the cheapest measured task cost to the accuracy ceiling:
18
+ Five displayed highest-effort model configurations sit on the frontier, from the cheapest measured task cost to the accuracy ceiling:
19
19
 
20
- - **deepseek-v4-flash [max]** — 53% for $0.10. The cheapest point, with lower accuracy and 153 average steps.
21
- - **deepseek-v4-pro [max]** — 63% for $0.24. A large accuracy gain for another $0.14 per task, with 155 average steps.
22
- - **gpt-5.6-luna [max]** — 67% for $0.61. The best broad value on the board.
23
- - **glm-5.3 [max]** — 69% for $3.99. The open-weights mid-tier point; matches Kimi K3's rounded score for less.
24
- - **gpt-5.6-sol [max]** — 73% for $8.39. The lower-cost near-peer to the accuracy leader.
25
- - **claude-opus-5 [max]** — 74% for $11.84. The current accuracy ceiling.
20
+ - **glm-5.3-flash [max]**: 63% for $0.24 with 123 average steps. This is the cheapest point.
21
+ - **gpt-5.6-luna [max]**: 67% for $0.61. This is the best broad value on the board.
22
+ - **glm-5.3 [max]**: 69% for $3.99 with 124 average steps. This is the open-weights mid-tier point and matches Kimi K3's rounded score for less.
23
+ - **gpt-5.6-sol [max]**: 73% for $6.46 with 61 average steps. This is the lower-cost near-peer to the accuracy leader.
24
+ - **claude-opus-5 [max]**: 74% for $11.84 with 99 average steps. This is the current accuracy ceiling.
26
25
 
27
- ## What changed — the frontier moved
26
+ ## What changed
28
27
 
29
- The August 20 refresh reshuffled the middle of the frontier:
28
+ The August 26 snapshot moves the budget end of the frontier and lowers the cost of its upper end:
30
29
 
31
- - **GLM-5.3 [max]** is now measured 69% for $3.99 with 124 average steps and takes the mid-tier frontier slot.
32
- - **GPT-5.6 Terra [max]** no longer appears on the live board; its July measurement (70% for $3.96) is retained in [Model Selection](/models/model-selection) as history, not a current frontier point.
33
- - **Kimi K3 [max]** is now strictly dominated: GLM-5.3 matches its rounded score for $0.66 less per task.
30
+ - **GLM-5.3 Flash [max]** now appears at 63% for $0.24 with 123 average steps. It replaces both DeepSeek V4 configurations on the budget frontier.
31
+ - **DeepSeek V4 Pro [max]** now costs $1.67 per task after DeepSeek's August 16 price change. GLM-5.3 Flash has a higher unrounded score (63.4% versus 62.8%), costs about one seventh as much, and averages 32 fewer steps.
32
+ - **DeepSeek V4 Flash [max]** now costs $0.46 per task. GLM-5.3 Flash is ten rounded points more accurate and costs about half as much.
33
+ - **GPT-5.6 Sol [max]** now costs $6.46 per task after OpenAI's August 20 promotional price cut, down from $8.39 in the previous snapshot. The reduced input and output rates run through at least November 21, 2026.
34
34
 
35
- ## Dominated models and why
35
+ DeepSWE's August 21, 2026 changelog says these DeepSeek costs use peak rates and off-peak rates are half as much. DeepSeek V4 Pro remains dominated at either rate. At the off-peak rate, DeepSeek V4 Flash costs about $0.23, marginally less than GLM-5.3 Flash's $0.24, but remains ten rounded points less accurate; these pages report the frontier from DeepSWE's published peak-rate costs.
36
36
 
37
- - **claude-fable-5 [max]** — Sol is more accurate and much cheaper; GLM-5.3 comes within a point for less than one fifth of the task cost.
38
- - **kimi-k3 [max]** — GLM-5.3 matches its rounded score and is $0.66 cheaper; Kimi remains useful for Moonshot-family diversity.
39
- - **gpt-5.5 [xhigh]** and **grok-4.6 [xhigh]** Luna matches their rounded 67% for $0.61.
40
- - **gemini-3.7-flash [high]** Luna is two points more accurate and less than one third of its task cost.
41
- - **grok-4.5 [high]**, **muse-spark-1.1 [xhigh]**, **muse-spark-1.2 [xhigh]**, and **gpt-5.4 [xhigh]** the new DeepSeek and GPT-5.6 points dominate these former budget choices.
42
- - **claude-opus-4.8 [max]** and **claude-sonnet-5 [max]** dominated on both cost and accuracy.
43
- - **qwen3.8-max [xhigh]**, **gemini-3.6-flash [high]**, **gemini-3.5-flash [high]**, **glm-5.2 [max]**, **kimi-k2.7-code [low]**, **claude-sonnet-4.6 [high]**, and **gemini-3.1-pro [high]** each has a cheaper, more accurate measured alternative.
37
+ ## Dominated models and why
38
+
39
+ - **deepseek-v4-pro [max]**: GLM-5.3 Flash has a higher unrounded score, costs $1.43 less, and averages 123 steps instead of 155.
40
+ - **deepseek-v4-flash [max]**: GLM-5.3 Flash is ten rounded points more accurate and costs $0.22 less.
41
+ - **claude-fable-5 [max]**: Sol is more accurate and much cheaper; GLM-5.3 comes within a point for less than one fifth of the task cost. This row is Fable 5 only; `claude-fable-5-1` released after this snapshot and has no measured position on the frontier.
42
+ - **kimi-k3 [max]**: GLM-5.3 matches its rounded score and is $0.66 cheaper; Kimi remains useful for Moonshot-family diversity.
43
+ - **gpt-5.5 [xhigh]** and **grok-4.6 [xhigh]**: Luna matches their rounded 67% for $0.61.
44
+ - **gemini-3.7-flash [high]**: Luna is two points more accurate and costs less than one third as much.
45
+ - **muse-spark-1.2 [xhigh]**: GLM-5.3 Flash is eight points more accurate and costs $3.46 less.
46
+ - **claude-opus-4.8 [max]** and **claude-sonnet-5 [max]**: each is dominated on both cost and accuracy.
47
+ - **qwen3.8-max [xhigh]**, **gemini-3.6-flash [high]**, **gemini-3.5-flash [high]**, and **glm-5.2 [max]**: each has a cheaper, more accurate displayed alternative.
48
+
49
+ Seven measured configurations are no longer displayed on the live leaderboard and are excluded from this current frontier calculation. [Model Selection](/models/model-selection) keeps their last published values as clearly labeled history: GPT-5.6 Terra, Grok 4.5, Muse Spark 1.1, GPT-5.4, Kimi K2.7 Code, Claude Sonnet 4.6, and Gemini 3.1 Pro Preview.
44
50
 
45
51
  ## Diversity and role-fit exceptions
46
52
 
47
53
  Efficiency is not the only axis. A dominated model can still earn a slot when it decorrelates errors or fills a niche:
48
54
 
49
- - **grok-4.6** retained as the operational xAI and OpenRouter provider-diversity fallback.
50
- - **glm-5.2 [max]** kept only as a measured predecessor; its results are never relabeled as GLM-5.3, which is now measured directly on the frontier.
51
- - **kimi-k3** retained as a Moonshot-family provider-diversity option despite GLM-5.3's strict DeepSWE dominance.
52
- - **claude-opus-4.8 [max]** retained where Anthropic diversity or its long-context behavior has separate value.
53
- - **claude-fable-5** kept where Anthropic-family behavior is specifically wanted, such as the quality-first, unbenchmarked design chain.
54
- - **Unmeasured models** a family without current DeepSWE or Artificial Analysis coverage may remain an operational default, but should not inherit a predecessor's score.
55
+ - **deepseek-v4-pro** and **deepseek-v4-flash** remain DeepSeek provider-diversity options, not budget-frontier choices.
56
+ - **grok-4.6** remains the operational xAI and OpenRouter provider-diversity fallback.
57
+ - **glm-5.2 [max]** remains only as a measured predecessor; its results are never relabeled as GLM-5.3 or GLM-5.3 Flash.
58
+ - **kimi-k3** remains a Moonshot-family provider-diversity option despite GLM-5.3's strict DeepSWE dominance.
59
+ - **claude-opus-4.8 [max]** remains useful where Anthropic diversity or its long-context behavior has separate value.
60
+ - **claude-fable-5** remains useful where Anthropic-family behavior is specifically wanted, such as the quality-first, unbenchmarked design chain.
61
+ - **claude-fable-5-1** is available in Atomic's catalog but released September 1, 2026, after the August 26 snapshot, so it is unmeasured here and holds no frontier position. Its published cache-read price is $0.25 per million tokens against Fable 5's $1.00, which can change the economics of a long cached-prefix session, but that is a price fact and not an accuracy result. Evaluate it before substituting it for a measured configuration.
62
+ - **Unmeasured models** may remain operational defaults when a family lacks current DeepSWE or Artificial Analysis coverage, but they should not inherit a predecessor's score.
55
63
 
56
64
  ## How to use this
57
65
 
58
66
  1. Default to a frontier model for the role's accuracy needs (see [Model Selection](/models/model-selection)).
59
- 2. Only reach for a dominated model when you have an explicit reason provider diversity, a long-context or token-price niche, or an unbenchmarked domain like design.
67
+ 2. Only reach for a dominated model when you have an explicit reason, such as provider diversity, a long-context or token-price niche, or an unbenchmarked domain like design.
60
68
  3. Re-read the frontier off the [DeepSWE live chart](https://deepswe.datacurve.ai/) when prices or benchmarks change, and update the timestamp on these pages.
61
69
 
62
70
  ## Related
package/docs/models.md CHANGED
@@ -218,6 +218,7 @@ In `models.json`, `headers` values must be strings. A `null` suppression marker
218
218
  Current behavior:
219
219
  - `/model`, `--list-models`, and the interactive footer display entries by model `id`.
220
220
  - The configured `name` is used for model matching and secondary model detail text. It does not replace the footer/status-bar model id.
221
+ - `input` lists the modalities **Atomic can send**. `["text"]`, `["text", "image"]`, and `["text", "image", "pdf"]` are the possible values. PDF is a platform capability rather than a per-model one — Anthropic documents that ["All active models support PDF processing"](https://platform.claude.com/docs/en/build-with-claude/pdf-support), routed through the same vision path as images — so upstream metadata carries it on every Claude entry. Atomic advertises it only where a runtime can serialize a document block: the Anthropic Messages and Amazon Bedrock Converse paths. A Claude mirror on any other provider stays at `["text", "image"]`, and a document sent to such a model is replaced by a visible placeholder rather than dropped silently. Note that Bedrock's Converse API needs citations enabled for full visual PDF understanding; without them it falls back to text extraction. `"pdf"` means PDF specifically: a document block's media type must be `application/pdf`, and any other value is rejected by name rather than sent mislabelled, because both request builders hardcode PDF rather than reading the field.
221
222
 
222
223
 
223
224
  ### Sampling Parameters
@@ -491,6 +492,48 @@ By default, Atomic sends per-tool `eager_input_streaming: true`. If a proxy or A
491
492
  | --------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
492
493
  | `supportsEagerToolInputStreaming` | Whether the provider accepts per-tool `eager_input_streaming`. Default: `true`. Set to `false` to omit that field and use the legacy fine-grained tool streaming beta header on tool-enabled requests. |
493
494
  | `supportsLongCacheRetention` | Whether the provider accepts Anthropic long cache retention (`cache_control.ttl: "1h"`) when cache retention is `long`. Default: `true`. |
495
+ | `delegatesThinkingModelBinding` | Whether the API decides for itself which thinking blocks the target model may read, dropping the rest. Default: `false`. See [Preserved thinking and model switches](#preserved-thinking-and-model-switches). |
496
+ | `enforcesPreservedThinkingBinding` | Whether the model rejects a thinking block replayed behind a changed conversation prefix. Default: `false`. When `true`, Atomic sends the `thinking-binding-controls-2026-08-01` beta header and `prefix_mismatch_behavior: "drop_block"`. |
497
+ | `supportsForcedToolChoice` | Whether the model accepts forced tool use (`tool_choice` `any` or a named tool). Default: `true`. When `false`, Atomic rejects a forced choice with an error rather than sending a request the model refuses. `auto` and `none` are never altered. |
498
+
499
+ `supportsForcedToolChoice` and `supportsTemperature` also exist on the Amazon Bedrock and OpenAI-compatible completions `compat` objects, with the same meanings and the same `true` defaults. Unlike the two preserved-thinking flags, which describe Anthropic's first-party endpoint, these describe the **model**, so Atomic applies them to every mirror that reaches it rather than only to `provider: "anthropic"`.
500
+
501
+ On the completions adapter, `supportsTemperature: false` also strips `temperature`, `top_p`, and `top_k` out of `samplingParams`. That merge is documented as last-wins so its keys override the named request fields, which means it would otherwise reopen exactly the parameters the model rejects. The strip runs after the merge, so it also covers a model-level `samplingParams` default, and it removes only those three keys — every other custom key you pass still overrides as before.
502
+
503
+ ### Forced tool use on Claude Fable 5.1
504
+
505
+ Claude Fable 5.1 rejects forced tool use — `tool_choice: {"type": "any"}` and `{"type": "tool", ...}` — on every request with a 400, whichever platform serves it. Anthropic's guidance is to use `tool_choice: {"type": "auto"}` with strict tool use or structured outputs instead.
506
+
507
+ Atomic's own agent loop only ever asks for `auto` or `none`, so no interactive session can reach this. It is reachable through the `@bastani/pi-ai` library's Anthropic, Bedrock, and OpenAI-completions entry points, which accept the wider tool-choice shape. For a model marked `supportsForcedToolChoice: false`, all three **fail the request with an error naming the model and the remedy**, before the round trip. That matters most on a gateway: OpenRouter drops parameters a model does not support, so an unguarded forced choice would vanish silently and return a plausible answer that ignored the instruction.
508
+
509
+ Atomic deliberately does not substitute `auto` on your behalf. Asking the model to call a specific tool and asking it to decide for itself are different requests, and silently swapping one for the other would discard an instruction you gave explicitly. If the substitution is what you want, make it yourself — branch on `compat.supportsForcedToolChoice` to decide. Every other model passes forced choices through unchanged, and `auto` and `none` are never altered on any model.
510
+
511
+ On the OpenAI-completions path the tool-choice union is wider than Anthropic's, and four of its members force a call: `"required"`, `{"type": "function", ...}`, `{"type": "custom", ...}`, and `{"type": "allowed_tools", "allowed_tools": {"mode": "required", ...}}`. All four are rejected. `allowed_tools` with `"mode": "auto"` is **not** rejected: OpenAI documents that mode as letting the model pick from the allowed tools *and generate a message*, so it narrows the candidate set rather than forcing a call, and it reaches the provider unchanged.
512
+
513
+ Two scoping details are worth knowing. **Claude Fable 5 is not restricted** — Anthropic names Fable 5.1 and Mythos 5.1 as the exceptions to forced tool use working, and OpenRouter's own metadata agrees, so the guard is version-scoped rather than family-scoped. That is the opposite of `supportsTemperature`, which Anthropic's sampling-parameter sentence applies to both Fable generations. And a provider "latest" alias such as OpenRouter's `~anthropic/claude-fable-latest` is **not** covered: its id names no version, so no rule keyed on the id can stay true if the alias re-points at a model that accepts forced tool use. If you use such an alias and need a forced choice guarded, pin the versioned id instead.
514
+
515
+ ### Preserved thinking and model switches
516
+
517
+ Anthropic binds every `thinking` and `redacted_thinking` block to the model that produced it, and on Claude Fable 5.1 also to the conversation prefix — the `system` prompt, the `tools` array, and every earlier message — that it was produced from. See [Preserved thinking](https://platform.claude.com/docs/en/build-with-claude/preserved-thinking).
518
+
519
+ Two separate checks follow from that, and Atomic handles them differently.
520
+
521
+ **The model check is the API's job.** A block is readable by the model that produced it or a newer one. Claude Fable 5.1 reads every earlier Claude model's blocks; no earlier model reads Fable 5.1's. A block the target model cannot read is always dropped by the API before the prompt reaches the model, unbilled, and the request succeeds. For first-party Anthropic models, Atomic therefore replays signed thinking blocks unchanged when you switch models mid-conversation and lets the API adjudicate:
522
+
523
+ - **Switching up** to Claude Fable 5.1 from another Claude model keeps the conversation's reasoning, because Fable 5.1 is allowed to read it.
524
+ - **Switching down** from Claude Fable 5.1 to an earlier Claude model drops that reasoning server-side, and the earlier model reasons again from the visible messages.
525
+
526
+ In both directions the visible assistant text, tool calls, and tool results are preserved exactly, so the conversation stays coherent. Atomic no longer rewrites another Claude model's reasoning into visible assistant text on a switch: that both discarded reasoning the newer model was entitled to read and destabilized the prefix later blocks are bound to.
527
+
528
+ **The conversation check can fail the request, so Atomic opts out of failing.** On Claude Fable 5.1, replaying a thinking block behind a changed prefix returns a 400. Anthropic enforces this by default for organizations created on or after August 31, 2026, which is why a session could fail on a new account but not an older one. Atomic sends the `thinking-binding-controls-2026-08-01` beta header with `thinking.block_binding.prefix_mismatch_behavior: "drop_block"` for that model, so a changed system prompt, a tool that appeared or disappeared, or a model switch drops the affected thinking blocks and the turn still answers. The field is sent on every request for that model, including turns with no reasoning level: the header alone would leave `prefix_mismatch_behavior` at its `"error"` default, which is the failure being avoided. When the API reports drops, Atomic records them on the assistant message's `diagnostics` array as an `anthropic_input_transformations` entry with the count, reasons, and block paths.
529
+
530
+ **Compaction is handled structurally, not by `drop_block`.** Atomic's client-side `preserve_recent` compaction serializes the protected tail into a single boundary message rather than replaying it as structured assistant and tool-result messages, so no signed thinking block survives a boundary to be replayed behind it. Compaction therefore **intentionally resets the signed reasoning chain**: reasoning produced before a boundary is not carried across it, while the tail's text, tool calls, and tool results are preserved losslessly. This is exactly the first remedy Anthropic documents for keep-tail compaction — strip `thinking` and `redacted_thinking` from turns you carry across and keep `text` and `tool_use` — reached by Atomic's transcript design rather than by a stripping pass. `drop_block` covers live prefix mismatches *between* boundaries; it is not what makes compaction safe. See [Compaction](/compaction).
531
+
532
+ **Server-side fallback leaves a boundary marker in the turn.** Claude Fable 5.1 is generated with the fallback targets Anthropic publishes for it — Claude Opus 4.8 and Claude Opus 5 — so a classifier refusal can be retried server-side on the same stream. When the decline happens partway through a response, the API emits a `fallback` content block marking where one model's output gives way to the next, then the fallback model continues. Atomic keeps that marker in the assistant turn as a `fallback` content block, and re-attributes the message to the serving model so usage is costed at that model's rates rather than the requested model's.
533
+
534
+ On the next turn the marker's **position** is load-bearing: Anthropic validates the surrounding thinking blocks against it, and a request that echoes thinking from both sides of the boundary is rejected if the marker is missing or moved. Atomic therefore replays the marker exactly where it appeared, drops the declining model's `thinking`, `redacted_thinking`, and unexecuted client-side tool calls that precede it, and keeps all visible text plus everything after it. A turn with no fallback boundary is unaffected.
535
+
536
+ **Provider restriction.** Both behaviors are scoped to first-party Anthropic models on the `anthropic-messages` API, which is where Anthropic documents the signature adjudication. Claude on Amazon Bedrock, Google Vertex, and Anthropic-compatible proxies keep the previous behavior: their thinking blocks are not replayed across a model switch, and Atomic does not send the block-binding beta on those paths. This includes two mirrors that could look eligible — opencode zen and the Vercel AI Gateway both ride `anthropic-messages`, and neither receives either capability. If you run Claude Fable 5.1 through one of those providers on a new Anthropic-backed account, a prefix change can still surface as a provider error. Custom providers known to adjudicate signatures the same way can opt in with the two `compat` fields above.
494
537
 
495
538
  ## OpenAI Compatibility
496
539
 
@@ -120,6 +120,21 @@ Add to `keybindings.json`:
120
120
  }
121
121
  ```
122
122
 
123
+ ## Zed (Integrated Terminal)
124
+
125
+ Add these key bindings to your Zed `keymap.json`:
126
+
127
+ ```json
128
+ {
129
+ "context": "Terminal",
130
+ "bindings": {
131
+ "shift-enter": ["terminal::SendText", "\u001b[13;2u"],
132
+ "ctrl--": ["terminal::SendText", "\u001b[45;5u"],
133
+ "ctrl-alt-]": ["terminal::SendText", "\u001b[93;7u"]
134
+ }
135
+ }
136
+ ```
137
+
123
138
  ## Windows Terminal
124
139
 
125
140
  Add to `settings.json` (CTRL+SHIFT+, or Settings → Open JSON file) to forward the modified Enter keys Atomic uses:
@@ -1,16 +1,16 @@
1
1
  {
2
2
  "name": "@bastani/atomic",
3
- "version": "0.9.18-alpha.3",
3
+ "version": "0.9.18-alpha.5",
4
4
  "lockfileVersion": 3,
5
5
  "requires": true,
6
6
  "packages": {
7
7
  "": {
8
8
  "name": "@bastani/atomic",
9
- "version": "0.9.18-alpha.3",
9
+ "version": "0.9.18-alpha.5",
10
10
  "license": "MIT",
11
11
  "dependencies": {
12
- "@bastani/atomic-natives": "0.9.18-alpha.3",
13
- "@bastani/pi-ai": "0.9.18-alpha.3",
12
+ "@bastani/atomic-natives": "0.9.18-alpha.5",
13
+ "@bastani/pi-ai": "0.9.18-alpha.5",
14
14
  "@dbos-inc/dbos-sdk": "4.25.14",
15
15
  "@earendil-works/pi-agent-core": "^0.84.4",
16
16
  "@earendil-works/pi-client": "^0.84.4",
@@ -517,18 +517,18 @@
517
517
  }
518
518
  },
519
519
  "node_modules/@bastani/atomic-natives": {
520
- "version": "0.9.18-alpha.3",
521
- "resolved": "https://registry.npmjs.org/@bastani/atomic-natives/-/atomic-natives-0.9.18-alpha.3.tgz",
520
+ "version": "0.9.18-alpha.5",
521
+ "resolved": "https://registry.npmjs.org/@bastani/atomic-natives/-/atomic-natives-0.9.18-alpha.5.tgz",
522
522
  "license": "MIT",
523
523
  "optionalDependencies": {
524
- "@bastani/atomic-natives-darwin-arm64": "0.9.18-alpha.3",
525
- "@bastani/atomic-natives-darwin-x64": "0.9.18-alpha.3",
526
- "@bastani/atomic-natives-linux-arm64-gnu": "0.9.18-alpha.3",
527
- "@bastani/atomic-natives-linux-arm64-musl": "0.9.18-alpha.3",
528
- "@bastani/atomic-natives-linux-x64-gnu": "0.9.18-alpha.3",
529
- "@bastani/atomic-natives-linux-x64-musl": "0.9.18-alpha.3",
530
- "@bastani/atomic-natives-win32-arm64-msvc": "0.9.18-alpha.3",
531
- "@bastani/atomic-natives-win32-x64-msvc": "0.9.18-alpha.3"
524
+ "@bastani/atomic-natives-darwin-arm64": "0.9.18-alpha.5",
525
+ "@bastani/atomic-natives-darwin-x64": "0.9.18-alpha.5",
526
+ "@bastani/atomic-natives-linux-arm64-gnu": "0.9.18-alpha.5",
527
+ "@bastani/atomic-natives-linux-arm64-musl": "0.9.18-alpha.5",
528
+ "@bastani/atomic-natives-linux-x64-gnu": "0.9.18-alpha.5",
529
+ "@bastani/atomic-natives-linux-x64-musl": "0.9.18-alpha.5",
530
+ "@bastani/atomic-natives-win32-arm64-msvc": "0.9.18-alpha.5",
531
+ "@bastani/atomic-natives-win32-x64-msvc": "0.9.18-alpha.5"
532
532
  },
533
533
  "engines": {
534
534
  "bun": ">=1.4.0",
@@ -536,8 +536,8 @@
536
536
  }
537
537
  },
538
538
  "node_modules/@bastani/atomic-natives-darwin-arm64": {
539
- "version": "0.9.18-alpha.3",
540
- "resolved": "https://registry.npmjs.org/@bastani/atomic-natives-darwin-arm64/-/atomic-natives-darwin-arm64-0.9.18-alpha.3.tgz",
539
+ "version": "0.9.18-alpha.5",
540
+ "resolved": "https://registry.npmjs.org/@bastani/atomic-natives-darwin-arm64/-/atomic-natives-darwin-arm64-0.9.18-alpha.5.tgz",
541
541
  "license": "MIT",
542
542
  "os": [
543
543
  "darwin"
@@ -548,8 +548,8 @@
548
548
  "optional": true
549
549
  },
550
550
  "node_modules/@bastani/atomic-natives-darwin-x64": {
551
- "version": "0.9.18-alpha.3",
552
- "resolved": "https://registry.npmjs.org/@bastani/atomic-natives-darwin-x64/-/atomic-natives-darwin-x64-0.9.18-alpha.3.tgz",
551
+ "version": "0.9.18-alpha.5",
552
+ "resolved": "https://registry.npmjs.org/@bastani/atomic-natives-darwin-x64/-/atomic-natives-darwin-x64-0.9.18-alpha.5.tgz",
553
553
  "license": "MIT",
554
554
  "os": [
555
555
  "darwin"
@@ -560,8 +560,8 @@
560
560
  "optional": true
561
561
  },
562
562
  "node_modules/@bastani/atomic-natives-linux-arm64-gnu": {
563
- "version": "0.9.18-alpha.3",
564
- "resolved": "https://registry.npmjs.org/@bastani/atomic-natives-linux-arm64-gnu/-/atomic-natives-linux-arm64-gnu-0.9.18-alpha.3.tgz",
563
+ "version": "0.9.18-alpha.5",
564
+ "resolved": "https://registry.npmjs.org/@bastani/atomic-natives-linux-arm64-gnu/-/atomic-natives-linux-arm64-gnu-0.9.18-alpha.5.tgz",
565
565
  "license": "MIT",
566
566
  "os": [
567
567
  "linux"
@@ -575,8 +575,8 @@
575
575
  "optional": true
576
576
  },
577
577
  "node_modules/@bastani/atomic-natives-linux-arm64-musl": {
578
- "version": "0.9.18-alpha.3",
579
- "resolved": "https://registry.npmjs.org/@bastani/atomic-natives-linux-arm64-musl/-/atomic-natives-linux-arm64-musl-0.9.18-alpha.3.tgz",
578
+ "version": "0.9.18-alpha.5",
579
+ "resolved": "https://registry.npmjs.org/@bastani/atomic-natives-linux-arm64-musl/-/atomic-natives-linux-arm64-musl-0.9.18-alpha.5.tgz",
580
580
  "license": "MIT",
581
581
  "os": [
582
582
  "linux"
@@ -590,8 +590,8 @@
590
590
  "optional": true
591
591
  },
592
592
  "node_modules/@bastani/atomic-natives-linux-x64-gnu": {
593
- "version": "0.9.18-alpha.3",
594
- "resolved": "https://registry.npmjs.org/@bastani/atomic-natives-linux-x64-gnu/-/atomic-natives-linux-x64-gnu-0.9.18-alpha.3.tgz",
593
+ "version": "0.9.18-alpha.5",
594
+ "resolved": "https://registry.npmjs.org/@bastani/atomic-natives-linux-x64-gnu/-/atomic-natives-linux-x64-gnu-0.9.18-alpha.5.tgz",
595
595
  "license": "MIT",
596
596
  "os": [
597
597
  "linux"
@@ -605,8 +605,8 @@
605
605
  "optional": true
606
606
  },
607
607
  "node_modules/@bastani/atomic-natives-linux-x64-musl": {
608
- "version": "0.9.18-alpha.3",
609
- "resolved": "https://registry.npmjs.org/@bastani/atomic-natives-linux-x64-musl/-/atomic-natives-linux-x64-musl-0.9.18-alpha.3.tgz",
608
+ "version": "0.9.18-alpha.5",
609
+ "resolved": "https://registry.npmjs.org/@bastani/atomic-natives-linux-x64-musl/-/atomic-natives-linux-x64-musl-0.9.18-alpha.5.tgz",
610
610
  "license": "MIT",
611
611
  "os": [
612
612
  "linux"
@@ -620,8 +620,8 @@
620
620
  "optional": true
621
621
  },
622
622
  "node_modules/@bastani/atomic-natives-win32-arm64-msvc": {
623
- "version": "0.9.18-alpha.3",
624
- "resolved": "https://registry.npmjs.org/@bastani/atomic-natives-win32-arm64-msvc/-/atomic-natives-win32-arm64-msvc-0.9.18-alpha.3.tgz",
623
+ "version": "0.9.18-alpha.5",
624
+ "resolved": "https://registry.npmjs.org/@bastani/atomic-natives-win32-arm64-msvc/-/atomic-natives-win32-arm64-msvc-0.9.18-alpha.5.tgz",
625
625
  "license": "MIT",
626
626
  "os": [
627
627
  "win32"
@@ -632,8 +632,8 @@
632
632
  "optional": true
633
633
  },
634
634
  "node_modules/@bastani/atomic-natives-win32-x64-msvc": {
635
- "version": "0.9.18-alpha.3",
636
- "resolved": "https://registry.npmjs.org/@bastani/atomic-natives-win32-x64-msvc/-/atomic-natives-win32-x64-msvc-0.9.18-alpha.3.tgz",
635
+ "version": "0.9.18-alpha.5",
636
+ "resolved": "https://registry.npmjs.org/@bastani/atomic-natives-win32-x64-msvc/-/atomic-natives-win32-x64-msvc-0.9.18-alpha.5.tgz",
637
637
  "license": "MIT",
638
638
  "os": [
639
639
  "win32"
@@ -644,8 +644,8 @@
644
644
  "optional": true
645
645
  },
646
646
  "node_modules/@bastani/pi-ai": {
647
- "version": "0.9.18-alpha.3",
648
- "resolved": "https://registry.npmjs.org/@bastani/pi-ai/-/pi-ai-0.9.18-alpha.3.tgz",
647
+ "version": "0.9.18-alpha.5",
648
+ "resolved": "https://registry.npmjs.org/@bastani/pi-ai/-/pi-ai-0.9.18-alpha.5.tgz",
649
649
  "license": "MIT",
650
650
  "dependencies": {
651
651
  "@anthropic-ai/sdk": "0.91.1",