0sec-cli 0.15.0 → 0.17.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (209) hide show
  1. package/0sec.js +322 -0
  2. package/README.md +63 -52
  3. package/attacks/data-exfiltration/pii-leakage.yaml +27 -0
  4. package/attacks/encoding-bypass/base64-encoding.yaml +24 -0
  5. package/attacks/jailbreak/dan-roleplay.yaml +27 -0
  6. package/attacks/jailbreak/hypothetical-scenario.yaml +25 -0
  7. package/attacks/jailbreak/multilingual-bypass.yaml +22 -0
  8. package/attacks/output-manipulation/harmful-content.yaml +25 -0
  9. package/attacks/prompt-injection/context-manipulation.yaml +32 -0
  10. package/attacks/prompt-injection/direct-injection.yaml +28 -0
  11. package/attacks/prompt-injection/indirect-injection.yaml +33 -0
  12. package/attacks/system-prompt-extraction/direct-ask.yaml +30 -0
  13. package/attacks/system-prompt-extraction/markdown-exfil.yaml +26 -0
  14. package/attacks/tool-misuse/ssrf-via-tools.yaml +27 -0
  15. package/chunks/adapt-loop-ECQG7454.js +18 -0
  16. package/chunks/adgraph-JLGA6RYI.js +54 -0
  17. package/chunks/agent/skills/frameworks/entra-id.yaml +63 -0
  18. package/chunks/agent/skills/frameworks/graphql-introspection.yaml +116 -0
  19. package/chunks/agent/skills/frameworks/nextjs.yaml +82 -0
  20. package/chunks/agent/skills/frameworks/python-web.yaml +85 -0
  21. package/chunks/agent/skills/frameworks/supabase.yaml +83 -0
  22. package/chunks/agent/skills/frameworks/wordpress-deep.yaml +122 -0
  23. package/chunks/agent/skills/techniques/ad-attack-paths.yaml +57 -0
  24. package/chunks/agent/skills/techniques/advisory-disclosure.yaml +59 -0
  25. package/chunks/agent/skills/techniques/assumption-mining.yaml +53 -0
  26. package/chunks/agent/skills/techniques/blind-exploitation.yaml +143 -0
  27. package/chunks/agent/skills/techniques/crypto-misuse.yaml +88 -0
  28. package/chunks/agent/skills/techniques/cve-poc-adaptation.yaml +82 -0
  29. package/chunks/agent/skills/techniques/entra-attack-paths.yaml +55 -0
  30. package/chunks/agent/skills/techniques/http-conformance-diff.yaml +48 -0
  31. package/chunks/agent/skills/techniques/jwt-attacks.yaml +137 -0
  32. package/chunks/agent/skills/techniques/kernel-weaponization.yaml +53 -0
  33. package/chunks/agent/skills/techniques/llm-excessive-agency.yaml +55 -0
  34. package/chunks/agent/skills/techniques/llm-insecure-output-handling.yaml +56 -0
  35. package/chunks/agent/skills/techniques/llm-prompt-injection.yaml +54 -0
  36. package/chunks/agent/skills/techniques/llm-prompt-layer-write.yaml +68 -0
  37. package/chunks/agent/skills/techniques/llm-rag-poisoning.yaml +57 -0
  38. package/chunks/agent/skills/techniques/llm-safety-eval.yaml +51 -0
  39. package/chunks/agent/skills/techniques/npm-ecosystem.yaml +51 -0
  40. package/chunks/agent/skills/techniques/poc-verification.yaml +55 -0
  41. package/chunks/agent/skills/techniques/race-condition.yaml +122 -0
  42. package/chunks/agent/skills/techniques/scoped-fix.yaml +49 -0
  43. package/chunks/agent/skills/techniques/seedless-depth-review.yaml +59 -0
  44. package/chunks/agent/skills/techniques/spec-differential.yaml +48 -0
  45. package/chunks/agent/skills/techniques/variant-hunting.yaml +55 -0
  46. package/chunks/agent/skills/vulnerabilities/cardano-eutxo-validators.yaml +72 -0
  47. package/chunks/agent/skills/vulnerabilities/command-injection.yaml +99 -0
  48. package/chunks/agent/skills/vulnerabilities/deserialization-chains.yaml +106 -0
  49. package/chunks/agent/skills/vulnerabilities/native-memory-safety.yaml +93 -0
  50. package/chunks/agent/skills/vulnerabilities/path-traversal.yaml +78 -0
  51. package/chunks/agent/skills/vulnerabilities/prototype-pollution.yaml +132 -0
  52. package/chunks/agent/skills/vulnerabilities/request-smuggling.yaml +129 -0
  53. package/chunks/agent/skills/vulnerabilities/sqli-advanced.yaml +102 -0
  54. package/chunks/agent/skills/vulnerabilities/ssrf-bypass.yaml +84 -0
  55. package/chunks/agent/skills/vulnerabilities/ssti-exploitation.yaml +112 -0
  56. package/chunks/agent/skills/vulnerabilities/structural-sqli.yaml +91 -0
  57. package/chunks/appsec-catalog-CBWHGTEL.js +24 -0
  58. package/chunks/artifact-scraper-GJEZMOHF.js +34 -0
  59. package/chunks/assumption-mining-BVMMUZHB.js +79 -0
  60. package/chunks/chunk-2CJ776PV.js +2430 -0
  61. package/chunks/chunk-2RMLOJVB.js +47 -0
  62. package/chunks/chunk-2SANI5RH.js +4728 -0
  63. package/chunks/chunk-3MOLBTLS.js +953 -0
  64. package/chunks/chunk-3QFDYBZQ.js +5108 -0
  65. package/chunks/chunk-47TQJWDH.js +3729 -0
  66. package/chunks/chunk-4Y4KQJXE.js +600 -0
  67. package/chunks/chunk-53G27VPS.js +136 -0
  68. package/chunks/chunk-57ZENEX2.js +169 -0
  69. package/chunks/chunk-5G3ZXFBW.js +182 -0
  70. package/chunks/chunk-6LKRLK2R.js +542 -0
  71. package/chunks/chunk-7DQEV5QI.js +287 -0
  72. package/chunks/chunk-A6CLR72I.js +235 -0
  73. package/chunks/chunk-BKZFDZ23.js +361 -0
  74. package/chunks/chunk-CLHCDHP4.js +471 -0
  75. package/chunks/chunk-D6S3IIPG.js +518 -0
  76. package/chunks/chunk-DN25OQQA.js +1070 -0
  77. package/chunks/chunk-DW5UWPFY.js +275 -0
  78. package/chunks/chunk-F3WBKITT.js +687 -0
  79. package/chunks/chunk-H2FFLZNK.js +3 -0
  80. package/chunks/chunk-H44E2CRN.js +193 -0
  81. package/chunks/chunk-HDWV7PGI.js +596 -0
  82. package/chunks/chunk-HE7LCA7G.js +381 -0
  83. package/chunks/chunk-HT6P7RY3.js +2597 -0
  84. package/chunks/chunk-IOL7D5YV.js +515 -0
  85. package/chunks/chunk-IORC6MQX.js +72828 -0
  86. package/chunks/chunk-IR537GON.js +49 -0
  87. package/chunks/chunk-K26SZ37E.js +5735 -0
  88. package/chunks/chunk-KKGE5RQE.js +1094 -0
  89. package/chunks/chunk-KLTNTE2Z.js +121 -0
  90. package/chunks/chunk-KTQDLNSR.js +1732 -0
  91. package/chunks/chunk-LOHTE223.js +2572 -0
  92. package/chunks/chunk-LP3HHYQU.js +245 -0
  93. package/chunks/chunk-MMLDQR4H.js +852 -0
  94. package/chunks/chunk-MYVT64FN.js +16 -0
  95. package/chunks/chunk-O462Y7P2.js +33019 -0
  96. package/chunks/chunk-OEFNRYI2.js +150 -0
  97. package/chunks/chunk-P6WKNFWX.js +1968 -0
  98. package/chunks/chunk-QI233I24.js +333 -0
  99. package/chunks/chunk-QJQKHO7G.js +917 -0
  100. package/chunks/chunk-QKICO43A.js +808 -0
  101. package/chunks/chunk-QOKTUOU2.js +102 -0
  102. package/chunks/chunk-RDZYQQLW.js +596 -0
  103. package/chunks/chunk-RJWOYEOG.js +271 -0
  104. package/chunks/chunk-RRMJC3ZE.js +105 -0
  105. package/chunks/chunk-RWONANDA.js +973 -0
  106. package/chunks/chunk-SAFFWQW4.js +5921 -0
  107. package/chunks/chunk-SF4KZ4O3.js +502 -0
  108. package/chunks/chunk-SNKTC4BP.js +1422 -0
  109. package/chunks/chunk-SOG2U7B3.js +662 -0
  110. package/chunks/chunk-SZJCPG2I.js +110 -0
  111. package/chunks/chunk-UJ4IK5PO.js +2010 -0
  112. package/chunks/chunk-UM3ZNQIM.js +9329 -0
  113. package/chunks/chunk-VGRDNSHA.js +3249 -0
  114. package/chunks/chunk-VQG4FT5D.js +685 -0
  115. package/chunks/chunk-WKNNZVJS.js +2798 -0
  116. package/chunks/chunk-WU6AFRAZ.js +1167 -0
  117. package/chunks/chunk-WVTBZEQO.js +1531 -0
  118. package/chunks/chunk-YLMN3N25.js +4054 -0
  119. package/chunks/commands-U3AMV5ZU.js +16795 -0
  120. package/chunks/corpus-v1.json +403 -0
  121. package/chunks/cost-ledger-ZCKMBJEN.js +13 -0
  122. package/chunks/data/appsec-archetypes.json +102 -0
  123. package/chunks/data/chromium-archetypes.json +204 -0
  124. package/chunks/data/freebsd-archetypes.json +171 -0
  125. package/chunks/data/kernel-archetypes.json +611 -0
  126. package/chunks/db-26NGQFKO.js +16 -0
  127. package/chunks/disclose-A6AEIEDA.js +132 -0
  128. package/chunks/dist-FDKALV4R.js +2772 -0
  129. package/chunks/dist-GL66KMUX.js +74 -0
  130. package/chunks/dist-JR67XKYO.js +159 -0
  131. package/chunks/eval-runner-DDLE5RQ3.js +27 -0
  132. package/chunks/example-manifest.json +95 -0
  133. package/chunks/exploit-agent-XP23ISN7.js +98 -0
  134. package/chunks/exploit-autoclimb-E6T7RRSC.js +81 -0
  135. package/chunks/exploit-climb-2FAUP6XU.js +358 -0
  136. package/chunks/fix-IG7LYXH3.js +12 -0
  137. package/chunks/github-issues-MJ6OYOOU.js +157 -0
  138. package/chunks/harness-HWOYFKFY.js +22 -0
  139. package/chunks/http-conformance-DU66MZIU.js +11 -0
  140. package/chunks/http-sender-GWH2IYEA.js +10 -0
  141. package/chunks/hunt-scan-XPELSPQQ.js +38 -0
  142. package/chunks/identity-6ZAIIWOR.js +191 -0
  143. package/chunks/kernel-primitive-TENE3R7T.js +44 -0
  144. package/chunks/kernel-vm-runner-4F6QSNFY.js +62 -0
  145. package/chunks/memsafety-scan-J7SECV5W.js +16 -0
  146. package/chunks/native-loop-NTONUDP7.js +54 -0
  147. package/chunks/npm-detectors-KB5Y5ZFX.js +69 -0
  148. package/chunks/npm-dynamic-discovery-AHROOZDB.js +13 -0
  149. package/chunks/orchestrate-ZUWAUWBC.js +62 -0
  150. package/chunks/pipeline-FCARFZU3.js +14 -0
  151. package/chunks/pre-recon-cve-66EB6G4M.js +351 -0
  152. package/chunks/prepare-MYY743TK.js +13 -0
  153. package/chunks/process-3Q7QJIOZ.js +13 -0
  154. package/chunks/replay-runner-RRR4E2A7.js +39 -0
  155. package/chunks/run-ESNN4V5W.js +32632 -0
  156. package/chunks/runtime-62ZQAH7H.js +43 -0
  157. package/chunks/runtime-J7PLXZNM.js +12 -0
  158. package/chunks/scan-stream-FHI2FYZE.js +96 -0
  159. package/chunks/scope-BI7BF4ZY.js +18 -0
  160. package/chunks/session-store-BCFQYDCE.js +28 -0
  161. package/chunks/source-files-PWRQR6LY.js +12 -0
  162. package/chunks/specdrift-WCLH6TTQ.js +19 -0
  163. package/chunks/variant-candidates-6KM6F6MD.js +13 -0
  164. package/chunks/web-recon-prepass-DPVCBRZA.js +1148 -0
  165. package/dashboard/assets/0sec-icon-66SreztZ.gif +0 -0
  166. package/dashboard/assets/bot-yILPnRgD.js +1 -0
  167. package/dashboard/assets/chevron-down-BeDJ-Vka.js +1 -0
  168. package/dashboard/assets/circle-alert-BEnMBMT-.js +1 -0
  169. package/dashboard/assets/client-DRYsuOGl.js +9 -0
  170. package/dashboard/assets/copy-Bi5vzoF9.js +1 -0
  171. package/dashboard/assets/desktop-CXTQonNm.js +32 -0
  172. package/dashboard/assets/desktop-NOekk4QS.css +1 -0
  173. package/dashboard/assets/dist-CmX-h7a4.js +1 -0
  174. package/dashboard/assets/dist-DAwYQmf_.js +45 -0
  175. package/dashboard/assets/findings-page-CACW9nw2.js +5 -0
  176. package/dashboard/assets/format-PoISqsES.js +1 -0
  177. package/dashboard/assets/geist-cyrillic-ext-wght-normal-DjL33-gN.woff2 +0 -0
  178. package/dashboard/assets/geist-cyrillic-wght-normal-BEAKL7Jp.woff2 +0 -0
  179. package/dashboard/assets/geist-latin-ext-wght-normal-DC-KSUi6.woff2 +0 -0
  180. package/dashboard/assets/geist-latin-wght-normal-BgDaEnEv.woff2 +0 -0
  181. package/dashboard/assets/geist-vietnamese-wght-normal-6IgcOCM7.woff2 +0 -0
  182. package/dashboard/assets/ibm-plex-mono-cyrillic-400-normal-BSMlKf0J.woff2 +0 -0
  183. package/dashboard/assets/ibm-plex-mono-cyrillic-400-normal-CEL4l2ZJ.woff +0 -0
  184. package/dashboard/assets/ibm-plex-mono-cyrillic-ext-400-normal-DMdlQ8Kv.woff +0 -0
  185. package/dashboard/assets/ibm-plex-mono-cyrillic-ext-400-normal-xuaO2J-f.woff2 +0 -0
  186. package/dashboard/assets/ibm-plex-mono-latin-400-normal-CvHOgSBP.woff +0 -0
  187. package/dashboard/assets/ibm-plex-mono-latin-400-normal-DMJ8VG8y.woff2 +0 -0
  188. package/dashboard/assets/ibm-plex-mono-latin-ext-400-normal-BmRBH3aV.woff2 +0 -0
  189. package/dashboard/assets/ibm-plex-mono-latin-ext-400-normal-D3D2R8hC.woff +0 -0
  190. package/dashboard/assets/ibm-plex-mono-vietnamese-400-normal-BulugwFq.woff2 +0 -0
  191. package/dashboard/assets/ibm-plex-mono-vietnamese-400-normal-DDuiU_S-.woff +0 -0
  192. package/dashboard/assets/jsx-runtime-C7oxC63R.js +1 -0
  193. package/dashboard/assets/live-page-DrgQqQK5.js +1 -0
  194. package/dashboard/assets/meta-tile-CcskIV_o.js +1 -0
  195. package/dashboard/assets/operations-BKE8T-vY.css +2 -0
  196. package/dashboard/assets/operations-D8nUP5_m.js +4 -0
  197. package/dashboard/assets/operations-app-CQQhG54n.js +2 -0
  198. package/dashboard/assets/overview-page-DTGS8dPo.js +1 -0
  199. package/dashboard/assets/page-header-MFVvEtZW.js +1 -0
  200. package/dashboard/assets/play-Dwo-9xt6.js +1 -0
  201. package/dashboard/assets/scans-page-ChVtq8Us.js +1 -0
  202. package/dashboard/assets/search-CiooTLck.js +1 -0
  203. package/dashboard/assets/siren-CCBwTSuk.js +1 -0
  204. package/dashboard/assets/table-BYS-nIVA.js +1 -0
  205. package/dashboard/assets/tabs-DivxbhQy.js +1 -0
  206. package/dashboard/desktop.html +21 -0
  207. package/dashboard/index.html +15 -0
  208. package/package.json +30 -18
  209. package/bin/0sec.cjs +0 -361
@@ -0,0 +1,4728 @@
1
+ #!/usr/bin/env node
2
+ import { createRequire as __0secCreateRequire } from "node:module";
3
+ const require = __0secCreateRequire(import.meta.url);
4
+ import {
5
+ VERSION,
6
+ homeStateDir
7
+ } from "./chunk-KKGE5RQE.js";
8
+
9
+ // packages/core/dist/agent/feature-presets.js
10
+ var FP_MOAT_FLAGS = [
11
+ "0SEC_FEATURE_REACHABILITY_GATE",
12
+ "0SEC_FEATURE_MULTIMODAL",
13
+ "0SEC_FEATURE_PUBLISHABILITY_GATE",
14
+ "0SEC_FEATURE_POV_GATE",
15
+ "0SEC_FEATURE_POC_GEN_STATIC",
16
+ "0SEC_FEATURE_CONSENSUS_VERIFY"
17
+ ];
18
+ var FEATURE_PRESETS = Object.freeze({
19
+ "fp-moat": FP_MOAT_FLAGS
20
+ });
21
+ var PRESET_ALIASES = Object.freeze({
22
+ "fp-moat": "fp-moat",
23
+ fp_moat: "fp-moat",
24
+ fpmoat: "fp-moat",
25
+ moat: "fp-moat"
26
+ });
27
+ function resolveFeaturePreset(token) {
28
+ return PRESET_ALIASES[token.trim().toLowerCase()];
29
+ }
30
+ function applyFeaturePreset(preset, env2 = process.env) {
31
+ const applied = [];
32
+ const preserved = [];
33
+ for (const flag of FEATURE_PRESETS[preset]) {
34
+ if (env2[flag] !== void 0) {
35
+ preserved.push(flag);
36
+ continue;
37
+ }
38
+ env2[flag] = "1";
39
+ applied.push(flag);
40
+ }
41
+ return { preset, applied, preserved };
42
+ }
43
+ function applyFeaturePresetFromEnv(env2 = process.env) {
44
+ const raw = env2["0SEC_FEATURE_PRESET"];
45
+ if (!raw)
46
+ return void 0;
47
+ const preset = resolveFeaturePreset(raw);
48
+ if (!preset)
49
+ return void 0;
50
+ return applyFeaturePreset(preset, env2);
51
+ }
52
+
53
+ // packages/core/dist/agent/features.js
54
+ var features = {
55
+ /**
56
+ * Early-stop at 50% budget if no findings, retry with a different strategy.
57
+ *
58
+ * OPT-IN (default OFF). It reads "no finding saved by the halfway turn" as
59
+ * "making no progress", which is wrong for the analysis it most often
60
+ * interrupts: a deep source audit legitimately spends its first half reading
61
+ * and narrowing before it has anything worth saving. Stopping there discards
62
+ * that work and reports it as a failure to the operator.
63
+ *
64
+ * Turn count is also a cost PROXY, and cost is already bounded directly by
65
+ * the token budget, so this never was the control that kept spend in check.
66
+ * Enable with `0SEC_FEATURE_EARLY_STOP=1` for benchmark or A/B runs where a
67
+ * fixed turn budget per attempt is the point.
68
+ */
69
+ get earlyStopRetry() {
70
+ return env("0SEC_FEATURE_EARLY_STOP", false);
71
+ },
72
+ /** Detect A-A-A and A-B-A-B loop patterns, inject warning */
73
+ get loopDetection() {
74
+ return env("0SEC_FEATURE_LOOP_DETECTION", true);
75
+ },
76
+ /** Compress middle messages when context exceeds 30k tokens */
77
+ get contextCompaction() {
78
+ return env("0SEC_FEATURE_CONTEXT_COMPACTION", true);
79
+ },
80
+ /**
81
+ * Re-send the opaque, model-bound Responses output item array on the next
82
+ * turn. Default ON; set to 0 only for matched retained-reasoning A/B runs.
83
+ */
84
+ get retainedReasoning() {
85
+ return env("0SEC_FEATURE_RETAINED_REASONING", true);
86
+ },
87
+ /** Exploit script templates in shell prompt (blind SQLi, SSTI, auth chain) */
88
+ get scriptTemplates() {
89
+ return env("0SEC_FEATURE_SCRIPT_TEMPLATES", true);
90
+ },
91
+ /** Dynamic vulnerability playbooks injected after recon phase */
92
+ get dynamicPlaybooks() {
93
+ return env("0SEC_FEATURE_DYNAMIC_PLAYBOOKS", false);
94
+ },
95
+ /** Just-in-time atomic DO/DON'T rules injected on a matching tool action */
96
+ get ruleInjection() {
97
+ return env("0SEC_FEATURE_RULE_INJECTION", false);
98
+ },
99
+ /** Agent writes plan/creds to disk, injected at reflection checkpoints */
100
+ get externalMemory() {
101
+ return env("0SEC_FEATURE_EXTERNAL_MEMORY", false);
102
+ },
103
+ /** Inject prior attempt findings when retrying (LLM-summarized progress handoff) */
104
+ get progressHandoff() {
105
+ return env("0SEC_FEATURE_PROGRESS_HANDOFF", true);
106
+ },
107
+ /** Allow the agent to search the web for CVE details, docs, and technique references */
108
+ get webSearch() {
109
+ return env("0SEC_FEATURE_WEB_SEARCH", false);
110
+ },
111
+ /** Interactive PTY sessions for exploits requiring interactivity (reverse shells, DB clients, SSH) */
112
+ get ptySession() {
113
+ return env("0SEC_FEATURE_PTY_SESSION", false);
114
+ },
115
+ /**
116
+ * Persistent, COMPUTE-ONLY Python REPL (`python_exec`, Phase-0). A framed
117
+ * python3 kernel keeps state across calls for payload/parse/crypto/encode
118
+ * work; networking is blocked at the socket source whenever an engagement is
119
+ * active. Default OFF — opt in via 0SEC_FEATURE_PYTHON_EXEC=1. Getter so
120
+ * the CLI `--features` flag (set after this module is imported) is honored at
121
+ * tool-dispatch time.
122
+ */
123
+ get pythonExec() {
124
+ return env("0SEC_FEATURE_PYTHON_EXEC", false);
125
+ },
126
+ /**
127
+ * Expose the path-confined `analyze_binary` bridge to 0verse. Default OFF:
128
+ * a model may request a long-running binary analysis only after an operator
129
+ * opts in with 0SEC_FEATURE_ZEROVERSE=1.
130
+ */
131
+ get zeroverse() {
132
+ return env("0SEC_FEATURE_ZEROVERSE", false);
133
+ },
134
+ /**
135
+ * EGATS specialist routing (#557, HPTSA-inspired). When ON, an EGATS branch
136
+ * whose hypothesis names a concrete vuln class (SQLi/XSS/SSRF/SSTI/IDOR/
137
+ * auth-bypass) runs as a per-class SPECIALIST: a class system prompt built
138
+ * from the technique sections in prompts.ts, the matching methodology skill
139
+ * auto-loaded into context, and a class-tuned tool subset. Hypotheses that
140
+ * are ambiguous (zero or multiple classes) fall back to the generic branch
141
+ * agent — beam search / scoring are untouched. Emits an `egats_specialist`
142
+ * event per routed node.
143
+ *
144
+ * Default OFF: this changes how branch mini-loops are configured, so it must
145
+ * be explicitly opted into before any A/B / multiplier claim on the
146
+ * benchmark harness. Implemented as a getter so the CLI `--features` flag
147
+ * (which sets the env var inside the command action, AFTER this module has
148
+ * been imported) is honored at routing time. Enable via
149
+ * 0SEC_FEATURE_SPECIALIST_ROUTING=1.
150
+ */
151
+ get specialistRouting() {
152
+ return env("0SEC_FEATURE_SPECIALIST_ROUTING", false);
153
+ },
154
+ /** Self-consistency voting: run the structured verify pipeline N times and take the majority vote */
155
+ get selfConsistencyVerify() {
156
+ return env("0SEC_FEATURE_CONSENSUS_VERIFY", false);
157
+ },
158
+ /** Multi-modal agreement: cross-validate findings against foxguard (Rust pattern scanner) */
159
+ get multiModalAgreement() {
160
+ return env("0SEC_FEATURE_MULTIMODAL", false);
161
+ },
162
+ /** Reachability gate: suppress findings whose sink is not reachable from an application entry point */
163
+ get reachabilityGate() {
164
+ return env("0SEC_FEATURE_REACHABILITY_GATE", false);
165
+ },
166
+ /**
167
+ * Publishability / in-scope gate (issue #537 / #539). Decides
168
+ * disclosure-worthiness per finding: SECURITY.md threat-model exclusion
169
+ * (by_design), global advisory dedup (duplicate) with the fix-bypass
170
+ * exception, latest-version (fixed), and public-API reachability
171
+ * (unreachable). Never auto-drops high-severity/high-impact findings — those
172
+ * are routed to needs_verify + human review via canAutoSuppress.
173
+ *
174
+ * Default OFF: this gate can suppress reproducible findings, so it must be
175
+ * explicitly opted into before any A/B claim. Disable/enable via
176
+ * 0SEC_FEATURE_PUBLISHABILITY_GATE.
177
+ */
178
+ get publishabilityGate() {
179
+ return env("0SEC_FEATURE_PUBLISHABILITY_GATE", false);
180
+ },
181
+ /** PoV gate: require a working, executable PoC per finding or downgrade to info */
182
+ get povGate() {
183
+ return env("0SEC_FEATURE_POV_GATE", false);
184
+ },
185
+ /**
186
+ * Intra-scan semantic dedupe post-pass (anchored incremental LLM
187
+ * clustering over the final finding set, `triage/semantic-dedupe.ts`).
188
+ * Marks duplicates with a canonical mapping + cluster reason instead of
189
+ * dropping them. Default OFF: it spends an LLM call per ≤50-finding batch
190
+ * after the scan, so it must be explicitly opted into before any A/B
191
+ * claim. Toggle via 0SEC_FEATURE_SEMANTIC_DEDUPE.
192
+ */
193
+ get semanticDedupe() {
194
+ return env("0SEC_FEATURE_SEMANTIC_DEDUPE", false);
195
+ },
196
+ /**
197
+ * Finding-specific remediation written by the model
198
+ * (`generateRemediationWithLLM`) instead of the static category knowledge
199
+ * base. Default OFF: it spends one extra LLM call per non-false-positive
200
+ * finding at report-assembly time, which is real money on a noisy scan and
201
+ * buys nothing on a scan with no findings.
202
+ *
203
+ * Worth turning on for disclosure-bound work: the static KB emits the same
204
+ * generic snippet for every finding in a category, whereas the model sees
205
+ * this finding's evidence and can name the actual sink. The call is
206
+ * fail-open — any error falls back to the KB answer — so enabling it can
207
+ * degrade cost, never correctness.
208
+ */
209
+ get llmRemediation() {
210
+ return env("0SEC_FEATURE_LLM_REMEDIATION", false);
211
+ },
212
+ /**
213
+ * Per-finding impact assessment (`assessImpact`, `triage/impact-assessment.ts`)
214
+ * written by the model: reachability tier, weaponizability, blast radius,
215
+ * business-impact tier. Default OFF — one extra LLM call per non-false-positive
216
+ * finding at report time.
217
+ *
218
+ * When on, the assessment feeds three things it is otherwise absent from:
219
+ * a real CVSS exploitability vector (AV/PR/UI from the reachability tier
220
+ * rather than the AV:N/severity-floor guess), the advisory's Impact +
221
+ * attack-prerequisites section, and the vendor-notification impact line. When
222
+ * off, all three fall back to today's category/severity heuristics — so this
223
+ * flag strictly adds fidelity, never changes the no-assessment output.
224
+ */
225
+ get impactAssessment() {
226
+ return env("0SEC_FEATURE_IMPACT_ASSESSMENT", false);
227
+ },
228
+ /**
229
+ * Incremental finding ranking post-pass (decimal-insertion between ranked
230
+ * anchors, `triage/incremental-rank.ts`). Orders the report by comparative
231
+ * promise (exploitability × impact × evidence strength). Default OFF: it
232
+ * spends an LLM call per ≤50-finding batch; opt in before any A/B claim.
233
+ * Toggle via 0SEC_FEATURE_INCREMENTAL_RANK.
234
+ */
235
+ get incrementalRank() {
236
+ return env("0SEC_FEATURE_INCREMENTAL_RANK", false);
237
+ },
238
+ /**
239
+ * Static-finding PoC generation (#666 / EPIC #674 Part A). For findings that
240
+ * ship with NO executable PoC (`pocSteps` empty — the static / code-analysis
241
+ * path), run an agentic PoC-gen pass that builds + runs a minimal PoC in the
242
+ * scan substrate (reuses the PoV mini-loop). On reproduce it synthesizes a
243
+ * runnable `pocSteps` graph so the verify runner stops skipping the finding;
244
+ * on no-repro it flags the finding `poc:none` for manual / inconclusive
245
+ * review instead of silently dropping it. Root cause: 112 high/crit findings
246
+ * with `poc_steps IS NULL` were silently `skipped` by the verify fan-out.
247
+ *
248
+ * Default OFF: it spends LLM + execution budget per static finding and must
249
+ * be explicitly opted into before any A/B claim (A/B-able via the #656
250
+ * harness). Toggle via 0SEC_FEATURE_POC_GEN_STATIC.
251
+ */
252
+ get pocGenStatic() {
253
+ return env("0SEC_FEATURE_POC_GEN_STATIC", false);
254
+ },
255
+ /**
256
+ * Inline validation / validate-on-save (#554). When ON, the native attack
257
+ * loop runs a fast deterministic category oracle the moment a high/critical
258
+ * finding is saved (`onFindingSaved` hook → `verifyOracleByCategory`, the
259
+ * cheap end of the #553 PoV-gate→oracle delegation). The verdict is injected
260
+ * back into the loop as a context note (confirmed → stop piling on;
261
+ * unconfirmed → "do not assume success"), stamped on `finding.inlineValidation`
262
+ * so EGATS `scoreEvidence` lets a confirmed finding dominate the regex signals
263
+ * and the batch oracle/PoV gate can skip the redundant re-run. Inline errors
264
+ * are inconclusive, never false-positive. Emits `inline_validation` events.
265
+ *
266
+ * Default OFF: it adds a per-finding network probe inside the attack loop and
267
+ * changes EGATS scoring, so it must be explicitly opted into before any A/B /
268
+ * cost_per_flag claim. Implemented as a getter so the CLI `--features` flag
269
+ * (which sets the env var inside the command action, AFTER this module is
270
+ * imported) is honored at loop time. Enable via
271
+ * 0SEC_FEATURE_INLINE_VALIDATION=1.
272
+ */
273
+ get inlineValidation() {
274
+ return env("0SEC_FEATURE_INLINE_VALIDATION", false);
275
+ },
276
+ /**
277
+ * WordPress plugin/theme fingerprinter + OSV CVE lookup.
278
+ * Exposes the `wp_fingerprint` tool to the attack agent. Off by default —
279
+ * can be disabled via `--features no-wp_fingerprint` / env if needed.
280
+ * WordPress detection is cheap and the resulting plugin/CVE hints are
281
+ * broadly useful on real web targets, so the default is ON. See
282
+ * packages/core/src/agent/wp-fingerprint.ts for the implementation.
283
+ *
284
+ * Implemented as a getter so the CLI `--features` flag — which sets the env
285
+ * var inside the command action, AFTER this module has been imported — is
286
+ * still honored at tool-dispatch time.
287
+ */
288
+ get wpFingerprint() {
289
+ return env("0SEC_FEATURE_WP_FINGERPRINT", true);
290
+ },
291
+ /**
292
+ * MongoDB ObjectID forge tool. Exposes the `mongo_objectid` tool to the
293
+ * attack agent so it can compute valid 24-char hex ObjectIds with arbitrary
294
+ * timestamps + counters (e.g. forge the "first user" ObjectId in an IDOR
295
+ * challenge by setting timestamp = appStartTimestamp and counter = 0).
296
+ *
297
+ * Default ON — this is a pure-computation utility with no network or
298
+ * filesystem side effects, so there's no reason to gate it off. Disable
299
+ * via 0SEC_FEATURE_MONGO_OBJECTID_FORGE=0 or `--no-mongo-objectid-forge`
300
+ * for ablation. Implemented as a getter so the CLI `--features` flag
301
+ * (which sets the env var inside the command action, AFTER this module
302
+ * has been imported) is still honored at tool-dispatch time. Matches
303
+ * the wpFingerprint pattern above. See packages/core/src/agent/objectid-forge.ts.
304
+ */
305
+ get mongoObjectIdForge() {
306
+ return env("0SEC_FEATURE_MONGO_OBJECTID_FORGE", true);
307
+ },
308
+ /**
309
+ * Live cloud-surface testing (0sec#925). Exposes `cloud_s3_probe` and
310
+ * `cloud_validate_credentials` to the attack agent so it can test S3 buckets
311
+ * for public access + orphaned-bucket takeover and safely validate harvested
312
+ * AWS credentials (read-only). All probes are anonymous or read/verify-only —
313
+ * no writes, no data exfiltration beyond minimal proof.
314
+ *
315
+ * Default OFF (opt-in via 0SEC_FEATURE_CLOUD_SURFACE=1). Probing a target
316
+ * org's bucket-name space or validating its harvested credentials is recon
317
+ * AGAINST THAT ORG, so it is deny-by-default at two layers: this enablement
318
+ * flag, AND an engagement-scope check in the tool handlers (a configured
319
+ * ScopePolicy that authorizes the bucket endpoint — see cloud-surface.ts
320
+ * `bucketInScope`). The read-only action allowlist (`assertReadOnlyAction`)
321
+ * stays on top of both. Getter so the CLI `--features` flag (set AFTER this
322
+ * module loads) is honored at dispatch time — matches the wpFingerprint /
323
+ * mongoObjectIdForge pattern above. See packages/core/src/agent/cloud-surface.ts.
324
+ */
325
+ get cloudSurface() {
326
+ return env("0SEC_FEATURE_CLOUD_SURFACE", false);
327
+ },
328
+ /**
329
+ * #978 (ADR-060) — agent fan-out. When ON, the agent gets the `start_scan`
330
+ * tool: it can dispatch CHILD scans (via the same POST /scans the UI uses)
331
+ * that run independently and report up the scan tree — the recursive
332
+ * sub-agent orchestration. Default OFF: fan-out multiplies scans/cost, so it
333
+ * stays opt-in even though the orchestrator enforces budget + a tree-level
334
+ * cap (max children/depth). Enable with 0SEC_FEATURE_AGENT_FANOUT=1.
335
+ * Getter so the CLI `--features` flag is honored at dispatch time.
336
+ */
337
+ get agentFanout() {
338
+ return env("0SEC_FEATURE_AGENT_FANOUT", false);
339
+ },
340
+ // ── Phase-2 offensive-engine feature flags (dev-live-engine-recovery) ──
341
+ // Each gates a tool that RUNS/BUILDS untrusted code or WEAPONIZES. They are
342
+ // deny-by-default and mirror `cloudSurface` exactly: the env flag is only one
343
+ // of two layers — the tool is ALSO engagement-scope gated in getToolsForRole,
344
+ // and the highest-risk ones add their own runtime precondition (kernel-VM
345
+ // artifacts present). Getters so the CLI `--features` flag (set AFTER this
346
+ // module loads) is honored at dispatch time.
347
+ /**
348
+ * `memsafety_fuzz` — clones/builds/fuzzes a source tree (sanitizer builds +
349
+ * a fuzz harness), executing attacker-adjacent build scripts and native
350
+ * fuzz targets. Default OFF; opt in via 0SEC_FEATURE_MEMSAFETY=1. Building
351
+ * and running an untrusted tree is code execution, so it is deny-by-default
352
+ * behind this flag AND an engagement scope.
353
+ */
354
+ get memsafetyFuzz() {
355
+ return env("0SEC_FEATURE_MEMSAFETY", false);
356
+ },
357
+ /**
358
+ * `npm_dynamic_discovery` — installs and RUNS untrusted npm packages under
359
+ * instrumentation to observe malicious install/runtime behaviour. Executing
360
+ * arbitrary package code is the whole point, so it is deny-by-default behind
361
+ * this flag AND an engagement scope. Opt in via 0SEC_FEATURE_NPM_DISCOVERY=1.
362
+ */
363
+ get npmDynamicDiscovery() {
364
+ return env("0SEC_FEATURE_NPM_DISCOVERY", false);
365
+ },
366
+ /**
367
+ * `weaponize_kernel` — the kernel-exploit weaponization ladder. It only ever
368
+ * runs inside a DISPOSABLE kernel VM and requires kernel-VM artifacts to be
369
+ * present. Highest-caution capability: deny-by-default behind this flag AND
370
+ * an engagement scope AND a runtime artifact-presence check (the handler
371
+ * refuses when the kernel-VM assets are absent). Opt in via
372
+ * 0SEC_FEATURE_KERNEL_WEAPONIZE=1.
373
+ */
374
+ get kernelWeaponize() {
375
+ return env("0SEC_FEATURE_KERNEL_WEAPONIZE", false);
376
+ },
377
+ /**
378
+ * `cve_adapt` — adapts a public CVE PoC to the target and RUNS it to confirm
379
+ * exploitability. Running an adapted exploit is code execution against the
380
+ * target, so it is deny-by-default behind this flag AND an engagement scope.
381
+ * Opt in via 0SEC_FEATURE_CVE_ADAPT=1.
382
+ */
383
+ get cveAdapt() {
384
+ return env("0SEC_FEATURE_CVE_ADAPT", false);
385
+ },
386
+ /**
387
+ * Anti-honeypot flag-shape validator. When the agent calls the `done`
388
+ * tool with a proposed `FLAG{...}`, the tool runs `validateFlagShape`
389
+ * first; low-confidence ("looks like a decoy") flags are rejected once
390
+ * with a hint to keep exploring. The agent can override by retrying the
391
+ * same flag — the heuristic is a speed bump, not a hard wall.
392
+ *
393
+ * Default ON because legitimate flags pass the shape check trivially
394
+ * and the false-positive rate on real flags should be near zero. Turn
395
+ * off via `0SEC_FEATURE_DECOY_DETECTION=0` or the CLI flag
396
+ * `--no-decoy-detection` for ablation/testing.
397
+ *
398
+ * Implemented as a getter so the CLI flag (which flips the env var
399
+ * inside the command action, AFTER this module has been imported) is
400
+ * still honored at tool-dispatch time. Matches the wpFingerprint
401
+ * pattern above. See GitHub issue #82 and
402
+ * packages/core/src/agent/flag-validator.ts.
403
+ */
404
+ get decoyDetection() {
405
+ return env("0SEC_FEATURE_DECOY_DETECTION", true);
406
+ },
407
+ // ── Always-on triage filters (default ON, ablatable for A/B testing) ──
408
+ /**
409
+ * `holding-it-wrong` regex blocklist (`packages/core/src/triage/holding-it-wrong.ts`).
410
+ * Matches finding text against documented I/O / eval / compile / persistence
411
+ * sink names and rejects findings that look like "the function did its job".
412
+ *
413
+ * Default ON because that's the existing v0.6.0 behavior. Can be disabled
414
+ * via 0SEC_FEATURE_HOLDING_IT_WRONG=0 to test whether this filter is
415
+ * suppressing real signal — the ceiling-analysis from 2026-04-06 identified
416
+ * this as the strongest candidate for the unexplained XBOW finding-density
417
+ * collapse from 14 → 4 between `features=none` and `features=all`.
418
+ */
419
+ get holdingItWrong() {
420
+ return env("0SEC_FEATURE_HOLDING_IT_WRONG", true);
421
+ },
422
+ /**
423
+ * `evidence_completeness <= 0.5` reject (`packages/core/src/agentic-scanner.ts:591`).
424
+ * Drops findings whose extracted feature vector says the agent didn't
425
+ * gather enough cross-source evidence (request + response + analysis + ...).
426
+ *
427
+ * Default ON because that's the existing v0.6.0 behavior. Can be disabled
428
+ * via 0SEC_FEATURE_EVIDENCE_GATE=0 for ablation.
429
+ */
430
+ get evidenceGate() {
431
+ return env("0SEC_FEATURE_EVIDENCE_GATE", true);
432
+ },
433
+ /**
434
+ * Learned per-finding triage router (`packages/core/src/triage/learned-router.ts`).
435
+ * When enabled, findings are scored by hand-coded rules derived from the
436
+ * XGBoost model trained on triage-dataset-v2.jsonl (1514 rows). High-confidence
437
+ * findings auto-accept (skipping expensive layers); low-confidence findings
438
+ * auto-reject; the middle band gets routed to a subset of layers based on
439
+ * the scan's slice type (xbow-wb, xbow-bb, npm).
440
+ *
441
+ * Default OFF until the router is validated via A/B testing on xbow-bench
442
+ * and npm-bench. See 0sec#113 for the design doc.
443
+ */
444
+ get learnedRouter() {
445
+ return env("0SEC_FEATURE_LEARNED_ROUTER", false);
446
+ },
447
+ /**
448
+ * Dynamic per-finding triage routing (`packages/core/src/triage/router/`).
449
+ * When enabled, every finding is sent through a `RouterModel` that
450
+ * decides which subset of the 11 triage layers to invoke for that
451
+ * specific finding. v0 ships an explicit-rule router encoded from the
452
+ * 0sec#72 per-profile ablation; a learned classifier replaces the
453
+ * rules in a follow-up PR without touching the dispatch site.
454
+ *
455
+ * Distinct from `learnedRouter` above: `learnedRouter` is the XGBoost
456
+ * TP/FP score model that decides accept/reject; `dynamicTriageRouting`
457
+ * is the per-layer dispatch decision. Both can be on at the same time;
458
+ * the dispatch router gates which layers run AFTER the TP/FP score
459
+ * model has spoken.
460
+ *
461
+ * Default OFF — opt in via 0SEC_FEATURE_DYNAMIC_TRIAGE=1. See
462
+ * 0sec#113 for the design doc and 0sec#67 for the joint paper plan.
463
+ */
464
+ get dynamicTriageRouting() {
465
+ return env("0SEC_FEATURE_DYNAMIC_TRIAGE", false);
466
+ },
467
+ /**
468
+ * Opt-in cloud-sink webhook integration (`packages/core/src/cloud-sink.ts`).
469
+ * When enabled AND the user has set 0SEC_CLOUD_SINK + 0SEC_CLOUD_SCAN_ID,
470
+ * every finding and the final scan report are POSTed to the configured
471
+ * remote endpoint in real time.
472
+ *
473
+ * Default ON so the env-var trio is sufficient to enable streaming, but the
474
+ * flag exists so operators can force-disable the integration in environments
475
+ * where outbound HTTP from the scanner is not desired (e.g. air-gapped CI).
476
+ * Disable via 0SEC_FEATURE_CLOUD_SINK=0.
477
+ */
478
+ get cloudSink() {
479
+ return env("0SEC_FEATURE_CLOUD_SINK", true);
480
+ },
481
+ /**
482
+ * Pre-recon CVE check (`packages/core/src/pre-recon-cve.ts`).
483
+ * In white-box mode (`--repo` set), runs `npm audit` / `pip-audit`
484
+ * against the source tree before the attack agent starts and injects
485
+ * any high/critical advisories into the system prompt as priority
486
+ * leads. Defends against expensive thrash on CVE-tagged challenges
487
+ * where the agent has source access but no concrete leads.
488
+ *
489
+ * Default ON in white-box mode (no-op in black-box). Disable via
490
+ * 0SEC_FEATURE_PRE_RECON_CVE=0 for ablation.
491
+ */
492
+ get preReconCve() {
493
+ return env("0SEC_FEATURE_PRE_RECON_CVE", true);
494
+ },
495
+ /**
496
+ * Deterministic web-recon pre-pass (`packages/core/src/stages/web-recon-prepass.ts`).
497
+ * On web scans, runs cheap non-destructive HTTP/DNS probes before the attack
498
+ * agent's first turn: baseline web checks, stack fingerprint → version→CVE
499
+ * lookup, JS source-map/secret scan, DNS/email posture, passive subdomain
500
+ * enumeration, and (Next.js only, on positive proof) framework-CVE active
501
+ * checks. It EMITS findings directly for what it can prove and injects a
502
+ * "pursue these leads" block into the system prompt for what it can only hint.
503
+ *
504
+ * Default ON (no-op in non-web modes). Gated behind 0SEC_FEATURE_WEB_RECON
505
+ * so it can be disabled for ablation or offline runs. Implemented as a getter
506
+ * so the CLI `--features` flag (which sets the env var inside the command
507
+ * action, AFTER this module has been imported) is honored at stage time.
508
+ */
509
+ get webRecon() {
510
+ return env("0SEC_FEATURE_WEB_RECON", true);
511
+ },
512
+ /**
513
+ * Best-effort target-history preflight for source review. When a local repo
514
+ * path is known, 0sec infers repository/package/product hints, queries live
515
+ * prior-vulnerability intel, and injects a compact audit-graph summary into
516
+ * the review prompt before the agent starts.
517
+ *
518
+ * Default ON for white-box/source-review modes. Disable via
519
+ * 0SEC_FEATURE_TARGET_HISTORY_PRESEED=0 for offline or ablation runs.
520
+ */
521
+ get targetHistoryPreseed() {
522
+ return env("0SEC_FEATURE_TARGET_HISTORY_PRESEED", true);
523
+ },
524
+ /**
525
+ * Preserve credential / exploit-bearing messages verbatim during
526
+ * `compactMessagesWithLLM` (`packages/core/src/agent/native-loop.ts`).
527
+ * When the conversation is compacted, middle messages whose serialized
528
+ * text matches the critical-message regex (passwords, credentials,
529
+ * shells, exploits, login/auth tokens, etc.) are appended verbatim
530
+ * after the LLM summary block, instead of being replaced by a paraphrase.
531
+ *
532
+ * Default ON: the win on long-tail challenges where a credential is
533
+ * recovered in turn 12 and needed in turn 38 is large, and the cost
534
+ * (a handful of extra messages preserved verbatim in the user
535
+ * compaction-summary block) is small. BoxPwnr-inspired: see
536
+ * `src/boxpwnr/solvers/single_loop_compactation.py` in 0ca/BoxPwnr,
537
+ * and 0sec#229 for the design discussion.
538
+ *
539
+ * Implemented as a getter so the CLI `--features` flag — which sets
540
+ * the env var inside the command action AFTER this module is imported
541
+ * — is still honored at compaction time. Disable via
542
+ * 0SEC_FEATURE_PRESERVE_CRITICAL_MESSAGES=0 for ablation.
543
+ */
544
+ get preserveCriticalMessages() {
545
+ return env("0SEC_FEATURE_PRESERVE_CRITICAL_MESSAGES", true);
546
+ },
547
+ /**
548
+ * Two-stage budget-warning injection in the agent loop (#408).
549
+ *
550
+ * Strix's `base_agent.py:186-211` injects a soft warning at 85% of the
551
+ * turn budget and a sharper warning at `maxTurns − 3` so the model gets
552
+ * a clean signal to call `done` (or `save_finding`+`done`) instead of
553
+ * being cut off mid-thought when the hard turn limit triggers. Each
554
+ * warning fires AT MOST ONCE per run; the small turn-state field
555
+ * `budgetWarningsFired` lives on the loop's local closure.
556
+ *
557
+ * Default ON per the issue acceptance criteria — the warnings are a
558
+ * single short user-message injection at two specific turn boundaries,
559
+ * and the win on long benchmarks (clean handoff instead of stray
560
+ * exploration on the last turn) is well-documented in Strix's
561
+ * implementation. Disable via 0SEC_FEATURE_BUDGET_WARNINGS=0 for
562
+ * ablation. Implemented as a getter so the CLI `--features` flag —
563
+ * which sets the env var inside the command action AFTER this module
564
+ * is imported — is still honored at injection time (matches the
565
+ * wpFingerprint / preserveCriticalMessages pattern).
566
+ */
567
+ get budgetWarnings() {
568
+ return env("0SEC_FEATURE_BUDGET_WARNINGS", true);
569
+ },
570
+ /**
571
+ * Per-file orchestration for the research and audit stages (#285).
572
+ *
573
+ * When enabled (default), the research and audit stages call the agent
574
+ * once per source file with a focused per-file system prompt rather than
575
+ * one shared session that nominally walks all files but in practice
576
+ * skips, dedupes, or condenses past the first ~30. Mirrors the
577
+ * per-finding verify loop pattern from pov-gate.ts.
578
+ *
579
+ * Trade-off: total token spend grows roughly N × per-file budget instead
580
+ * of capped at a single session's budget. For a 50-file package, that
581
+ * could be a 5-10× cost increase on research. Disable via
582
+ * `0SEC_FEATURE_PER_ITEM_ORCHESTRATION=0` to revert to the shared-session
583
+ * behavior — useful for cost-bounded benchmarks.
584
+ *
585
+ * Implemented as a getter so the env var is honored at orchestration time
586
+ * (matches the wpFingerprint / mongoObjectIdForge pattern).
587
+ */
588
+ get perItemOrchestration() {
589
+ return env("0SEC_FEATURE_PER_ITEM_ORCHESTRATION", true);
590
+ },
591
+ /**
592
+ * JIT skill loading (`packages/core/src/agent/skills/`).
593
+ * When enabled, the agent gains `list_skills` and `load_skill` tools
594
+ * that let it browse a registry of focused methodology guides and load
595
+ * them into working context mid-scan. Skills replace the monolithic
596
+ * playbook injection with targeted, on-demand knowledge (#410, #457).
597
+ *
598
+ * Default OFF until the skill registry is validated via A/B testing.
599
+ * Implemented as a getter so the CLI `--features` flag — which sets
600
+ * the env var inside the command action, AFTER this module has been
601
+ * imported — is still honored at tool-dispatch time.
602
+ */
603
+ get jitSkills() {
604
+ return env("0SEC_FEATURE_JIT_SKILLS", false);
605
+ },
606
+ /**
607
+ * Execution-journal shadow mode (#494, first additive slice).
608
+ *
609
+ * When ON, the live agent loop ALSO writes append-only journal entries
610
+ * (`tool_call`, `tool_result`, `finding`, `done`) to
611
+ * `~/.0sec/runs/<scanId>/journal.jsonl` as it runs — a durable,
612
+ * replayable trace alongside the existing in-memory conversation window.
613
+ * This is strictly additive: the loop continues to drive off its own
614
+ * conversation state, the journal is write-only here, and a failed
615
+ * journal write is swallowed so it can never abort a scan. The journal is
616
+ * NOT yet the source of truth — routing the loop off `rehydrateContext`
617
+ * is the next slice (see docs/research/agent-execution-journal-design.md).
618
+ *
619
+ * Default OFF: shadow writes add a small per-turn fsync cost and the
620
+ * format is still settling, so it must be explicitly opted into for the
621
+ * moat-ablation harness before any A/B claim. Implemented as a getter so
622
+ * the CLI `--features` flag (which sets the env var inside the command
623
+ * action, AFTER this module has been imported) is honored at loop time.
624
+ * Enable via 0SEC_FEATURE_EXECUTION_JOURNAL=1 or `--features
625
+ * execution-journal`.
626
+ */
627
+ get executionJournal() {
628
+ return env("0SEC_FEATURE_EXECUTION_JOURNAL", false);
629
+ },
630
+ /**
631
+ * Execution-journal context routing (#494, slice 2).
632
+ *
633
+ * When ON, the native agent loop seeds its initial/resume conversation
634
+ * context from the on-disk execution journal via
635
+ * `rehydrateContext(loadJournal(...))` instead of (or fronting) the
636
+ * truncated 40-message DB session blob. This is the slice that finally
637
+ * routes the loop's context OFF the journal — the IronCurtain "every
638
+ * agent begins with a fresh context window and rehydrates from disk"
639
+ * primitive becomes load-bearing.
640
+ *
641
+ * Independent of `executionJournal` (the shadow-WRITE flag) on purpose so
642
+ * the moat-ablation harness can toggle write and route separately for a
643
+ * clean A/B. Rehydrate is a READER, though, so it only does anything when
644
+ * a journal was written for the run — it reads `~/.0sec/runs/<scanId>/
645
+ * journal.jsonl` regardless of how it got there (shadow mode this slice,
646
+ * or specialists in a later slice). When the journal is missing, empty, or
647
+ * corrupt the loop falls back to the existing DB-blob / fresh-prompt
648
+ * seeding and never crashes; the fallback is logged. A FRESH run (no
649
+ * journal yet) rehydrates to empty state, which is byte-equivalent to
650
+ * today's initial-prompt seeding — so `journalRehydrate` only changes
651
+ * behaviour on RESUME of an already-journaled run.
652
+ *
653
+ * Default OFF: this changes the loop's source of truth for resume, so it
654
+ * must be explicitly opted into before any A/B claim. Implemented as a
655
+ * getter so the CLI `--features` flag (which sets the env var inside the
656
+ * command action, AFTER this module has been imported) is honored at loop
657
+ * time. Enable via 0SEC_FEATURE_JOURNAL_REHYDRATE=1 or `--features
658
+ * journal-rehydrate`.
659
+ */
660
+ get journalRehydrate() {
661
+ return env("0SEC_FEATURE_JOURNAL_REHYDRATE", false);
662
+ },
663
+ /**
664
+ * Loot / foothold ledger for opportunistic exploit chaining (#567).
665
+ *
666
+ * When ON, the attack/discovery/verify agents maintain a typed `LootLedger`
667
+ * (credential | token | path | endpoint | hash | cookie) populated from
668
+ * `save_finding` evidence AND from evidence-bearing tool results
669
+ * (http_request / crawl / submit_form / send_prompt / browser / read_file /
670
+ * bash). A compact "known footholds" block is re-injected into the agent's
671
+ * context each turn (re-rendered from structured state, so it survives
672
+ * compaction), and a `use_loot` tool lets the agent retrieve full artifact
673
+ * values on demand to replay them in follow-up requests. This is the cheap,
674
+ * deterministic alternative to EGATS tree-search (which is disabled) — it
675
+ * stays inside the existing single agent loop, adds no new search layer.
676
+ *
677
+ * Default ON: it's purely additive (extra context awareness + one read-only
678
+ * tool), matches the `preserveCriticalMessages` rationale — recovering a
679
+ * credential in turn 12 that's needed in turn 38 is a large win on long-tail
680
+ * challenges — and the cost (a short, size-capped block per turn) is small.
681
+ * Disable via 0SEC_FEATURE_LOOT_LEDGER=0 or `--no-loot-ledger` for
682
+ * ablation. Implemented as a getter so the CLI `--features` flag (which sets
683
+ * the env var inside the command action, AFTER this module has been
684
+ * imported) is honored at tool-dispatch / injection time — matches the
685
+ * wpFingerprint / preserveCriticalMessages pattern.
686
+ */
687
+ get lootLedger() {
688
+ return env("0SEC_FEATURE_LOOT_LEDGER", true);
689
+ },
690
+ /**
691
+ * Typed TODO / plan ledger (`packages/core/src/agent/task-ledger.ts`).
692
+ *
693
+ * When ON, the agent gets a `plan` tool that maintains a typed, validated
694
+ * task list (add / start / complete / drop / note / list), and the loop
695
+ * re-injects a compact plan block re-rendered from that structured state —
696
+ * so the plan survives `compactMessagesWithLLM` eating the message that
697
+ * carried it. Prior art: Tencent Xuanwu's Atuin moved 68.7% → 84.0% on
698
+ * CyberGym holding the model fixed, and agent-maintained TODO lists are a
699
+ * named component of that harness design.
700
+ *
701
+ * Default ON, on the same reasoning as `lootLedger` above: it is additive
702
+ * (one tool schema plus a size-capped block), it is structured state
703
+ * re-rendered per turn rather than a new search or reasoning layer, and the
704
+ * failure mode of an unused tool is a few hundred wasted schema tokens
705
+ * rather than wrong behavior. Note for whoever publishes benchmark numbers
706
+ * next: this DOES change the default tool list, so re-baseline before
707
+ * quoting a figure across this change. Disable via 0SEC_FEATURE_AGENT_PLAN=0
708
+ * or `--features no-agent-plan` for ablation. Getter so the CLI `--features`
709
+ * flag (which sets the env var AFTER this module is imported) is honored at
710
+ * tool-dispatch time.
711
+ */
712
+ get agentPlan() {
713
+ return env("0SEC_FEATURE_AGENT_PLAN", false);
714
+ },
715
+ /**
716
+ * Task-drift detection (`packages/core/src/agent/drift.ts`).
717
+ *
718
+ * When ON, the loop tracks lexical "anchor contact" between each turn's
719
+ * activity and the objective + open plan tasks, and injects a re-anchoring
720
+ * message when contact has been absent for several consecutive turns. It is
721
+ * a pure function of the trajectory — no LLM call, no network, no per-turn
722
+ * cost. Complements `loopDetection`, which catches an agent repeating itself;
723
+ * drift is the opposite shape (novel activity every turn, none of it on-task)
724
+ * and is invisible to the loop detector.
725
+ *
726
+ * Default OFF, unlike `agentPlan` above, and the asymmetry is deliberate: the
727
+ * plan tool is a capability the model chooses to use, whereas this INJECTS
728
+ * unsolicited steering into a running agent based on a lexical heuristic
729
+ * whose false-positive rate has not been measured. The measurement needs
730
+ * labelled trajectories (replay stored benchmark runs, human-mark which fires
731
+ * were genuine derails) and that corpus does not exist yet — see the honest
732
+ * limitations section in the module doc, particularly that a legitimate pivot
733
+ * to a newly-discovered lead is lexically indistinguishable from a derail.
734
+ * Repo convention is explicit that behavior-steering features stay opt-in
735
+ * until A/B'd, and this is squarely one. Enable via
736
+ * 0SEC_FEATURE_DRIFT_DETECTION=1 or `--features drift-detection`.
737
+ */
738
+ get driftDetection() {
739
+ return env("0SEC_FEATURE_DRIFT_DETECTION", false);
740
+ },
741
+ /**
742
+ * OAST out-of-band interaction collaborator + oracle (#659).
743
+ *
744
+ * When ON, the attack/verify agents get `oast_register` / `oast_poll` to
745
+ * confirm blind/out-of-band classes (blind SSRF/XSS, OOB RCE/SQLi, XXE-OOB,
746
+ * JNDI) via a hosted DNS+HTTP callback server we control, with
747
+ * correlation-token matching. A confirmed callback is disclosure-grade
748
+ * evidence and feeds the loot ledger.
749
+ *
750
+ * Default OFF — the tools are inert without a deployed collaborator. Enable
751
+ * with 0SEC_FEATURE_OAST=1 AND point 0SEC_OAST_URL at the self-hosted
752
+ * collaborator server (see packages/core/src/oast/server.ts). Getter (not a
753
+ * const) so the CLI `--features` flag is honored at tool-dispatch time.
754
+ */
755
+ get oastCollaborator() {
756
+ return env("0SEC_FEATURE_OAST", false);
757
+ },
758
+ /**
759
+ * Anthropic prompt caching (`cache_control: {type: "ephemeral"}`) over the
760
+ * stable request prefix — tool schemas, system prompt, and the settled part
761
+ * of the conversation. See `runtime/prompt-cache.ts` for the placement
762
+ * strategy and the wire contract it encodes.
763
+ *
764
+ * Default ON — and unlike every other flag in this file, that default is not
765
+ * an A/B judgement call. Where a moat layer trades cost for recall, caching
766
+ * is strictly dominant on the axes we care about: the same prompt, the same
767
+ * tokens, the same model output, at ~0.1x input price and materially lower
768
+ * per-turn prefill latency on every turn after the first. The engine
769
+ * re-sends the entire transcript each turn (stateless Messages API), so the
770
+ * saving compounds with conversation length — precisely where the pain is.
771
+ * There is no recall or behaviour dimension to ablate here, which is why
772
+ * this ships enabled rather than waiting on a benchmark.
773
+ *
774
+ * It is also gated on provider support and fails closed: only providers
775
+ * verified to honour `cache_control` receive it, so a non-Anthropic wire can
776
+ * never see an Anthropic-shaped field regardless of this flag (see
777
+ * `providerSupportsPromptCache`).
778
+ *
779
+ * Disable via 0SEC_FEATURE_PROMPT_CACHE=0 — worth doing only to isolate a
780
+ * suspected provider-side caching bug, or to measure the uncached baseline.
781
+ * Implemented as a getter so a late env mutation (CLI `--features`, which
782
+ * runs after this module is imported) is honoured at request-build time.
783
+ */
784
+ get promptCache() {
785
+ return env("0SEC_FEATURE_PROMPT_CACHE", true);
786
+ }
787
+ };
788
+ function presetRaisesDefault(key) {
789
+ const raw = process.env["0SEC_FEATURE_PRESET"];
790
+ if (!raw)
791
+ return false;
792
+ const preset = resolveFeaturePreset(raw);
793
+ if (!preset)
794
+ return false;
795
+ return FEATURE_PRESETS[preset].includes(key);
796
+ }
797
+ function env(key, defaultValue) {
798
+ const val = process.env[key];
799
+ if (val === void 0)
800
+ return defaultValue || presetRaisesDefault(key);
801
+ return val !== "0" && val !== "false";
802
+ }
803
+
804
+ // packages/core/dist/diagnostics/channel.js
805
+ var MAX_MESSAGE_LENGTH = 512;
806
+ var MAX_FIELD_VALUE_LENGTH = 256;
807
+ var MAX_CODE_LENGTH = 64;
808
+ var MAX_FIELDS = 32;
809
+ var MAX_BUFFERED = 200;
810
+ var UNKNOWN_CODE = "unknown";
811
+ var ANSI_PATTERN = /[\u001B\u009B][[\]()#;?]*(?:(?:(?:[a-zA-Z\d]*(?:;[-a-zA-Z\d/#&.:=?%@~_]*)*)?\u0007)|(?:(?:\d{1,4}(?:;\d{0,4})*)?[\dA-PR-TZcf-nq-uy=><~]))/g;
812
+ var CONTROL_PATTERN = /[\u0000-\u001F\u007F-\u009F]/g;
813
+ var UNICODE_BREAK_PATTERN = /[\u2028\u2029]/g;
814
+ function sanitizeText(input, maxLength) {
815
+ let out = input.replace(ANSI_PATTERN, "");
816
+ out = out.replace(CONTROL_PATTERN, " ");
817
+ out = out.replace(UNICODE_BREAK_PATTERN, " ");
818
+ out = out.replace(/\s+/g, " ").trim();
819
+ if (out.length > maxLength) {
820
+ out = out.slice(0, Math.max(0, maxLength - 1)).trimEnd() + "\u2026";
821
+ }
822
+ return out;
823
+ }
824
+ function sanitizeCode(input) {
825
+ const raw = typeof input === "string" ? input : String(input ?? "");
826
+ const slug = raw.toLowerCase().replace(ANSI_PATTERN, "").replace(/[^a-z0-9_.-]+/g, "_").replace(/^[_.-]+|[_.-]+$/g, "").slice(0, MAX_CODE_LENGTH);
827
+ return slug.length > 0 ? slug : UNKNOWN_CODE;
828
+ }
829
+ function sanitizeFieldValue(value) {
830
+ if (value === null)
831
+ return null;
832
+ switch (typeof value) {
833
+ case "undefined":
834
+ return void 0;
835
+ case "boolean":
836
+ return value;
837
+ case "number":
838
+ return Number.isFinite(value) ? value : String(value);
839
+ case "bigint":
840
+ return `${value}n`;
841
+ case "string":
842
+ return sanitizeText(value, MAX_FIELD_VALUE_LENGTH);
843
+ case "function":
844
+ return "[function]";
845
+ case "symbol":
846
+ return sanitizeText(String(value), MAX_FIELD_VALUE_LENGTH);
847
+ default:
848
+ break;
849
+ }
850
+ if (value instanceof Error) {
851
+ return sanitizeText(`${value.name}: ${value.message}`, MAX_FIELD_VALUE_LENGTH);
852
+ }
853
+ if (value instanceof Date) {
854
+ const t = value.getTime();
855
+ return Number.isFinite(t) ? value.toISOString() : "[invalid date]";
856
+ }
857
+ try {
858
+ const json = JSON.stringify(value);
859
+ return json === void 0 ? "[unserializable]" : sanitizeText(json, MAX_FIELD_VALUE_LENGTH);
860
+ } catch {
861
+ return "[unserializable]";
862
+ }
863
+ }
864
+ var EMPTY_FIELDS = Object.freeze({});
865
+ function sanitizeFields(input) {
866
+ if (input == null || typeof input !== "object")
867
+ return EMPTY_FIELDS;
868
+ const out = {};
869
+ let count = 0;
870
+ for (const key of Object.keys(input)) {
871
+ if (count >= MAX_FIELDS)
872
+ break;
873
+ const safeKey = sanitizeCode(key);
874
+ if (Object.prototype.hasOwnProperty.call(out, safeKey))
875
+ continue;
876
+ const value = sanitizeFieldValue(input[key]);
877
+ if (value === void 0)
878
+ continue;
879
+ out[safeKey] = value;
880
+ count++;
881
+ }
882
+ return Object.freeze(out);
883
+ }
884
+ var LEVEL_RANK = { info: 10, warn: 20, error: 30 };
885
+ var OFF_RANK = Number.POSITIVE_INFINITY;
886
+ function minimumRank() {
887
+ const raw = process.env["0SEC_DIAG_LEVEL"];
888
+ if (!raw)
889
+ return LEVEL_RANK.info;
890
+ switch (raw.trim().toLowerCase()) {
891
+ case "off":
892
+ case "none":
893
+ case "silent":
894
+ return OFF_RANK;
895
+ case "error":
896
+ return LEVEL_RANK.error;
897
+ case "warn":
898
+ case "warning":
899
+ return LEVEL_RANK.warn;
900
+ default:
901
+ return LEVEL_RANK.info;
902
+ }
903
+ }
904
+ function formatDiagnosticLine(event) {
905
+ const keys = Object.keys(event.fields);
906
+ if (keys.length === 0)
907
+ return `[0sec] ${event.message}`;
908
+ const rendered = keys.map((k) => `${k}=${event.fields[k]}`).join(" ");
909
+ return `[0sec] ${event.message} (${rendered})`;
910
+ }
911
+ var stderrDiagnosticSink = {
912
+ emit(event) {
913
+ try {
914
+ process.stderr.write(formatDiagnosticLine(event) + "\n");
915
+ } catch {
916
+ }
917
+ }
918
+ };
919
+ function tryEmit(sink, event) {
920
+ try {
921
+ sink.emit(event);
922
+ return true;
923
+ } catch {
924
+ return false;
925
+ }
926
+ }
927
+ var DiagnosticsChannel = class {
928
+ /** Exclusive owner of delivery. When set, the stderr sink is bypassed. */
929
+ claimed = null;
930
+ /** Additive observers. Always receive, claimed or not. */
931
+ observers = [];
932
+ /** Bounded replay ring. */
933
+ buffer = [];
934
+ dropped = 0;
935
+ // ── Emit ────────────────────────────────────────────────────────────────
936
+ info(code, message, fields) {
937
+ this.emit("info", code, message, fields);
938
+ }
939
+ warn(code, message, fields) {
940
+ this.emit("warn", code, message, fields);
941
+ }
942
+ error(code, message, fields) {
943
+ this.emit("error", code, message, fields);
944
+ }
945
+ /**
946
+ * Build, sanitize and deliver one diagnostic.
947
+ *
948
+ * The entire body sits inside a `try` with an empty `catch`: this is called
949
+ * from retry loops, stream readers and `catch` blocks deep inside a scan, and
950
+ * no diagnostic is important enough to justify aborting the work it is
951
+ * describing.
952
+ */
953
+ emit(level, code, message, fields) {
954
+ try {
955
+ if (LEVEL_RANK[level] < minimumRank())
956
+ return;
957
+ const event = Object.freeze({
958
+ level,
959
+ code: sanitizeCode(code),
960
+ message: sanitizeText(typeof message === "string" ? message : String(message ?? ""), MAX_MESSAGE_LENGTH),
961
+ fields: sanitizeFields(fields),
962
+ ts: Date.now()
963
+ });
964
+ this.remember(event);
965
+ this.deliver(event);
966
+ } catch {
967
+ }
968
+ }
969
+ remember(event) {
970
+ this.buffer.push(event);
971
+ while (this.buffer.length > MAX_BUFFERED) {
972
+ this.buffer.shift();
973
+ this.dropped++;
974
+ }
975
+ }
976
+ deliver(event) {
977
+ const claimed = this.claimed;
978
+ if (claimed) {
979
+ if (!tryEmit(claimed, event))
980
+ stderrDiagnosticSink.emit(event);
981
+ } else {
982
+ stderrDiagnosticSink.emit(event);
983
+ }
984
+ for (const observer of this.observers.slice())
985
+ tryEmit(observer, event);
986
+ }
987
+ // ── Subscription ────────────────────────────────────────────────────────
988
+ /**
989
+ * Take exclusive ownership of delivery. While claimed, the built-in stderr
990
+ * sink is bypassed — this is how the TUI stops core from writing to the
991
+ * terminal it is painting.
992
+ *
993
+ * Claims nest: a second claim supersedes the first, and releasing it restores
994
+ * the previous owner. Release is idempotent and only un-claims while this
995
+ * claim is still the live one, so an out-of-order teardown cannot resurrect a
996
+ * dead sink.
997
+ */
998
+ claim(sink, options = {}) {
999
+ const previous = this.claimed;
1000
+ this.claimed = sink;
1001
+ if (options.replay) {
1002
+ for (const event of this.buffer.slice())
1003
+ tryEmit(sink, event);
1004
+ }
1005
+ let released = false;
1006
+ return () => {
1007
+ if (released)
1008
+ return;
1009
+ released = true;
1010
+ if (this.claimed === sink)
1011
+ this.claimed = previous;
1012
+ };
1013
+ }
1014
+ /**
1015
+ * Add an observer that receives every diagnostic regardless of who holds the
1016
+ * claim. For tracing exporters, test spies, and audit logs — anything that
1017
+ * wants a copy but must not silence stderr.
1018
+ */
1019
+ subscribe(sink) {
1020
+ this.observers.push(sink);
1021
+ return () => {
1022
+ const idx = this.observers.indexOf(sink);
1023
+ if (idx >= 0)
1024
+ this.observers.splice(idx, 1);
1025
+ };
1026
+ }
1027
+ /** True when some sink holds the exclusive claim (stderr is bypassed). */
1028
+ get isClaimed() {
1029
+ return this.claimed !== null;
1030
+ }
1031
+ /** Number of additive observers. */
1032
+ get observerCount() {
1033
+ return this.observers.length;
1034
+ }
1035
+ /** Recent diagnostics, oldest first. Bounded by `MAX_BUFFERED`. */
1036
+ recent() {
1037
+ return this.buffer.slice();
1038
+ }
1039
+ /** How many buffered diagnostics were evicted to stay within the bound. */
1040
+ droppedCount() {
1041
+ return this.dropped;
1042
+ }
1043
+ /** Test-only: drop all sinks, the replay buffer, and the drop counter. */
1044
+ resetForTests() {
1045
+ this.claimed = null;
1046
+ this.observers = [];
1047
+ this.buffer = [];
1048
+ this.dropped = 0;
1049
+ }
1050
+ };
1051
+ var diag = new DiagnosticsChannel();
1052
+ function claimDiagnostics(sink, options) {
1053
+ return diag.claim(sink, options);
1054
+ }
1055
+ function subscribeDiagnostics(sink) {
1056
+ return diag.subscribe(sink);
1057
+ }
1058
+ function isDiagnosticsClaimed() {
1059
+ return diag.isClaimed;
1060
+ }
1061
+ function recentDiagnostics() {
1062
+ return diag.recent();
1063
+ }
1064
+ function _resetDiagnosticsForTests() {
1065
+ diag.resetForTests();
1066
+ }
1067
+
1068
+ // packages/core/dist/cloud/credentials.js
1069
+ import { readFileSync, statSync } from "node:fs";
1070
+ import { join } from "node:path";
1071
+ var DEFAULT_CLOUD_HOST = "https://cloud.0.security";
1072
+ var CloudAuthMissingError = class extends Error {
1073
+ constructor(message) {
1074
+ super(message);
1075
+ this.name = "CloudAuthMissingError";
1076
+ }
1077
+ };
1078
+ var CloudAuthError = class extends Error {
1079
+ status;
1080
+ constructor(message, status) {
1081
+ super(message);
1082
+ this.status = status;
1083
+ this.name = "CloudAuthError";
1084
+ }
1085
+ };
1086
+ function loadCloudCredentials(opts = {}) {
1087
+ const env2 = opts.env ?? process.env;
1088
+ const warn = opts.warn ?? ((m) => process.stderr.write(`${m}
1089
+ `));
1090
+ const envTok = env2["0SEC_CLOUD_TOKEN"]?.trim();
1091
+ if (envTok) {
1092
+ const envHost = normaliseHost(env2["0SEC_CLOUD_HOST"]?.trim() ?? DEFAULT_CLOUD_HOST);
1093
+ return { host: envHost, token: envTok, source: "env" };
1094
+ }
1095
+ const path = join(homeStateDir(opts.homeDir), "cloud.env");
1096
+ let raw;
1097
+ try {
1098
+ raw = readFileSync(path, "utf-8");
1099
+ } catch (err) {
1100
+ const code = err.code;
1101
+ if (code === "ENOENT") {
1102
+ throw new CloudAuthMissingError(`0sec-cloud credentials not found. Run \`0sec auth login\` or set 0SEC_CLOUD_TOKEN in env, or create ${path} (chmod 600) with 0SEC_CLOUD_TOKEN=\u2026 (optionally 0SEC_CLOUD_HOST=\u2026).`);
1103
+ }
1104
+ throw err;
1105
+ }
1106
+ try {
1107
+ const st = statSync(path);
1108
+ const mode = st.mode & 511;
1109
+ if (mode !== 384) {
1110
+ warn(`[0sec cloud] WARNING: ${path} mode is ${mode.toString(8).padStart(3, "0")} (expected 600). Run: chmod 600 ${path}`);
1111
+ }
1112
+ } catch {
1113
+ }
1114
+ const parsed = parseEnvFile(raw);
1115
+ const fileTok = parsed["0SEC_CLOUD_TOKEN"]?.trim();
1116
+ if (!fileTok) {
1117
+ throw new CloudAuthMissingError(`0sec-cloud credentials in ${path} are incomplete: 0SEC_CLOUD_TOKEN is required.`);
1118
+ }
1119
+ const fileHost = normaliseHost(parsed["0SEC_CLOUD_HOST"]?.trim() ?? DEFAULT_CLOUD_HOST);
1120
+ return { host: fileHost, token: fileTok, source: "file" };
1121
+ }
1122
+ function normaliseHost(host) {
1123
+ let h = host;
1124
+ if (!/^https?:\/\//.test(h)) {
1125
+ throw new CloudAuthMissingError(`0SEC_CLOUD_HOST must be an http(s) URL (got ${JSON.stringify(host)}).`);
1126
+ }
1127
+ while (h.endsWith("/"))
1128
+ h = h.slice(0, -1);
1129
+ return h;
1130
+ }
1131
+ function parseEnvFile(raw) {
1132
+ const out = {};
1133
+ const lines = raw.split(/\r?\n/);
1134
+ for (let i = 0; i < lines.length; i++) {
1135
+ const line = lines[i];
1136
+ const trimmed = line.trim();
1137
+ if (trimmed.length === 0)
1138
+ continue;
1139
+ if (trimmed.startsWith("#"))
1140
+ continue;
1141
+ const eq = trimmed.indexOf("=");
1142
+ if (eq <= 0) {
1143
+ throw new CloudAuthMissingError(`Malformed cloud.env at line ${i + 1}: expected KEY=VALUE, got ${JSON.stringify(line)}`);
1144
+ }
1145
+ const key = trimmed.slice(0, eq).trim();
1146
+ const value = trimmed.slice(eq + 1).trim();
1147
+ if (!/^[A-Z0-9_]+$/.test(key)) {
1148
+ throw new CloudAuthMissingError(`Malformed cloud.env at line ${i + 1}: invalid key ${JSON.stringify(key)}`);
1149
+ }
1150
+ out[key] = value;
1151
+ }
1152
+ return out;
1153
+ }
1154
+
1155
+ // packages/core/dist/cloud/client.js
1156
+ var CloudError = class extends Error {
1157
+ status;
1158
+ path;
1159
+ code;
1160
+ constructor(message, status, path, code) {
1161
+ super(message);
1162
+ this.status = status;
1163
+ this.path = path;
1164
+ this.code = code;
1165
+ this.name = "CloudError";
1166
+ }
1167
+ };
1168
+ var CloudUnauthorizedError = class extends CloudError {
1169
+ constructor(path) {
1170
+ super(`0sec-cloud auth rejected (HTTP 401) on ${path}. Run \`0sec auth login\` to refresh.`, 401, path);
1171
+ this.name = "CloudUnauthorizedError";
1172
+ }
1173
+ };
1174
+ var CloudForbiddenError = class extends CloudError {
1175
+ constructor(path) {
1176
+ super(`0sec-cloud forbidden (HTTP 403) on ${path}. Token lacks scope for this resource.`, 403, path);
1177
+ this.name = "CloudForbiddenError";
1178
+ }
1179
+ };
1180
+ var CloudNetworkError = class extends CloudError {
1181
+ constructor(message, path) {
1182
+ super(`0sec-cloud network error on ${path}: ${message}`, void 0, path);
1183
+ this.name = "CloudNetworkError";
1184
+ }
1185
+ };
1186
+ function healthPath(host) {
1187
+ try {
1188
+ const hostname = new URL(host).hostname.toLowerCase();
1189
+ if (hostname === "cloud.0sec.ai" || hostname === "cloud.0.security") {
1190
+ return "/api/health";
1191
+ }
1192
+ } catch {
1193
+ }
1194
+ return "/health";
1195
+ }
1196
+ var CloudClient = class {
1197
+ host;
1198
+ token;
1199
+ fetchImpl;
1200
+ constructor(opts) {
1201
+ this.host = opts.host;
1202
+ this.token = opts.token;
1203
+ this.fetchImpl = opts.fetchImpl ?? fetch;
1204
+ }
1205
+ /**
1206
+ * Verify cloud reachability through its health route. The hosted dashboard
1207
+ * uses `/api/health`; a self-hosted receiver uses `/health`.
1208
+ */
1209
+ async pingHealth() {
1210
+ return this.getJson(healthPath(this.host));
1211
+ }
1212
+ /**
1213
+ * Fetch the hosted inference model catalog — available models, pricing,
1214
+ * wire API protocol, and context limits. Used at runtime for model
1215
+ * selection and by the hosted provider to determine per-model capabilities.
1216
+ * Returns the raw list response; the caller caches/filters as needed.
1217
+ */
1218
+ async getInferenceModels() {
1219
+ return this.getJson("/api/inference/v1/models");
1220
+ }
1221
+ /**
1222
+ * Fetch the organization's hosted inference credit availability. The service
1223
+ * calculates the percentage from Autumn's current pool; holds reduce availability.
1224
+ * Older gateways without percentage metadata remain explicitly unavailable.
1225
+ */
1226
+ async getInferenceAccount() {
1227
+ const account = await this.getJson("/api/inference/account");
1228
+ const credits = account.credits;
1229
+ if (!credits || typeof credits !== "object" || typeof credits.featureId !== "string" || !credits.featureId.trim() || typeof credits.remaining !== "number" || !Number.isFinite(credits.remaining) || credits.remaining < 0 || credits.granted !== null && (typeof credits.granted !== "number" || !Number.isFinite(credits.granted) || credits.granted < 0)) {
1230
+ return { ...account, credits: null };
1231
+ }
1232
+ const remainingPercent = typeof credits.remainingPercent === "number" && Number.isFinite(credits.remainingPercent) && credits.remainingPercent >= 0 && credits.remainingPercent <= 100 && credits.granted !== null && credits.granted > 0 && credits.remaining <= credits.granted ? credits.remainingPercent : null;
1233
+ const nextResetAt = typeof credits.nextResetAt === "number" && Number.isSafeInteger(credits.nextResetAt) && credits.nextResetAt > 0 && credits.nextResetAt <= 864e13 ? credits.nextResetAt : null;
1234
+ return { ...account, credits: {
1235
+ featureId: credits.featureId,
1236
+ granted: credits.granted,
1237
+ remaining: credits.remaining,
1238
+ remainingPercent,
1239
+ nextResetAt
1240
+ } };
1241
+ }
1242
+ /**
1243
+ * Fetch request-level usage metadata for the operator's hosted
1244
+ * inference sessions. Returns lightweight metadata records (model,
1245
+ * tokens, provider, timestamp) — no prompt/response payload.
1246
+ */
1247
+ async getInferenceUsage() {
1248
+ return this.getJson("/api/inference/usage");
1249
+ }
1250
+ /**
1251
+ * Generic JSON GET helper. Public so future modules (scans, findings)
1252
+ * can reuse the same error mapping without duplicating it. Not exported
1253
+ * past the package boundary — see ./index.ts.
1254
+ */
1255
+ async getJson(path) {
1256
+ const url = `${this.host}${path}`;
1257
+ let res;
1258
+ try {
1259
+ res = await this.fetchImpl(url, {
1260
+ method: "GET",
1261
+ headers: this.headers()
1262
+ });
1263
+ } catch (err) {
1264
+ const msg = err instanceof Error ? err.message : String(err);
1265
+ throw new CloudNetworkError(this.scrub(msg), path);
1266
+ }
1267
+ if (!res.ok) {
1268
+ let code;
1269
+ try {
1270
+ const body = await res.json();
1271
+ const raw = body?.error?.code;
1272
+ if (typeof raw === "string" && raw.length > 0)
1273
+ code = raw;
1274
+ } catch {
1275
+ }
1276
+ this.throwForStatus(res.status, path, code);
1277
+ }
1278
+ return await res.json();
1279
+ }
1280
+ /** Throw a typed error for a non-2xx status, carrying the gateway's body `code`. */
1281
+ throwForStatus(status, path, code) {
1282
+ if (status === 401)
1283
+ throw new CloudUnauthorizedError(path);
1284
+ if (status === 403)
1285
+ throw new CloudForbiddenError(path);
1286
+ throw new CloudError(`0sec-cloud request failed (HTTP ${status}${code ? ` ${code}` : ""}) on ${path}.`, status, path, code);
1287
+ }
1288
+ /**
1289
+ * Throw a typed error for non-2xx responses. Public so direct callers
1290
+ * (e.g. an integration test driving raw fetch) can reuse the mapping.
1291
+ */
1292
+ assertOk(res, path) {
1293
+ if (res.ok)
1294
+ return;
1295
+ if (res.status === 401)
1296
+ throw new CloudUnauthorizedError(path);
1297
+ if (res.status === 403)
1298
+ throw new CloudForbiddenError(path);
1299
+ throw new CloudError(`0sec-cloud request failed (HTTP ${res.status}) on ${path}.`, res.status, path);
1300
+ }
1301
+ // ── internals ──
1302
+ headers() {
1303
+ return {
1304
+ Authorization: `Bearer ${this.token}`,
1305
+ Accept: "application/json",
1306
+ "User-Agent": `0sec-cli/${VERSION}`
1307
+ };
1308
+ }
1309
+ /**
1310
+ * Strip anything that looks like our own token from a string. The
1311
+ * cloud token may be interpolated into a TLS-layer error message in
1312
+ * exotic failure modes — we redact it to keep the no-leak invariant
1313
+ * local to this module.
1314
+ */
1315
+ scrub(s) {
1316
+ if (!this.token)
1317
+ return s;
1318
+ return s.split(this.token).join("[REDACTED]");
1319
+ }
1320
+ };
1321
+
1322
+ // packages/core/dist/runtime/llm-api.js
1323
+ import { randomUUID } from "node:crypto";
1324
+ import { appendFileSync, existsSync, readFileSync as readFileSync2, renameSync, writeFileSync } from "node:fs";
1325
+ import { homedir } from "node:os";
1326
+ import { join as join2 } from "node:path";
1327
+
1328
+ // packages/core/dist/runtime/prompt-cache.js
1329
+ var MAX_CACHE_BREAKPOINTS = 4;
1330
+ var MESSAGE_CACHE_BREAKPOINTS = MAX_CACHE_BREAKPOINTS - 1;
1331
+ var BREAKPOINT_SPACING_BLOCKS = 15;
1332
+ var NATIVE_CACHE_PROVIDERS = /* @__PURE__ */ new Set(["anthropic"]);
1333
+ var OPT_IN_CACHE_PROVIDERS = /* @__PURE__ */ new Set(["z-ai", "kimi"]);
1334
+ function providerSupportsPromptCache(provider) {
1335
+ if (NATIVE_CACHE_PROVIDERS.has(provider))
1336
+ return true;
1337
+ if (!OPT_IN_CACHE_PROVIDERS.has(provider))
1338
+ return false;
1339
+ return readExtraCacheProviders().has(provider);
1340
+ }
1341
+ function readExtraCacheProviders() {
1342
+ const raw = process.env["0SEC_PROMPT_CACHE_EXTRA_PROVIDERS"];
1343
+ if (!raw)
1344
+ return /* @__PURE__ */ new Set();
1345
+ return new Set(raw.split(",").map((entry) => entry.trim().toLowerCase()).filter((entry) => entry.length > 0));
1346
+ }
1347
+ function planMessageBreakpoints(messages, budget = MESSAGE_CACHE_BREAKPOINTS) {
1348
+ if (messages.length === 0 || budget <= 0)
1349
+ return [];
1350
+ const lastIndex = messages.length - 1;
1351
+ const picks = /* @__PURE__ */ new Set([lastIndex]);
1352
+ const rollingBudget = Math.max(1, budget - 1);
1353
+ let blocksSincePick = blockCount(messages[lastIndex]);
1354
+ for (let i = lastIndex - 1; i >= 0 && picks.size < rollingBudget; i--) {
1355
+ if (blocksSincePick >= BREAKPOINT_SPACING_BLOCKS) {
1356
+ picks.add(i);
1357
+ blocksSincePick = 0;
1358
+ }
1359
+ blocksSincePick += blockCount(messages[i]);
1360
+ }
1361
+ if (picks.size < budget)
1362
+ picks.add(0);
1363
+ return [...picks].sort((a, b) => a - b).slice(0, budget);
1364
+ }
1365
+ function blockCount(message) {
1366
+ return message?.content.length ?? 0;
1367
+ }
1368
+ function withCacheControl(block) {
1369
+ return { ...block, cache_control: { type: "ephemeral" } };
1370
+ }
1371
+ function readCacheUsage(raw) {
1372
+ if (typeof raw !== "object" || raw === null)
1373
+ return void 0;
1374
+ const usage = raw;
1375
+ const uncachedInput = numberOr(usage.input_tokens, 0);
1376
+ const outputTokens = numberOr(usage.output_tokens, 0);
1377
+ const cacheRead = numberOr(usage.cache_read_input_tokens, void 0);
1378
+ const cacheWrite = numberOr(usage.cache_creation_input_tokens, void 0);
1379
+ return {
1380
+ inputTokens: uncachedInput + (cacheRead ?? 0) + (cacheWrite ?? 0),
1381
+ outputTokens,
1382
+ ...cacheRead !== void 0 ? { cachedInputTokens: cacheRead } : {},
1383
+ ...cacheWrite !== void 0 ? { cacheWriteTokens: cacheWrite } : {}
1384
+ };
1385
+ }
1386
+ function numberOr(value, fallback) {
1387
+ return typeof value === "number" && Number.isFinite(value) ? value : fallback;
1388
+ }
1389
+
1390
+ // packages/core/dist/runtime/llm-api.js
1391
+ var NATIVE_COMPLETION_TOKEN_LIMIT = 8192;
1392
+ function readResponsesCachedTokens(usage) {
1393
+ const details = usage.input_tokens_details;
1394
+ const cached = Number(details?.cached_tokens ?? 0);
1395
+ return Number.isFinite(cached) && cached > 0 ? { cachedInputTokens: cached } : {};
1396
+ }
1397
+ function safeParseJson(raw) {
1398
+ if (!raw)
1399
+ return {};
1400
+ try {
1401
+ return JSON.parse(raw);
1402
+ } catch {
1403
+ return { _raw: raw };
1404
+ }
1405
+ }
1406
+ function isWireBlockArray(blocks) {
1407
+ return blocks.every((block) => block !== null && typeof block === "object" && !Array.isArray(block) && typeof block.type === "string");
1408
+ }
1409
+ var azureRegionCache = /* @__PURE__ */ new Map();
1410
+ async function probeAzureRegion(baseUrl, apiKey, fetchImpl = fetch) {
1411
+ const override = process.env["0SEC_REGION_OVERRIDE"];
1412
+ if (override && override.trim().length > 0) {
1413
+ return override.trim();
1414
+ }
1415
+ const cached = azureRegionCache.get(baseUrl);
1416
+ if (cached)
1417
+ return cached;
1418
+ try {
1419
+ const controller = new AbortController();
1420
+ const timer = setTimeout(() => controller.abort(), 5e3);
1421
+ const res = await fetchImpl(`${baseUrl.replace(/\/+$/, "")}/models`, {
1422
+ method: "GET",
1423
+ headers: { "api-key": apiKey },
1424
+ signal: controller.signal
1425
+ }).finally(() => clearTimeout(timer));
1426
+ const region = res.headers.get("x-ms-region");
1427
+ const resolved = region && region.trim().length > 0 ? prettyRegion(region.trim()) : "unknown";
1428
+ azureRegionCache.set(baseUrl, resolved);
1429
+ return resolved;
1430
+ } catch {
1431
+ azureRegionCache.set(baseUrl, "unknown");
1432
+ return "unknown";
1433
+ }
1434
+ }
1435
+ function prettyRegion(code) {
1436
+ const map = {
1437
+ eastus: "East US",
1438
+ eastus2: "East US 2",
1439
+ westus: "West US",
1440
+ westus2: "West US 2",
1441
+ westus3: "West US 3",
1442
+ centralus: "Central US",
1443
+ northcentralus: "North Central US",
1444
+ southcentralus: "South Central US",
1445
+ westcentralus: "West Central US",
1446
+ canadaeast: "Canada East",
1447
+ canadacentral: "Canada Central",
1448
+ brazilsouth: "Brazil South",
1449
+ northeurope: "North Europe",
1450
+ westeurope: "West Europe",
1451
+ uksouth: "UK South",
1452
+ ukwest: "UK West",
1453
+ francecentral: "France Central",
1454
+ germanywestcentral: "Germany West Central",
1455
+ switzerlandnorth: "Switzerland North",
1456
+ norwayeast: "Norway East",
1457
+ swedencentral: "Sweden Central",
1458
+ polandcentral: "Poland Central",
1459
+ italynorth: "Italy North",
1460
+ eastasia: "East Asia",
1461
+ southeastasia: "Southeast Asia",
1462
+ japaneast: "Japan East",
1463
+ japanwest: "Japan West",
1464
+ koreacentral: "Korea Central",
1465
+ australiaeast: "Australia East",
1466
+ centralindia: "Central India",
1467
+ southindia: "South India",
1468
+ uaenorth: "UAE North",
1469
+ southafricanorth: "South Africa North"
1470
+ };
1471
+ return map[code.toLowerCase()] ?? code;
1472
+ }
1473
+ var PROVIDER_BANNER_KEY = /* @__PURE__ */ Symbol.for("0sec.core.loggedProviderStartup");
1474
+ var loggedProviderStartup = (() => {
1475
+ const g = globalThis;
1476
+ if (!g[PROVIDER_BANNER_KEY])
1477
+ g[PROVIDER_BANNER_KEY] = /* @__PURE__ */ new Set();
1478
+ return g[PROVIDER_BANNER_KEY];
1479
+ })();
1480
+ function appendNativeTrace(record) {
1481
+ const file = process.env["0SEC_TRACE_NATIVE_RESPONSES"];
1482
+ if (!file)
1483
+ return;
1484
+ try {
1485
+ appendFileSync(file, `${JSON.stringify({ ts: (/* @__PURE__ */ new Date()).toISOString(), ...record })}
1486
+ `, "utf8");
1487
+ } catch {
1488
+ }
1489
+ }
1490
+ function shouldLogProviderStartup() {
1491
+ return process.env["0SEC_SUPPRESS_PROVIDER_STARTUP_LOG"] !== "1";
1492
+ }
1493
+ function isRetryableHttpStatus(status) {
1494
+ return status === 429 || status === 500 || status === 502 || status === 503 || status === 504;
1495
+ }
1496
+ var TRANSIENT_STREAM_ERROR_PATTERNS = [
1497
+ "stream completed without final response",
1498
+ "response stream failed"
1499
+ ];
1500
+ function llmStreamMaxAttempts() {
1501
+ const raw = process.env["0SEC_LLM_STREAM_MAX_ATTEMPTS"];
1502
+ if (raw == null || raw.trim() === "")
1503
+ return 3;
1504
+ const n = Number.parseInt(raw, 10);
1505
+ return Number.isFinite(n) && n >= 1 ? Math.min(n, 5) : 3;
1506
+ }
1507
+ function streamRetryBackoffMs(retry) {
1508
+ return retry <= 1 ? 500 : 1e3;
1509
+ }
1510
+ function delayWithAbort(ms, signal) {
1511
+ if (signal?.aborted)
1512
+ return Promise.resolve();
1513
+ return new Promise((resolve) => {
1514
+ const timer = setTimeout(() => {
1515
+ signal?.removeEventListener("abort", onAbort);
1516
+ resolve();
1517
+ }, ms);
1518
+ const onAbort = () => {
1519
+ clearTimeout(timer);
1520
+ resolve();
1521
+ };
1522
+ signal?.addEventListener("abort", onAbort, { once: true });
1523
+ });
1524
+ }
1525
+ function shouldRetryNativeStream(result) {
1526
+ if (result.stopReason !== "error")
1527
+ return false;
1528
+ if (result.cancelled)
1529
+ return false;
1530
+ const error = result.error ?? "";
1531
+ if (!TRANSIENT_STREAM_ERROR_PATTERNS.some((pattern) => error.includes(pattern)))
1532
+ return false;
1533
+ if (result.content.some((block) => block.type === "tool_use"))
1534
+ return false;
1535
+ return true;
1536
+ }
1537
+ function isRetryableTransportCode(code) {
1538
+ return [
1539
+ "EAI_AGAIN",
1540
+ "ECONNRESET",
1541
+ "ENOTFOUND",
1542
+ "ETIMEDOUT",
1543
+ "UND_ERR_CONNECT_TIMEOUT",
1544
+ "UND_ERR_SOCKET"
1545
+ ].includes(code);
1546
+ }
1547
+ function llmMaxRetries() {
1548
+ const raw = process.env["0SEC_LLM_MAX_RETRIES"];
1549
+ if (raw == null || raw.trim() === "")
1550
+ return 6;
1551
+ const n = Number.parseInt(raw, 10);
1552
+ return Number.isFinite(n) && n >= 0 ? n : 6;
1553
+ }
1554
+ function llmMaxRetryWaitMs() {
1555
+ const raw = process.env["0SEC_LLM_MAX_RETRY_WAIT_MS"];
1556
+ if (raw == null || raw.trim() === "")
1557
+ return 6e4;
1558
+ const n = Number.parseInt(raw, 10);
1559
+ return Number.isFinite(n) && n > 0 ? n : 6e4;
1560
+ }
1561
+ function llm429MaxRetries() {
1562
+ const raw = process.env["0SEC_LLM_429_MAX_RETRIES"] ?? process.env["0SEC_LLM_MAX_RETRIES"];
1563
+ if (raw == null || raw.trim() === "")
1564
+ return 12;
1565
+ const n = Number.parseInt(raw, 10);
1566
+ return Number.isFinite(n) && n >= 0 ? n : 12;
1567
+ }
1568
+ function llm429MaxRetryWaitMs() {
1569
+ const raw = process.env["0SEC_LLM_429_MAX_RETRY_WAIT_MS"] ?? process.env["0SEC_LLM_MAX_RETRY_WAIT_MS"];
1570
+ if (raw == null || raw.trim() === "")
1571
+ return 3e5;
1572
+ const n = Number.parseInt(raw, 10);
1573
+ return Number.isFinite(n) && n > 0 ? n : 3e5;
1574
+ }
1575
+ var QuotaExhaustedError = class extends Error {
1576
+ name = "QuotaExhaustedError";
1577
+ quotaKind;
1578
+ planType;
1579
+ resetsAtMs;
1580
+ resetsInSeconds;
1581
+ constructor(message, details) {
1582
+ super(message);
1583
+ this.quotaKind = details.quotaKind;
1584
+ this.planType = details.planType;
1585
+ this.resetsAtMs = details.resetsAtMs;
1586
+ this.resetsInSeconds = details.resetsInSeconds;
1587
+ }
1588
+ };
1589
+ var OperatorAbortError = class extends Error {
1590
+ name = "OperatorAbortError";
1591
+ constructor(message = "request cancelled by operator") {
1592
+ super(message);
1593
+ }
1594
+ };
1595
+ var NO_OPERATOR_ABORT = {
1596
+ operatorAborted: () => false,
1597
+ throwIfCancelled: () => {
1598
+ },
1599
+ dispose: () => {
1600
+ }
1601
+ };
1602
+ function manualAnySignal(sources, detach) {
1603
+ const merged = new AbortController();
1604
+ for (const source of sources) {
1605
+ if (source.aborted) {
1606
+ merged.abort(source.reason);
1607
+ return merged.signal;
1608
+ }
1609
+ source.addEventListener("abort", () => merged.abort(source.reason), {
1610
+ once: true,
1611
+ signal: detach
1612
+ });
1613
+ }
1614
+ return merged.signal;
1615
+ }
1616
+ function composeCallAbort(timeout, operator) {
1617
+ if (!operator)
1618
+ return { signal: timeout, ...NO_OPERATOR_ABORT };
1619
+ let operatorFired = operator.aborted;
1620
+ const detach = new AbortController();
1621
+ operator.addEventListener("abort", () => {
1622
+ if (!timeout.aborted)
1623
+ operatorFired = true;
1624
+ }, { once: true, signal: detach.signal });
1625
+ const signal = typeof AbortSignal.any === "function" ? AbortSignal.any([timeout, operator]) : manualAnySignal([timeout, operator], detach.signal);
1626
+ return {
1627
+ signal,
1628
+ operator,
1629
+ operatorAborted: () => operatorFired,
1630
+ throwIfCancelled: () => {
1631
+ if (operatorFired)
1632
+ throw new OperatorAbortError();
1633
+ },
1634
+ dispose: () => detach.abort()
1635
+ };
1636
+ }
1637
+ function parseUsageLimitReached(body) {
1638
+ let json;
1639
+ try {
1640
+ json = JSON.parse(body);
1641
+ } catch {
1642
+ return void 0;
1643
+ }
1644
+ const root = typeof json === "object" && json !== null ? json : void 0;
1645
+ const err = typeof root?.error === "object" && root.error !== null ? root.error : root;
1646
+ const errorType = typeof err?.type === "string" ? err.type : void 0;
1647
+ const errorCode = typeof err?.code === "string" ? err.code : void 0;
1648
+ const quotaKind = errorType === "usage_limit_reached" ? "usage_limit_reached" : errorType === "insufficient_quota" || errorCode === "insufficient_quota" ? "insufficient_quota" : void 0;
1649
+ if (!quotaKind || !err)
1650
+ return void 0;
1651
+ const details = { quotaKind };
1652
+ if (typeof err.plan_type === "string") {
1653
+ details.planType = err.plan_type;
1654
+ } else if (quotaKind === "insufficient_quota") {
1655
+ details.planType = "token-plan";
1656
+ }
1657
+ if (typeof err.resets_in_seconds === "number" && Number.isFinite(err.resets_in_seconds)) {
1658
+ details.resetsInSeconds = err.resets_in_seconds;
1659
+ }
1660
+ if (typeof err.resets_at === "number" && Number.isFinite(err.resets_at)) {
1661
+ details.resetsAtMs = err.resets_at > 1e12 ? err.resets_at : err.resets_at * 1e3;
1662
+ }
1663
+ if (details.resetsAtMs == null && details.resetsInSeconds != null) {
1664
+ details.resetsAtMs = Date.now() + details.resetsInSeconds * 1e3;
1665
+ }
1666
+ if (details.resetsAtMs == null && quotaKind === "insufficient_quota") {
1667
+ const message = typeof err.message === "string" ? err.message : "";
1668
+ const m = message.match(/resets? at (\d{2})-(\d{2}) (\d{2}):(\d{2}):(\d{2})(?:\s*UTC)?/);
1669
+ if (m) {
1670
+ const now = Date.now();
1671
+ const year = new Date(now).getUTCFullYear();
1672
+ const at = (y) => Date.UTC(y, Number(m[1]) - 1, Number(m[2]), Number(m[3]), Number(m[4]), Number(m[5]));
1673
+ const ts = at(year);
1674
+ details.resetsAtMs = ts > now ? ts : at(year + 1);
1675
+ }
1676
+ }
1677
+ return details;
1678
+ }
1679
+ function llmStreamIdleTimeoutMs() {
1680
+ const raw = process.env["0SEC_LLM_STREAM_IDLE_TIMEOUT_MS"];
1681
+ if (raw == null || raw.trim() === "")
1682
+ return 12e4;
1683
+ const n = Number.parseInt(raw, 10);
1684
+ return Number.isFinite(n) && n > 0 ? n : 12e4;
1685
+ }
1686
+ function parseRetryAfterMs(headerValue) {
1687
+ if (!headerValue)
1688
+ return void 0;
1689
+ const trimmed = headerValue.trim();
1690
+ if (trimmed === "")
1691
+ return void 0;
1692
+ if (/^\d+$/.test(trimmed))
1693
+ return Number.parseInt(trimmed, 10) * 1e3;
1694
+ const dateMs = Date.parse(trimmed);
1695
+ if (Number.isFinite(dateMs)) {
1696
+ const delta = dateMs - Date.now();
1697
+ return delta > 0 ? delta : 0;
1698
+ }
1699
+ return void 0;
1700
+ }
1701
+ function retryBackoffMs(attempt, ceilingMs = 2e4) {
1702
+ const ceiling = Math.min(ceilingMs, 500 * 2 ** attempt);
1703
+ return Math.floor(Math.random() * ceiling) + 250;
1704
+ }
1705
+ var RETRY_AFTER_CAP_MS = 12e4;
1706
+ function retryAfterMsFromHeaders(headers) {
1707
+ const msHeader = headers?.get?.("retry-after-ms");
1708
+ if (msHeader != null) {
1709
+ const n = Number.parseInt(msHeader.trim(), 10);
1710
+ if (Number.isFinite(n) && n >= 0)
1711
+ return Math.min(n, RETRY_AFTER_CAP_MS);
1712
+ }
1713
+ const parsed = parseRetryAfterMs(headers?.get?.("retry-after"));
1714
+ return parsed != null ? Math.min(parsed, RETRY_AFTER_CAP_MS) : void 0;
1715
+ }
1716
+ function sleepWithAbort(ms, signal) {
1717
+ return new Promise((resolve, reject) => {
1718
+ if (signal.aborted) {
1719
+ reject(new DOMException("Aborted during retry backoff", "AbortError"));
1720
+ return;
1721
+ }
1722
+ const t = setTimeout(resolve, ms);
1723
+ signal.addEventListener("abort", () => {
1724
+ clearTimeout(t);
1725
+ reject(new DOMException("Aborted during retry backoff", "AbortError"));
1726
+ }, { once: true });
1727
+ });
1728
+ }
1729
+ function defaultReasoningEffort(model) {
1730
+ const lower = model.toLowerCase();
1731
+ if (/gpt-[56](?:[-.]|$)/.test(lower) || /^o[134]/.test(lower))
1732
+ return "medium";
1733
+ return void 0;
1734
+ }
1735
+ async function logProviderStartup(provider, providerLabel, baseUrl, model, wireApi, apiKey, fetchImpl = fetch) {
1736
+ const key = `${provider}:${baseUrl}`;
1737
+ if (loggedProviderStartup.has(key))
1738
+ return;
1739
+ loggedProviderStartup.add(key);
1740
+ if (!shouldLogProviderStartup())
1741
+ return;
1742
+ if (provider !== "azure") {
1743
+ diag.info("provider_initialized", `${providerLabel} provider initialized`, {
1744
+ provider,
1745
+ endpoint: baseUrl,
1746
+ model
1747
+ });
1748
+ return;
1749
+ }
1750
+ const region = await probeAzureRegion(baseUrl, apiKey, fetchImpl);
1751
+ diag.info("provider_initialized", "Azure OpenAI provider initialized", {
1752
+ provider,
1753
+ endpoint: baseUrl,
1754
+ model,
1755
+ region,
1756
+ // Distinguishes "the header said westeurope" from "the probe could not
1757
+ // tell us", which matters when someone is debugging a data-residency
1758
+ // requirement and `region=unknown` is not the same as `region` missing.
1759
+ region_source: region === "unknown" ? "probe-failed" : "x-ms-region",
1760
+ wire_api: wireApi
1761
+ });
1762
+ }
1763
+ var DEFAULT_ANTHROPIC_MODEL = "claude-sonnet-4-6";
1764
+ var DEFAULT_OPENROUTER_MODEL = "anthropic/claude-sonnet-4.6";
1765
+ var FREE_OPENROUTER_MODEL = "nvidia/nemotron-3-super-120b-a12b:free";
1766
+ var DEFAULT_OPENAI_MODEL = "gpt-4o";
1767
+ var DEEPSEEK_DEFAULT_BASE_URL = "https://api.deepseek.com";
1768
+ var DEEPSEEK_DEFAULT_MODEL = "deepseek-flash";
1769
+ var QWEN_TOKEN_PLAN_DEEPSEEK_MODEL = "deepseek-v4-flash-0731";
1770
+ var XAI_DEFAULT_BASE_URL = "https://api.x.ai/v1";
1771
+ var XAI_DEFAULT_MODEL = "grok-4.6";
1772
+ var OPENCODE_DEFAULT_BASE_URL = "https://opencode.ai/zen/v1";
1773
+ var OPENCODE_DEFAULT_MODEL = "muse-spark-1.3-contributor-free";
1774
+ function openAICompatibleWireApi(env2, variable, fallback = "chat_completions") {
1775
+ const value = env2[variable];
1776
+ if (value === void 0)
1777
+ return fallback;
1778
+ if (value === "chat_completions" || value === "responses")
1779
+ return value;
1780
+ throw new Error(`${variable} must be "chat_completions" or "responses"`);
1781
+ }
1782
+ var googleFunctionCallSequence = 0;
1783
+ function opencodeModelId(model) {
1784
+ return model.replace(/^opencode\//i, "");
1785
+ }
1786
+ function opencodeWireApiForModel(model) {
1787
+ const bare = opencodeModelId(model ?? OPENCODE_DEFAULT_MODEL).toLowerCase();
1788
+ if (/^(muse-spark|gpt-|o[1-4](?:[-_]|$)|grok)/.test(bare))
1789
+ return "responses";
1790
+ if (/^(claude|qwen)/.test(bare))
1791
+ return "anthropic_messages";
1792
+ if (/^gemini/.test(bare))
1793
+ return "google_generate_content";
1794
+ if (/^(deepseek|mimo|ling|big-pickle|nemotron|minimax|glm|kimi|k3)/.test(bare)) {
1795
+ return "chat_completions";
1796
+ }
1797
+ throw new Error(`OpenCode Zen has no wire mapping for model "${bare}"`);
1798
+ }
1799
+ var COPILOT_API_BASE = "https://api.githubcopilot.com";
1800
+ var COPILOT_DEFAULT_MODEL = "gpt-4o";
1801
+ var COPILOT_STATIC_HEADERS = {
1802
+ "Copilot-Integration-Id": "vscode-chat",
1803
+ "Editor-Version": "vscode/1.99.3",
1804
+ "Editor-Plugin-Version": "copilot-chat/0.26.7",
1805
+ "X-GitHub-Api-Version": "2026-06-01",
1806
+ "Openai-Intent": "conversation-edits",
1807
+ "X-Initiator": "user"
1808
+ };
1809
+ function copilotModelId(model) {
1810
+ return model.replace(/^copilot\//i, "");
1811
+ }
1812
+ var GEMINI_OAUTH_CLIENT_ID = "681255809395-oo8ft2oprdrnp9e3aqf6av3hmdib135j.apps.googleusercontent.com";
1813
+ var GEMINI_OAUTH_CLIENT_SECRET = "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";
1814
+ var GEMINI_OAUTH_TOKEN_URL = "https://oauth2.googleapis.com/token";
1815
+ var CODE_ASSIST_ENDPOINT = "https://cloudcode-pa.googleapis.com";
1816
+ var CODE_ASSIST_API_VERSION = "v1internal";
1817
+ var GEMINI_DEFAULT_MODEL = "gemini-2.5-pro";
1818
+ var GEMINI_CLI_VERSION = "0.1.0";
1819
+ var AZURE_FOUNDRY_DEPLOYMENT_IDS = {
1820
+ "deepseek-v4-flash": true,
1821
+ "deepseek-v4-pro": true,
1822
+ "kimi-k2.7-code": true,
1823
+ "gpt-oss-120b": true,
1824
+ "gpt-5.4": true,
1825
+ "gpt-5.6-sol": true,
1826
+ "gpt-5.6-luna": true,
1827
+ "gpt-5.6-terra": true
1828
+ };
1829
+ function parseLlmFallbackChain(env2 = process.env) {
1830
+ const raw = env2["0SEC_LLM_FALLBACK"];
1831
+ if (!raw || raw.trim().length === 0)
1832
+ return [];
1833
+ const entries = [];
1834
+ const VALID_PROVIDERS = {
1835
+ openrouter: true,
1836
+ anthropic: true,
1837
+ openai: true,
1838
+ azure: true,
1839
+ deepseek: true,
1840
+ "chatgpt-codex": true,
1841
+ "z-ai": true,
1842
+ kimi: true,
1843
+ qwen: true,
1844
+ xai: true,
1845
+ opencode: true,
1846
+ copilot: true,
1847
+ google: true,
1848
+ hosted: true
1849
+ };
1850
+ for (const part of raw.split(",")) {
1851
+ const trimmed = part.trim();
1852
+ if (!trimmed)
1853
+ continue;
1854
+ const colonIdx = trimmed.indexOf(":");
1855
+ if (colonIdx < 1 || colonIdx === trimmed.length - 1) {
1856
+ diag.warn("fallback_chain_malformed_entry", `0SEC_LLM_FALLBACK: malformed entry "${trimmed}" (expected provider:model)`, { entry: trimmed, expected: "provider:model" });
1857
+ continue;
1858
+ }
1859
+ const provider = trimmed.slice(0, colonIdx);
1860
+ const model = trimmed.slice(colonIdx + 1).trim();
1861
+ if (!VALID_PROVIDERS[provider]) {
1862
+ diag.warn("fallback_chain_unknown_provider", `0SEC_LLM_FALLBACK: unknown provider "${provider}" in "${trimmed}"`, { entry: trimmed, provider });
1863
+ continue;
1864
+ }
1865
+ if (!model) {
1866
+ diag.warn("fallback_chain_empty_model", `0SEC_LLM_FALLBACK: empty model in "${trimmed}"`, { entry: trimmed, provider });
1867
+ continue;
1868
+ }
1869
+ entries.push({ provider, model });
1870
+ }
1871
+ return entries;
1872
+ }
1873
+ function resolveFailoverProvider(provider, model, env2 = process.env, apiKey) {
1874
+ switch (provider) {
1875
+ case "deepseek": {
1876
+ const key = apiKey ?? env2.DEEPSEEK_API_KEY;
1877
+ if (!key)
1878
+ return void 0;
1879
+ return { apiKey: key, baseUrl: env2.DEEPSEEK_BASE_URL ?? DEEPSEEK_DEFAULT_BASE_URL, wireApi: "responses" };
1880
+ }
1881
+ case "openrouter": {
1882
+ const key = apiKey ?? env2.OPENROUTER_API_KEY;
1883
+ if (!key)
1884
+ return void 0;
1885
+ return { apiKey: key, baseUrl: "https://openrouter.ai/api/v1", wireApi: openAICompatibleWireApi(env2, "OPENROUTER_WIRE_API") };
1886
+ }
1887
+ case "azure": {
1888
+ const key = apiKey ?? env2.AZURE_OPENAI_API_KEY;
1889
+ if (!key)
1890
+ return void 0;
1891
+ const url = env2.AZURE_OPENAI_BASE_URL ?? env2.OPENAI_BASE_URL;
1892
+ if (!url)
1893
+ return void 0;
1894
+ return { apiKey: key, baseUrl: url, wireApi: openAICompatibleWireApi(env2, "AZURE_OPENAI_WIRE_API") };
1895
+ }
1896
+ case "openai": {
1897
+ const key = apiKey ?? env2.OPENAI_API_KEY;
1898
+ if (!key)
1899
+ return void 0;
1900
+ return { apiKey: key, baseUrl: env2.OPENAI_BASE_URL ?? "https://api.openai.com/v1", wireApi: openAICompatibleWireApi(env2, "OPENAI_WIRE_API") };
1901
+ }
1902
+ case "anthropic": {
1903
+ const key = apiKey ?? env2.ANTHROPIC_API_KEY;
1904
+ if (!key)
1905
+ return void 0;
1906
+ return { apiKey: key, baseUrl: env2.ANTHROPIC_BASE_URL ?? "https://api.anthropic.com", wireApi: "chat_completions" };
1907
+ }
1908
+ case "chatgpt-codex": {
1909
+ if (!env2["0SEC_CHATGPT_ACCESS_TOKEN"] && !env2["0SEC_CHATGPT_OAUTH_REFRESH_TOKEN"] && !readChatGptCodexAuthFile(env2))
1910
+ return void 0;
1911
+ return { apiKey: "", baseUrl: CODEX_API_ENDPOINT, wireApi: "responses" };
1912
+ }
1913
+ case "z-ai": {
1914
+ const key = apiKey ?? env2.Z_AI_API_KEY;
1915
+ if (!key)
1916
+ return void 0;
1917
+ return { apiKey: key, baseUrl: env2.Z_AI_BASE_URL ?? ZAI_DEFAULT_BASE_URL, wireApi: "chat_completions" };
1918
+ }
1919
+ case "kimi": {
1920
+ const key = apiKey ?? env2.KIMI_API_KEY;
1921
+ if (!key)
1922
+ return void 0;
1923
+ return { apiKey: key, baseUrl: env2.KIMI_BASE_URL ?? KIMI_DEFAULT_BASE_URL, wireApi: "chat_completions" };
1924
+ }
1925
+ case "qwen": {
1926
+ const key = apiKey ?? env2.QWEN_API_KEY;
1927
+ if (!key)
1928
+ return void 0;
1929
+ return { apiKey: key, baseUrl: env2.QWEN_BASE_URL ?? QWEN_DEFAULT_BASE_URL, wireApi: "chat_completions" };
1930
+ }
1931
+ case "xai": {
1932
+ const key = apiKey ?? env2.XAI_API_KEY;
1933
+ if (!key)
1934
+ return void 0;
1935
+ return { apiKey: key, baseUrl: env2.XAI_BASE_URL ?? XAI_DEFAULT_BASE_URL, wireApi: openAICompatibleWireApi(env2, "XAI_WIRE_API") };
1936
+ }
1937
+ case "opencode": {
1938
+ const key = apiKey ?? env2.OPENCODE_API_KEY;
1939
+ if (!key)
1940
+ return void 0;
1941
+ return { apiKey: key, baseUrl: env2.OPENCODE_BASE_URL ?? OPENCODE_DEFAULT_BASE_URL, wireApi: opencodeWireApiForModel(model) };
1942
+ }
1943
+ case "copilot": {
1944
+ const key = apiKey ?? env2["0SEC_COPILOT_GITHUB_TOKEN"];
1945
+ if (!key)
1946
+ return void 0;
1947
+ return { apiKey: key, baseUrl: env2.COPILOT_BASE_URL ?? COPILOT_API_BASE, wireApi: "chat_completions" };
1948
+ }
1949
+ case "google": {
1950
+ if (!env2["0SEC_GEMINI_ACCESS_TOKEN"] && !env2["0SEC_GEMINI_OAUTH_REFRESH_TOKEN"])
1951
+ return void 0;
1952
+ return { apiKey: "", baseUrl: CODE_ASSIST_ENDPOINT, wireApi: "google_generate_content" };
1953
+ }
1954
+ case "hosted": {
1955
+ try {
1956
+ const creds = loadCloudCredentials({
1957
+ env: env2,
1958
+ warn: () => {
1959
+ }
1960
+ });
1961
+ return {
1962
+ apiKey: creds.token,
1963
+ baseUrl: `${creds.host}/api/inference/v1`,
1964
+ wireApi: "chat_completions"
1965
+ };
1966
+ } catch (err) {
1967
+ if (err instanceof CloudAuthMissingError)
1968
+ return void 0;
1969
+ throw err;
1970
+ }
1971
+ }
1972
+ }
1973
+ }
1974
+ var fallbackChainCache;
1975
+ function getFallbackChain(env2) {
1976
+ const raw = env2["0SEC_LLM_FALLBACK"];
1977
+ if (!fallbackChainCache || fallbackChainCache.raw !== raw) {
1978
+ fallbackChainCache = { raw, entries: parseLlmFallbackChain(env2) };
1979
+ }
1980
+ return fallbackChainCache.entries;
1981
+ }
1982
+ var ZAI_DEFAULT_BASE_URL = "https://api.z.ai/api/anthropic";
1983
+ var ZAI_DEFAULT_MODEL = "glm-5.3";
1984
+ var ZAI_DEFAULT_THINKING_BUDGET = 2048;
1985
+ function zaiThinkingBudget() {
1986
+ const raw = process.env["0SEC_ZAI_THINKING_BUDGET"];
1987
+ if (raw == null || raw.trim().length === 0)
1988
+ return ZAI_DEFAULT_THINKING_BUDGET;
1989
+ const n = Number.parseInt(raw, 10);
1990
+ return Number.isFinite(n) && n >= 0 ? n : ZAI_DEFAULT_THINKING_BUDGET;
1991
+ }
1992
+ var KIMI_DEFAULT_BASE_URL = "https://api.kimi.com/coding/v1";
1993
+ var KIMI_DEFAULT_MODEL = "k3";
1994
+ var QWEN_DEFAULT_BASE_URL = "https://token-plan.ap-southeast-1.maas.aliyuncs.com/compatible-mode/v1";
1995
+ var QWEN_DEFAULT_MODEL = "qwen3.8-max";
1996
+ var CODEX_API_ENDPOINT = "https://chatgpt.com/backend-api/codex/responses";
1997
+ var CODEX_OAUTH_ISSUER = "https://auth.openai.com";
1998
+ var CODEX_OAUTH_CLIENT_ID = "app_EMoamEEZ73f0CkXaXp7hrann";
1999
+ var CODEX_DEFAULT_MODEL = "gpt-5.5";
2000
+ var LOOP_SERVER_COMPACTION_TOKENS = 15e4;
2001
+ var PROCESS_SESSION_ID = `0sec-${Math.random().toString(36).slice(2, 10)}-${Date.now().toString(36)}`;
2002
+ var chatGptCodexAuthStates = /* @__PURE__ */ new Map();
2003
+ function codexAuthStateKey(state) {
2004
+ return JSON.stringify([state.authFilePath, state.accountId, state.refreshToken || state.accessToken]);
2005
+ }
2006
+ function readChatGptCodexEnv(env2 = process.env) {
2007
+ const access = env2["0SEC_CHATGPT_ACCESS_TOKEN"];
2008
+ const refresh = env2["0SEC_CHATGPT_OAUTH_REFRESH_TOKEN"];
2009
+ if ((!access || access.length === 0) && (!refresh || refresh.length === 0)) {
2010
+ return void 0;
2011
+ }
2012
+ const accountId = env2["0SEC_CHATGPT_ACCOUNT_ID"];
2013
+ return {
2014
+ accessToken: access && access.length > 0 ? access : void 0,
2015
+ refreshToken: refresh && refresh.length > 0 ? refresh : void 0,
2016
+ accountId
2017
+ };
2018
+ }
2019
+ function resolveChatGptCodexAuthPath(env2 = process.env) {
2020
+ return env2["0SEC_CHATGPT_AUTH_FILE"] ?? join2(env2.HOME ?? homedir(), ".codex", "auth.json");
2021
+ }
2022
+ function persistChatGptCodexAuthFile(authPath, tokens, usedRefreshToken) {
2023
+ try {
2024
+ let existing = {};
2025
+ try {
2026
+ existing = JSON.parse(readFileSync2(authPath, "utf8"));
2027
+ } catch {
2028
+ existing = {};
2029
+ }
2030
+ const prevTokens = existing.tokens ?? {};
2031
+ if (prevTokens.refresh_token !== usedRefreshToken)
2032
+ return;
2033
+ const nextTokens = {
2034
+ ...prevTokens,
2035
+ access_token: tokens.access_token,
2036
+ ...tokens.refresh_token ? { refresh_token: tokens.refresh_token } : {},
2037
+ ...tokens.id_token ? { id_token: tokens.id_token } : {}
2038
+ };
2039
+ const merged = {
2040
+ ...existing,
2041
+ tokens: nextTokens,
2042
+ last_refresh: (/* @__PURE__ */ new Date()).toISOString()
2043
+ };
2044
+ const tmp = `${authPath}.tmp-${process.pid}`;
2045
+ writeFileSync(tmp, `${JSON.stringify(merged, null, 2)}
2046
+ `, { mode: 384 });
2047
+ renameSync(tmp, authPath);
2048
+ } catch (err) {
2049
+ process.stderr.write(`[0sec] warning: could not persist rotated Codex refresh token to ${authPath}: ${err instanceof Error ? err.message : String(err)}
2050
+ `);
2051
+ }
2052
+ }
2053
+ function readChatGptCodexAuthFile(env2 = process.env) {
2054
+ const authPath = resolveChatGptCodexAuthPath(env2);
2055
+ if (!existsSync(authPath))
2056
+ return void 0;
2057
+ try {
2058
+ const auth = JSON.parse(readFileSync2(authPath, "utf8"));
2059
+ const tokens = auth.tokens;
2060
+ if (!tokens)
2061
+ return void 0;
2062
+ const accessToken = typeof tokens.access_token === "string" && tokens.access_token.length > 0 ? tokens.access_token : void 0;
2063
+ const refreshToken = typeof tokens.refresh_token === "string" && tokens.refresh_token.length > 0 ? tokens.refresh_token : void 0;
2064
+ if (!accessToken && !refreshToken)
2065
+ return void 0;
2066
+ return {
2067
+ ...accessToken ? { accessToken } : {},
2068
+ ...refreshToken ? { refreshToken } : {},
2069
+ ...typeof tokens.account_id === "string" && tokens.account_id.length > 0 ? { accountId: tokens.account_id } : {}
2070
+ };
2071
+ } catch {
2072
+ return void 0;
2073
+ }
2074
+ }
2075
+ function accessTokenExpiryMs(accessToken) {
2076
+ const claims = parseJwtPayload(accessToken);
2077
+ const exp = claims?.exp;
2078
+ if (typeof exp === "number" && Number.isFinite(exp)) {
2079
+ return exp * 1e3;
2080
+ }
2081
+ return Date.now() + 36e5;
2082
+ }
2083
+ async function refreshChatGptCodexAccessToken(refreshToken) {
2084
+ const res = await fetch(`${CODEX_OAUTH_ISSUER}/oauth/token`, {
2085
+ method: "POST",
2086
+ headers: { "Content-Type": "application/x-www-form-urlencoded" },
2087
+ body: new URLSearchParams({
2088
+ grant_type: "refresh_token",
2089
+ refresh_token: refreshToken,
2090
+ client_id: CODEX_OAUTH_CLIENT_ID
2091
+ }).toString()
2092
+ });
2093
+ if (!res.ok) {
2094
+ const body = await res.text().catch(() => "");
2095
+ throw new Error(`ChatGPT Codex token refresh failed: ${res.status} ${body.slice(0, 200)}`);
2096
+ }
2097
+ return await res.json();
2098
+ }
2099
+ function parseJwtPayload(token) {
2100
+ const parts = token.split(".");
2101
+ if (parts.length !== 3)
2102
+ return void 0;
2103
+ try {
2104
+ return JSON.parse(Buffer.from(parts[1], "base64url").toString("utf8"));
2105
+ } catch {
2106
+ return void 0;
2107
+ }
2108
+ }
2109
+ function extractChatGptAccountId(tokens) {
2110
+ const checkClaims = (claims) => {
2111
+ if (typeof claims.chatgpt_account_id === "string")
2112
+ return claims.chatgpt_account_id;
2113
+ const authClaim = claims["https://api.openai.com/auth"] ?? {};
2114
+ if (typeof authClaim.chatgpt_account_id === "string")
2115
+ return authClaim.chatgpt_account_id;
2116
+ const orgs = claims.organizations;
2117
+ if (Array.isArray(orgs) && orgs.length > 0 && orgs[0] && typeof orgs[0] === "object") {
2118
+ const id = orgs[0].id;
2119
+ if (typeof id === "string")
2120
+ return id;
2121
+ }
2122
+ return void 0;
2123
+ };
2124
+ for (const tok of [tokens.id_token, tokens.access_token]) {
2125
+ if (!tok)
2126
+ continue;
2127
+ const claims = parseJwtPayload(tok);
2128
+ if (claims) {
2129
+ const id = checkClaims(claims);
2130
+ if (id)
2131
+ return id;
2132
+ }
2133
+ }
2134
+ return void 0;
2135
+ }
2136
+ function resolveChatGptCodexAuthState(env2) {
2137
+ const fromEnvOnly = readChatGptCodexEnv(env2);
2138
+ const fromFile = fromEnvOnly ? void 0 : readChatGptCodexAuthFile(env2);
2139
+ const tokens = fromEnvOnly ?? fromFile;
2140
+ if (!tokens) {
2141
+ throw new Error("ChatGPT Codex auth: neither 0SEC_CHATGPT_ACCESS_TOKEN nor 0SEC_CHATGPT_OAUTH_REFRESH_TOKEN is set. Run `codex login` and either forward the access token via worker-controller (preferred for multi-sandbox dispatch \u2014 avoids the OAuth refresh-token rotation race) or keep a valid ~/.codex/auth.json on this host.");
2142
+ }
2143
+ const identity = {
2144
+ refreshToken: tokens.refreshToken ?? "",
2145
+ accessToken: tokens.accessToken,
2146
+ accountId: tokens.accountId ?? (tokens.accessToken ? extractChatGptAccountId({ access_token: tokens.accessToken }) : void 0),
2147
+ ...fromFile ? { authFilePath: resolveChatGptCodexAuthPath(env2) } : {}
2148
+ };
2149
+ const key = codexAuthStateKey(identity);
2150
+ const existing = chatGptCodexAuthStates.get(key);
2151
+ if (existing)
2152
+ return existing;
2153
+ const state = {
2154
+ ...identity,
2155
+ accessTokenExpiresAt: tokens.accessToken ? accessTokenExpiryMs(tokens.accessToken) : 0
2156
+ };
2157
+ chatGptCodexAuthStates.set(key, state);
2158
+ return state;
2159
+ }
2160
+ async function refreshChatGptCodexAuthState(state) {
2161
+ const now = Date.now();
2162
+ const needsRefresh = !state.accessToken || state.accessTokenExpiresAt - 6e4 <= now;
2163
+ if (needsRefresh && !state.refreshToken) {
2164
+ throw new Error("ChatGPT Codex access token expired and no refresh token is available. The worker-controller should forward a fresh access token at sandbox dispatch.");
2165
+ }
2166
+ if (needsRefresh) {
2167
+ if (!state.inflightRefresh) {
2168
+ const usedRefresh = state.refreshToken;
2169
+ state.inflightRefresh = (async () => {
2170
+ try {
2171
+ const tokens = await refreshChatGptCodexAccessToken(usedRefresh);
2172
+ state.accessToken = tokens.access_token;
2173
+ state.accessTokenExpiresAt = Date.now() + (tokens.expires_in ?? 3600) * 1e3;
2174
+ if (tokens.refresh_token) {
2175
+ state.refreshToken = tokens.refresh_token;
2176
+ if (state.authFilePath) {
2177
+ persistChatGptCodexAuthFile(state.authFilePath, tokens, usedRefresh);
2178
+ }
2179
+ }
2180
+ if (!state.accountId) {
2181
+ state.accountId = extractChatGptAccountId(tokens);
2182
+ }
2183
+ chatGptCodexAuthStates.set(codexAuthStateKey(state), state);
2184
+ } finally {
2185
+ state.inflightRefresh = void 0;
2186
+ }
2187
+ })();
2188
+ }
2189
+ await state.inflightRefresh;
2190
+ }
2191
+ if (!state.accessToken) {
2192
+ throw new Error("ChatGPT Codex auth: access token still unset after refresh \u2014 refresh must have failed.");
2193
+ }
2194
+ return { accessToken: state.accessToken, accountId: state.accountId };
2195
+ }
2196
+ var geminiCodeAssistAuthStates = /* @__PURE__ */ new Map();
2197
+ function geminiAuthStateKey(refreshToken, accessToken) {
2198
+ return JSON.stringify([refreshToken, refreshToken ? void 0 : accessToken]);
2199
+ }
2200
+ function readGeminiCodeAssistEnv(env2 = process.env) {
2201
+ const access = env2["0SEC_GEMINI_ACCESS_TOKEN"];
2202
+ const refresh = env2["0SEC_GEMINI_OAUTH_REFRESH_TOKEN"];
2203
+ if ((!access || access.length === 0) && (!refresh || refresh.length === 0))
2204
+ return void 0;
2205
+ return {
2206
+ accessToken: access && access.length > 0 ? access : void 0,
2207
+ refreshToken: refresh && refresh.length > 0 ? refresh : void 0
2208
+ };
2209
+ }
2210
+ function resolveGeminiCodeAssistAuthState(env2) {
2211
+ const tokens = readGeminiCodeAssistEnv(env2);
2212
+ if (!tokens) {
2213
+ throw new Error("Google Gemini Code Assist auth: neither 0SEC_GEMINI_ACCESS_TOKEN nor 0SEC_GEMINI_OAUTH_REFRESH_TOKEN is set. Sign in with your Google account (0sec connect) or forward a fresh access token.");
2214
+ }
2215
+ const key = geminiAuthStateKey(tokens.refreshToken ?? "", tokens.accessToken);
2216
+ const existing = geminiCodeAssistAuthStates.get(key);
2217
+ if (existing)
2218
+ return existing;
2219
+ const state = {
2220
+ refreshToken: tokens.refreshToken ?? "",
2221
+ accessToken: tokens.accessToken,
2222
+ // A forwarded access token with no `exp` we can read: treat as immediately
2223
+ // stale so the first call refreshes (when a refresh token is available).
2224
+ accessTokenExpiresAt: tokens.accessToken && tokens.refreshToken ? 0 : tokens.accessToken ? Date.now() + 36e5 : 0
2225
+ };
2226
+ geminiCodeAssistAuthStates.set(key, state);
2227
+ return state;
2228
+ }
2229
+ async function refreshGoogleAccessToken(refreshToken) {
2230
+ const res = await fetch(GEMINI_OAUTH_TOKEN_URL, {
2231
+ method: "POST",
2232
+ headers: { "Content-Type": "application/x-www-form-urlencoded" },
2233
+ body: new URLSearchParams({
2234
+ grant_type: "refresh_token",
2235
+ refresh_token: refreshToken,
2236
+ client_id: GEMINI_OAUTH_CLIENT_ID,
2237
+ client_secret: GEMINI_OAUTH_CLIENT_SECRET
2238
+ }).toString()
2239
+ });
2240
+ const bodyText = await res.text().catch(() => "");
2241
+ let parsed = {};
2242
+ try {
2243
+ parsed = bodyText ? JSON.parse(bodyText) : {};
2244
+ } catch {
2245
+ parsed = {};
2246
+ }
2247
+ if (!res.ok) {
2248
+ const err = new Error(`Google token refresh failed: ${res.status} ${bodyText.slice(0, 200)}`);
2249
+ err.invalidGrant = parsed.error === "invalid_grant";
2250
+ throw err;
2251
+ }
2252
+ return parsed;
2253
+ }
2254
+ async function refreshGeminiCodeAssistAuthState(state) {
2255
+ const now = Date.now();
2256
+ const needsRefresh = !state.accessToken || state.accessTokenExpiresAt - 3e5 <= now;
2257
+ if (needsRefresh && !state.refreshToken) {
2258
+ if (state.accessToken)
2259
+ return state.accessToken;
2260
+ throw new Error("Google Gemini Code Assist auth: no access or refresh token available.");
2261
+ }
2262
+ if (needsRefresh) {
2263
+ if (!state.inflightRefresh) {
2264
+ const usedRefresh = state.refreshToken;
2265
+ state.inflightRefresh = (async () => {
2266
+ try {
2267
+ const tokens = await refreshGoogleAccessToken(usedRefresh);
2268
+ if (!tokens.access_token)
2269
+ throw new Error("Google token refresh returned no access_token.");
2270
+ state.accessToken = tokens.access_token;
2271
+ state.accessTokenExpiresAt = Date.now() + (tokens.expires_in ?? 3600) * 1e3;
2272
+ if (tokens.refresh_token)
2273
+ state.refreshToken = tokens.refresh_token;
2274
+ } catch (err) {
2275
+ if (err.invalidGrant) {
2276
+ state.accessToken = void 0;
2277
+ state.accessTokenExpiresAt = 0;
2278
+ geminiCodeAssistAuthStates.delete(geminiAuthStateKey(state.refreshToken, state.accessToken));
2279
+ }
2280
+ throw err;
2281
+ } finally {
2282
+ state.inflightRefresh = void 0;
2283
+ }
2284
+ })();
2285
+ }
2286
+ await state.inflightRefresh;
2287
+ }
2288
+ if (!state.accessToken) {
2289
+ throw new Error("Google Gemini Code Assist auth: access token unset after refresh.");
2290
+ }
2291
+ return state.accessToken;
2292
+ }
2293
+ var GEMINI_CODE_ASSIST_METADATA = {
2294
+ ideType: "IDE_UNSPECIFIED",
2295
+ platform: "PLATFORM_UNSPECIFIED",
2296
+ pluginType: "GEMINI"
2297
+ };
2298
+ async function codeAssistPost(method, accessToken, body) {
2299
+ const res = await fetch(`${CODE_ASSIST_ENDPOINT}/${CODE_ASSIST_API_VERSION}:${method}`, {
2300
+ method: "POST",
2301
+ headers: {
2302
+ "Content-Type": "application/json",
2303
+ Authorization: `Bearer ${accessToken}`
2304
+ },
2305
+ body: JSON.stringify(body)
2306
+ });
2307
+ const text = await res.text().catch(() => "");
2308
+ if (!res.ok) {
2309
+ const err = new Error(`Code Assist ${method} failed: ${res.status} ${text.slice(0, 200)}`);
2310
+ err.securityPolicyViolated = text.includes("SECURITY_POLICY_VIOLATED");
2311
+ throw err;
2312
+ }
2313
+ try {
2314
+ return text ? JSON.parse(text) : {};
2315
+ } catch {
2316
+ return {};
2317
+ }
2318
+ }
2319
+ async function resolveGeminiCodeAssistProject(state, env2, sleep = (ms) => new Promise((r) => setTimeout(r, ms))) {
2320
+ if (state.projectId !== void 0)
2321
+ return state.projectId;
2322
+ if (state.inflightProjectResolve)
2323
+ return state.inflightProjectResolve;
2324
+ state.inflightProjectResolve = (async () => {
2325
+ try {
2326
+ const override = firstNonEmptyEnv(env2, "GOOGLE_CLOUD_PROJECT", "0SEC_GEMINI_PROJECT");
2327
+ const accessToken = await refreshGeminiCodeAssistAuthState(state);
2328
+ let load;
2329
+ try {
2330
+ load = await codeAssistPost("loadCodeAssist", accessToken, {
2331
+ ...override ? { cloudaicompanionProject: override } : {},
2332
+ metadata: GEMINI_CODE_ASSIST_METADATA
2333
+ });
2334
+ } catch (err) {
2335
+ if (err.securityPolicyViolated) {
2336
+ if (override)
2337
+ return override;
2338
+ throw new Error("Google Gemini Code Assist: this account is behind a VPC Service Controls perimeter \u2014 set GOOGLE_CLOUD_PROJECT (or 0SEC_GEMINI_PROJECT).");
2339
+ }
2340
+ throw err;
2341
+ }
2342
+ const bound = load.cloudaicompanionProject;
2343
+ if (typeof bound === "string" && bound.length > 0)
2344
+ return bound;
2345
+ const tiers = Array.isArray(load.allowedTiers) ? load.allowedTiers : [];
2346
+ const defaultTier = tiers.find((tier) => tier.isDefault === true);
2347
+ const currentTier = load.currentTier;
2348
+ const tierId = currentTier?.id ?? defaultTier?.id ?? "free-tier";
2349
+ const tierForProject = currentTier ?? defaultTier;
2350
+ const isFreeTier = tierForProject?.userDefinedCloudaicompanionProject === false || tierId === "free-tier";
2351
+ const onboardProject = isFreeTier ? void 0 : override;
2352
+ let op = await codeAssistPost("onboardUser", accessToken, {
2353
+ tierId,
2354
+ ...onboardProject ? { cloudaicompanionProject: onboardProject } : {},
2355
+ metadata: GEMINI_CODE_ASSIST_METADATA
2356
+ });
2357
+ let guard = 0;
2358
+ while (op.done !== true && guard < 60) {
2359
+ const opName = typeof op.name === "string" ? op.name : void 0;
2360
+ if (!opName)
2361
+ break;
2362
+ await sleep(5e3);
2363
+ const opRes = await fetch(`${CODE_ASSIST_ENDPOINT}/${CODE_ASSIST_API_VERSION}/${opName}`, {
2364
+ headers: { Authorization: `Bearer ${await refreshGeminiCodeAssistAuthState(state)}` }
2365
+ });
2366
+ const opText = await opRes.text().catch(() => "");
2367
+ if (!opRes.ok)
2368
+ throw new Error(`Code Assist onboard poll failed: ${opRes.status} ${opText.slice(0, 200)}`);
2369
+ op = opText ? JSON.parse(opText) : {};
2370
+ guard += 1;
2371
+ }
2372
+ const response = op.response;
2373
+ const project = response?.cloudaicompanionProject;
2374
+ const resolved2 = typeof project === "string" ? project : typeof project?.id === "string" ? project.id : void 0;
2375
+ return resolved2 && resolved2.length > 0 ? resolved2 : override ?? "";
2376
+ } finally {
2377
+ state.inflightProjectResolve = void 0;
2378
+ }
2379
+ })();
2380
+ const resolved = await state.inflightProjectResolve;
2381
+ state.projectId = resolved;
2382
+ return resolved;
2383
+ }
2384
+ function firstNonEmptyEnv(env2, ...names) {
2385
+ for (const name of names) {
2386
+ const value = env2[name];
2387
+ if (typeof value === "string" && value.trim().length > 0)
2388
+ return value.trim();
2389
+ }
2390
+ return void 0;
2391
+ }
2392
+ function geminiUserAgent(model) {
2393
+ return `GeminiCLI/${GEMINI_CLI_VERSION} (${process.platform}; ${process.arch}) ${model}`;
2394
+ }
2395
+ function parseCodexAzureConfig(env2 = process.env) {
2396
+ const configPath = `${env2.HOME ?? ""}/.codex/config.toml`;
2397
+ if (!existsSync(configPath))
2398
+ return {};
2399
+ try {
2400
+ const content = readFileSync2(configPath, "utf8");
2401
+ const azureSectionMatch = content.match(/\[model_providers\.azure\]([\s\S]*?)(?:\n\[|$)/);
2402
+ const activeProviderMatch = content.match(/^\s*model_provider\s*=\s*"([^"]+)"/m);
2403
+ const baseUrlMatch = azureSectionMatch?.[1]?.match(/base_url\s*=\s*"([^"]+)"/);
2404
+ const wireApiMatch = azureSectionMatch?.[1]?.match(/wire_api\s*=\s*"([^"]+)"/);
2405
+ const azureModelMatch = azureSectionMatch?.[1]?.match(/model\s*=\s*"([^"]+)"/);
2406
+ const topLevelModelMatch = content.match(/^\s*model\s*=\s*"([^"]+)"/m);
2407
+ const topLevelSection = content.split(/^\[/m)[0] ?? "";
2408
+ const reasoningMatch = azureSectionMatch?.[1]?.match(/model_reasoning_effort\s*=\s*"([^"]+)"/) ?? topLevelSection.match(/^\s*model_reasoning_effort\s*=\s*"([^"]+)"/m);
2409
+ return {
2410
+ baseUrl: baseUrlMatch?.[1],
2411
+ model: azureModelMatch?.[1] ?? (activeProviderMatch?.[1] === "azure" ? topLevelModelMatch?.[1] : void 0),
2412
+ wireApi: wireApiMatch?.[1] === "responses" ? "responses" : "chat_completions",
2413
+ reasoningEffort: reasoningMatch?.[1]
2414
+ };
2415
+ } catch {
2416
+ return {};
2417
+ }
2418
+ }
2419
+ function providerForModel(model, env2) {
2420
+ if (!model)
2421
+ return void 0;
2422
+ const m = model.toLowerCase();
2423
+ if (model === DEEPSEEK_DEFAULT_MODEL || model === "deepseek-v4-flash") {
2424
+ return env2.DEEPSEEK_API_KEY ? "deepseek" : void 0;
2425
+ }
2426
+ if (AZURE_FOUNDRY_DEPLOYMENT_IDS[m]) {
2427
+ return env2.AZURE_OPENAI_API_KEY ? "azure" : void 0;
2428
+ }
2429
+ if (m === QWEN_TOKEN_PLAN_DEEPSEEK_MODEL) {
2430
+ return env2.QWEN_API_KEY ? "qwen" : void 0;
2431
+ }
2432
+ if (m.startsWith("openrouter/"))
2433
+ return env2.OPENROUTER_API_KEY ? "openrouter" : void 0;
2434
+ if (m.startsWith("glm-") || m.startsWith("z-ai/") || m.includes("glm")) {
2435
+ return env2.Z_AI_API_KEY ? "z-ai" : void 0;
2436
+ }
2437
+ if (m.startsWith("k3") || m.startsWith("kimi")) {
2438
+ return env2.KIMI_API_KEY ? "kimi" : void 0;
2439
+ }
2440
+ if (m.startsWith("qwen")) {
2441
+ return env2.QWEN_API_KEY ? "qwen" : void 0;
2442
+ }
2443
+ if (m.startsWith("grok") || m.startsWith("xai/") || m.startsWith("x-ai/")) {
2444
+ return env2.XAI_API_KEY ? "xai" : void 0;
2445
+ }
2446
+ if (/^(opencode\/|muse-spark|mimo|ling|big-pickle|nemotron|minimax)/.test(m)) {
2447
+ return env2.OPENCODE_API_KEY ? "opencode" : void 0;
2448
+ }
2449
+ if (m.startsWith("copilot/")) {
2450
+ return env2["0SEC_COPILOT_GITHUB_TOKEN"] ? "copilot" : void 0;
2451
+ }
2452
+ if (m.startsWith("gemini") || m.startsWith("google/")) {
2453
+ return env2["0SEC_GEMINI_ACCESS_TOKEN"] || env2["0SEC_GEMINI_OAUTH_REFRESH_TOKEN"] ? "google" : void 0;
2454
+ }
2455
+ if (/^gpt-|^o[1-4](?:[-_]|$)/.test(m)) {
2456
+ if (env2["0SEC_CHATGPT_ACCESS_TOKEN"] || env2["0SEC_CHATGPT_OAUTH_REFRESH_TOKEN"])
2457
+ return "chatgpt-codex";
2458
+ if (env2.OPENAI_API_KEY)
2459
+ return "openai";
2460
+ return void 0;
2461
+ }
2462
+ if (m.startsWith("claude") || m.startsWith("anthropic/") || m.includes("sonnet") || m.includes("opus") || m.includes("haiku")) {
2463
+ if (env2.ANTHROPIC_API_KEY)
2464
+ return "anthropic";
2465
+ if (env2.OPENROUTER_API_KEY)
2466
+ return "openrouter";
2467
+ return void 0;
2468
+ }
2469
+ return void 0;
2470
+ }
2471
+ var DEFAULT_PROVIDER_MODELS = {
2472
+ openrouter: DEFAULT_OPENROUTER_MODEL,
2473
+ anthropic: DEFAULT_ANTHROPIC_MODEL,
2474
+ openai: DEFAULT_OPENAI_MODEL,
2475
+ azure: void 0,
2476
+ deepseek: DEEPSEEK_DEFAULT_MODEL,
2477
+ "chatgpt-codex": CODEX_DEFAULT_MODEL,
2478
+ "z-ai": ZAI_DEFAULT_MODEL,
2479
+ kimi: KIMI_DEFAULT_MODEL,
2480
+ qwen: QWEN_DEFAULT_MODEL,
2481
+ xai: XAI_DEFAULT_MODEL,
2482
+ opencode: OPENCODE_DEFAULT_MODEL,
2483
+ copilot: COPILOT_DEFAULT_MODEL,
2484
+ google: GEMINI_DEFAULT_MODEL,
2485
+ hosted: ""
2486
+ };
2487
+ var AUTO_MODEL_SENTINEL = "auto";
2488
+ function detectProvider(configApiKey, preferredModel, env2, configProvider) {
2489
+ if (configProvider !== void 0 && !Object.hasOwn(DEFAULT_PROVIDER_MODELS, configProvider)) {
2490
+ throw new Error(`RuntimeConfig.provider is unsupported: ${configProvider}`);
2491
+ }
2492
+ const selectedProviderRaw = configProvider ?? env2["0SEC_SELECTED_PROVIDER"]?.trim();
2493
+ const forcedProviderRaw = env2["0SEC_FORCE_PROVIDER"]?.trim() || void 0;
2494
+ if (selectedProviderRaw && forcedProviderRaw && selectedProviderRaw !== forcedProviderRaw) {
2495
+ throw new Error(`${configProvider !== void 0 ? "RuntimeConfig.provider" : "0SEC_SELECTED_PROVIDER"} conflicts with 0SEC_FORCE_PROVIDER`);
2496
+ }
2497
+ const primaryModel = env2["0SEC_MODEL"]?.trim();
2498
+ const selectedProviderApplies = configProvider !== void 0 || !preferredModel || !primaryModel || preferredModel === primaryModel;
2499
+ const pinnedProviderRaw = forcedProviderRaw ?? (selectedProviderApplies ? selectedProviderRaw : void 0);
2500
+ if (pinnedProviderRaw) {
2501
+ const source = pinnedProviderRaw === forcedProviderRaw ? "0SEC_FORCE_PROVIDER" : configProvider !== void 0 ? "RuntimeConfig.provider" : "0SEC_SELECTED_PROVIDER";
2502
+ if (!Object.hasOwn(DEFAULT_PROVIDER_MODELS, pinnedProviderRaw)) {
2503
+ throw new Error(`${source} is unsupported: ${pinnedProviderRaw}`);
2504
+ }
2505
+ const provider = pinnedProviderRaw;
2506
+ const model = preferredModel ?? env2["0SEC_MODEL"] ?? (configProvider !== void 0 || provider === "hosted" ? DEFAULT_PROVIDER_MODELS[provider] : void 0);
2507
+ if (model === void 0 || model === "" && provider !== "hosted") {
2508
+ throw new Error(`${source} requires an explicit model`);
2509
+ }
2510
+ if (configApiKey && (provider === "hosted" || provider === "chatgpt-codex")) {
2511
+ throw new Error(`${source}=${provider} requires its own authentication, not RuntimeConfig.apiKey`);
2512
+ }
2513
+ const resolved = resolveFailoverProvider(provider, model, env2, configApiKey);
2514
+ if (!resolved) {
2515
+ throw new Error(`${source}=${provider} has no configured credentials`);
2516
+ }
2517
+ return { provider, ...resolved, defaultModel: model };
2518
+ }
2519
+ if (configApiKey) {
2520
+ if (configApiKey.startsWith("sk-or-")) {
2521
+ return {
2522
+ provider: "openrouter",
2523
+ apiKey: configApiKey,
2524
+ baseUrl: "https://openrouter.ai/api/v1",
2525
+ defaultModel: DEFAULT_OPENROUTER_MODEL,
2526
+ wireApi: openAICompatibleWireApi(env2, "OPENROUTER_WIRE_API")
2527
+ };
2528
+ }
2529
+ if (configApiKey.startsWith("sk-ant-")) {
2530
+ return {
2531
+ provider: "anthropic",
2532
+ apiKey: configApiKey,
2533
+ baseUrl: "https://api.anthropic.com",
2534
+ defaultModel: DEFAULT_ANTHROPIC_MODEL,
2535
+ wireApi: "chat_completions"
2536
+ };
2537
+ }
2538
+ return {
2539
+ provider: "openai",
2540
+ apiKey: configApiKey,
2541
+ baseUrl: "https://api.openai.com/v1",
2542
+ defaultModel: DEFAULT_OPENAI_MODEL,
2543
+ wireApi: openAICompatibleWireApi(env2, "OPENAI_WIRE_API")
2544
+ };
2545
+ }
2546
+ switch (providerForModel(preferredModel, env2)) {
2547
+ case "deepseek":
2548
+ return {
2549
+ provider: "deepseek",
2550
+ apiKey: env2.DEEPSEEK_API_KEY,
2551
+ baseUrl: env2.DEEPSEEK_BASE_URL ?? DEEPSEEK_DEFAULT_BASE_URL,
2552
+ defaultModel: DEEPSEEK_DEFAULT_MODEL,
2553
+ wireApi: "responses"
2554
+ };
2555
+ case "azure": {
2556
+ const azureKey2 = env2.AZURE_OPENAI_API_KEY;
2557
+ if (!azureKey2)
2558
+ break;
2559
+ const azureConfig = parseCodexAzureConfig(env2);
2560
+ return {
2561
+ provider: "azure",
2562
+ apiKey: azureKey2,
2563
+ baseUrl: env2.AZURE_OPENAI_BASE_URL ?? env2.OPENAI_BASE_URL ?? azureConfig.baseUrl ?? "https://api.openai.com/v1",
2564
+ defaultModel: preferredModel ?? env2.AZURE_OPENAI_MODEL ?? azureConfig.model ?? DEFAULT_OPENAI_MODEL,
2565
+ wireApi: openAICompatibleWireApi(env2, "AZURE_OPENAI_WIRE_API", azureConfig.wireApi),
2566
+ reasoningEffort: azureConfig.reasoningEffort
2567
+ };
2568
+ }
2569
+ // z-ai (GLM) and kimi (Moonshot) ride the Anthropic Messages wire (routed by
2570
+ // LlmApiRuntime.isAnthropicWire — NOT by this `wireApi` field). The
2571
+ // "chat_completions" below is an inert default that is intentionally UNUSED
2572
+ // for these two providers; do NOT add them to isOpenAICompat.
2573
+ case "z-ai":
2574
+ return {
2575
+ provider: "z-ai",
2576
+ apiKey: env2.Z_AI_API_KEY,
2577
+ baseUrl: env2.Z_AI_BASE_URL ?? ZAI_DEFAULT_BASE_URL,
2578
+ defaultModel: ZAI_DEFAULT_MODEL,
2579
+ wireApi: "chat_completions"
2580
+ };
2581
+ case "kimi":
2582
+ return {
2583
+ provider: "kimi",
2584
+ apiKey: env2.KIMI_API_KEY,
2585
+ baseUrl: env2.KIMI_BASE_URL ?? KIMI_DEFAULT_BASE_URL,
2586
+ defaultModel: KIMI_DEFAULT_MODEL,
2587
+ wireApi: "chat_completions"
2588
+ };
2589
+ case "qwen":
2590
+ return {
2591
+ provider: "qwen",
2592
+ apiKey: env2.QWEN_API_KEY,
2593
+ baseUrl: env2.QWEN_BASE_URL ?? QWEN_DEFAULT_BASE_URL,
2594
+ defaultModel: QWEN_DEFAULT_MODEL,
2595
+ wireApi: "chat_completions"
2596
+ };
2597
+ case "xai":
2598
+ return {
2599
+ provider: "xai",
2600
+ apiKey: env2.XAI_API_KEY,
2601
+ baseUrl: env2.XAI_BASE_URL ?? XAI_DEFAULT_BASE_URL,
2602
+ defaultModel: XAI_DEFAULT_MODEL,
2603
+ wireApi: openAICompatibleWireApi(env2, "XAI_WIRE_API")
2604
+ };
2605
+ case "opencode":
2606
+ return {
2607
+ provider: "opencode",
2608
+ apiKey: env2.OPENCODE_API_KEY,
2609
+ baseUrl: env2.OPENCODE_BASE_URL ?? OPENCODE_DEFAULT_BASE_URL,
2610
+ defaultModel: OPENCODE_DEFAULT_MODEL,
2611
+ wireApi: opencodeWireApiForModel(preferredModel)
2612
+ };
2613
+ case "copilot":
2614
+ return {
2615
+ provider: "copilot",
2616
+ apiKey: env2["0SEC_COPILOT_GITHUB_TOKEN"],
2617
+ baseUrl: env2.COPILOT_BASE_URL ?? COPILOT_API_BASE,
2618
+ defaultModel: preferredModel ?? COPILOT_DEFAULT_MODEL,
2619
+ wireApi: "chat_completions"
2620
+ };
2621
+ case "google":
2622
+ return {
2623
+ provider: "google",
2624
+ apiKey: "",
2625
+ baseUrl: CODE_ASSIST_ENDPOINT,
2626
+ defaultModel: preferredModel ?? GEMINI_DEFAULT_MODEL,
2627
+ wireApi: "google_generate_content"
2628
+ };
2629
+ case "chatgpt-codex":
2630
+ return {
2631
+ provider: "chatgpt-codex",
2632
+ apiKey: "",
2633
+ baseUrl: CODEX_API_ENDPOINT,
2634
+ defaultModel: env2["0SEC_MODEL"] ?? CODEX_DEFAULT_MODEL,
2635
+ wireApi: "responses"
2636
+ };
2637
+ case "anthropic":
2638
+ return {
2639
+ provider: "anthropic",
2640
+ apiKey: env2.ANTHROPIC_API_KEY,
2641
+ baseUrl: env2.ANTHROPIC_BASE_URL ?? "https://api.anthropic.com",
2642
+ defaultModel: DEFAULT_ANTHROPIC_MODEL,
2643
+ wireApi: "chat_completions"
2644
+ };
2645
+ case "openrouter":
2646
+ return {
2647
+ provider: "openrouter",
2648
+ apiKey: env2.OPENROUTER_API_KEY,
2649
+ baseUrl: "https://openrouter.ai/api/v1",
2650
+ defaultModel: DEFAULT_OPENROUTER_MODEL,
2651
+ wireApi: openAICompatibleWireApi(env2, "OPENROUTER_WIRE_API")
2652
+ };
2653
+ case "openai":
2654
+ return {
2655
+ provider: "openai",
2656
+ apiKey: env2.OPENAI_API_KEY,
2657
+ baseUrl: env2.OPENAI_BASE_URL ?? "https://api.openai.com/v1",
2658
+ defaultModel: DEFAULT_OPENAI_MODEL,
2659
+ wireApi: openAICompatibleWireApi(env2, "OPENAI_WIRE_API")
2660
+ };
2661
+ default:
2662
+ break;
2663
+ }
2664
+ const chatGptAccess = env2["0SEC_CHATGPT_ACCESS_TOKEN"];
2665
+ const chatGptRefresh = env2["0SEC_CHATGPT_OAUTH_REFRESH_TOKEN"];
2666
+ const chatGptAuthFile = !chatGptAccess && !chatGptRefresh ? readChatGptCodexAuthFile(env2) : void 0;
2667
+ if (chatGptAccess && chatGptAccess.length > 0 || chatGptRefresh && chatGptRefresh.length > 0 || !!chatGptAuthFile) {
2668
+ return {
2669
+ provider: "chatgpt-codex",
2670
+ // No api key — auth flows via OAuth bearer that's refreshed on
2671
+ // demand by getChatGptCodexAccessToken(). Empty string keeps the
2672
+ // existing apiKey-required diagnostics from firing (those check
2673
+ // for empty strings; we want "valid but bearer-not-key").
2674
+ apiKey: "",
2675
+ // baseUrl is informational only — the runtime hardcodes
2676
+ // CODEX_API_ENDPOINT for this provider.
2677
+ baseUrl: CODEX_API_ENDPOINT,
2678
+ defaultModel: env2["0SEC_MODEL"] ?? CODEX_DEFAULT_MODEL,
2679
+ wireApi: "responses"
2680
+ };
2681
+ }
2682
+ const deepseekKey = env2.DEEPSEEK_API_KEY;
2683
+ if (deepseekKey) {
2684
+ return {
2685
+ provider: "deepseek",
2686
+ apiKey: deepseekKey,
2687
+ baseUrl: env2.DEEPSEEK_BASE_URL ?? DEEPSEEK_DEFAULT_BASE_URL,
2688
+ defaultModel: DEEPSEEK_DEFAULT_MODEL,
2689
+ wireApi: "responses"
2690
+ };
2691
+ }
2692
+ const openrouterKey = env2.OPENROUTER_API_KEY;
2693
+ if (openrouterKey) {
2694
+ return {
2695
+ provider: "openrouter",
2696
+ apiKey: openrouterKey,
2697
+ baseUrl: "https://openrouter.ai/api/v1",
2698
+ defaultModel: DEFAULT_OPENROUTER_MODEL,
2699
+ wireApi: openAICompatibleWireApi(env2, "OPENROUTER_WIRE_API")
2700
+ };
2701
+ }
2702
+ const azureKey = env2.AZURE_OPENAI_API_KEY;
2703
+ if (azureKey) {
2704
+ const azureConfig = parseCodexAzureConfig(env2);
2705
+ return {
2706
+ provider: "azure",
2707
+ apiKey: azureKey,
2708
+ baseUrl: env2.AZURE_OPENAI_BASE_URL ?? env2.OPENAI_BASE_URL ?? azureConfig.baseUrl ?? "https://api.openai.com/v1",
2709
+ defaultModel: env2.AZURE_OPENAI_MODEL ?? azureConfig.model ?? DEFAULT_OPENAI_MODEL,
2710
+ wireApi: openAICompatibleWireApi(env2, "AZURE_OPENAI_WIRE_API", azureConfig.wireApi),
2711
+ reasoningEffort: azureConfig.reasoningEffort
2712
+ };
2713
+ }
2714
+ const openaiKey = env2.OPENAI_API_KEY;
2715
+ if (openaiKey) {
2716
+ return {
2717
+ provider: "openai",
2718
+ apiKey: openaiKey,
2719
+ baseUrl: env2.OPENAI_BASE_URL ?? "https://api.openai.com/v1",
2720
+ defaultModel: DEFAULT_OPENAI_MODEL,
2721
+ wireApi: openAICompatibleWireApi(env2, "OPENAI_WIRE_API")
2722
+ };
2723
+ }
2724
+ const zaiKey = env2.Z_AI_API_KEY;
2725
+ if (zaiKey) {
2726
+ return {
2727
+ provider: "z-ai",
2728
+ apiKey: zaiKey,
2729
+ baseUrl: env2.Z_AI_BASE_URL ?? ZAI_DEFAULT_BASE_URL,
2730
+ defaultModel: ZAI_DEFAULT_MODEL,
2731
+ wireApi: "chat_completions"
2732
+ };
2733
+ }
2734
+ const kimiKey = env2.KIMI_API_KEY;
2735
+ if (kimiKey) {
2736
+ return {
2737
+ provider: "kimi",
2738
+ apiKey: kimiKey,
2739
+ baseUrl: env2.KIMI_BASE_URL ?? KIMI_DEFAULT_BASE_URL,
2740
+ defaultModel: KIMI_DEFAULT_MODEL,
2741
+ wireApi: "chat_completions"
2742
+ };
2743
+ }
2744
+ const qwenKey = env2.QWEN_API_KEY;
2745
+ if (qwenKey) {
2746
+ return {
2747
+ provider: "qwen",
2748
+ apiKey: qwenKey,
2749
+ baseUrl: env2.QWEN_BASE_URL ?? QWEN_DEFAULT_BASE_URL,
2750
+ defaultModel: QWEN_DEFAULT_MODEL,
2751
+ wireApi: "chat_completions"
2752
+ };
2753
+ }
2754
+ const xaiKey = env2.XAI_API_KEY;
2755
+ if (xaiKey) {
2756
+ return {
2757
+ provider: "xai",
2758
+ apiKey: xaiKey,
2759
+ baseUrl: env2.XAI_BASE_URL ?? XAI_DEFAULT_BASE_URL,
2760
+ defaultModel: XAI_DEFAULT_MODEL,
2761
+ wireApi: openAICompatibleWireApi(env2, "XAI_WIRE_API")
2762
+ };
2763
+ }
2764
+ const opencodeKey = env2.OPENCODE_API_KEY;
2765
+ if (opencodeKey) {
2766
+ return {
2767
+ provider: "opencode",
2768
+ apiKey: opencodeKey,
2769
+ baseUrl: env2.OPENCODE_BASE_URL ?? OPENCODE_DEFAULT_BASE_URL,
2770
+ defaultModel: OPENCODE_DEFAULT_MODEL,
2771
+ wireApi: opencodeWireApiForModel(preferredModel)
2772
+ };
2773
+ }
2774
+ const copilotToken = env2["0SEC_COPILOT_GITHUB_TOKEN"];
2775
+ if (copilotToken) {
2776
+ return {
2777
+ provider: "copilot",
2778
+ apiKey: copilotToken,
2779
+ baseUrl: env2.COPILOT_BASE_URL ?? COPILOT_API_BASE,
2780
+ defaultModel: preferredModel ?? COPILOT_DEFAULT_MODEL,
2781
+ wireApi: "chat_completions"
2782
+ };
2783
+ }
2784
+ const geminiAccess = env2["0SEC_GEMINI_ACCESS_TOKEN"];
2785
+ const geminiRefresh = env2["0SEC_GEMINI_OAUTH_REFRESH_TOKEN"];
2786
+ if (geminiAccess && geminiAccess.length > 0 || geminiRefresh && geminiRefresh.length > 0) {
2787
+ return {
2788
+ provider: "google",
2789
+ apiKey: "",
2790
+ baseUrl: CODE_ASSIST_ENDPOINT,
2791
+ defaultModel: env2["0SEC_MODEL"] ?? GEMINI_DEFAULT_MODEL,
2792
+ wireApi: "google_generate_content"
2793
+ };
2794
+ }
2795
+ const anthropicKey = env2.ANTHROPIC_API_KEY;
2796
+ if (anthropicKey) {
2797
+ return {
2798
+ provider: "anthropic",
2799
+ apiKey: anthropicKey,
2800
+ baseUrl: env2.ANTHROPIC_BASE_URL ?? "https://api.anthropic.com",
2801
+ defaultModel: DEFAULT_ANTHROPIC_MODEL,
2802
+ wireApi: "chat_completions"
2803
+ };
2804
+ }
2805
+ try {
2806
+ const hostedCreds = loadCloudCredentials({
2807
+ env: env2,
2808
+ warn: () => {
2809
+ }
2810
+ });
2811
+ return {
2812
+ provider: "hosted",
2813
+ apiKey: hostedCreds.token,
2814
+ baseUrl: `${hostedCreds.host}/api/inference/v1`,
2815
+ defaultModel: "",
2816
+ wireApi: "chat_completions"
2817
+ };
2818
+ } catch {
2819
+ }
2820
+ return {
2821
+ provider: "anthropic",
2822
+ apiKey: "",
2823
+ baseUrl: env2.ANTHROPIC_BASE_URL ?? "https://api.anthropic.com",
2824
+ defaultModel: DEFAULT_ANTHROPIC_MODEL,
2825
+ wireApi: "chat_completions"
2826
+ };
2827
+ }
2828
+ var LlmApiRuntime = class _LlmApiRuntime {
2829
+ type = "api";
2830
+ // These are set by the constructor via applyConfiguration() (root) or the
2831
+ // inherited fork branch; the `!` records that a fork sets them directly while
2832
+ // the root path assigns them through the shared applyConfiguration() helper.
2833
+ config;
2834
+ // Not readonly: a live provider switch re-freezes env from the new account.
2835
+ env;
2836
+ codexAuthState;
2837
+ geminiAuthState;
2838
+ provider;
2839
+ apiKey;
2840
+ baseUrl;
2841
+ model;
2842
+ wireApi;
2843
+ reasoningEffort;
2844
+ azureConfig;
2845
+ serverCompactionTokens;
2846
+ /** Ordered fallback chain (0SEC_LLM_FALLBACK). Empty = no failover. */
2847
+ fallbackChain;
2848
+ /** Index into fallbackChain — which entry to try next. */
2849
+ fallbackIndex;
2850
+ /** Resolve and validate the hosted model and wire protocol once per runtime. */
2851
+ hostedCatalogPromise = null;
2852
+ /** Catalog ceiling, resolved before hosted inference is submitted. */
2853
+ hostedMaxOutputTokens;
2854
+ constructor(config, inherited) {
2855
+ if (inherited) {
2856
+ const timeout = config.timeout ?? inherited.config.timeout ?? 12e4;
2857
+ if (!Number.isFinite(timeout) || timeout <= 0) {
2858
+ throw new Error("Subagent timeout must be a positive finite number");
2859
+ }
2860
+ if (inherited.provider === "hosted" && inherited.hostedMaxOutputTokens === void 0) {
2861
+ throw new Error("Hosted model catalog must resolve before creating a subagent");
2862
+ }
2863
+ const model = config.model ?? inherited.model;
2864
+ const modelChanged = model !== inherited.model;
2865
+ this.config = {
2866
+ type: "api",
2867
+ model,
2868
+ agentModels: inherited.config.agentModels,
2869
+ singleModel: inherited.config.singleModel,
2870
+ timeout: Math.min(timeout, inherited.config.timeout || 12e4)
2871
+ };
2872
+ this.env = inherited.env;
2873
+ this.provider = inherited.provider;
2874
+ this.apiKey = inherited.apiKey;
2875
+ this.baseUrl = inherited.baseUrl;
2876
+ this.model = model;
2877
+ this.wireApi = inherited.wireApi;
2878
+ if (modelChanged) {
2879
+ if (this.provider === "opencode")
2880
+ this.wireApi = opencodeWireApiForModel(model);
2881
+ if (this.provider === "openai")
2882
+ this.wireApi = openAICompatibleWireApi(this.env, "OPENAI_WIRE_API");
2883
+ if (this.provider === "azure")
2884
+ this.wireApi = openAICompatibleWireApi(this.env, "AZURE_OPENAI_WIRE_API", inherited.azureConfig.wireApi);
2885
+ this.applyModelWireApi();
2886
+ }
2887
+ this.reasoningEffort = modelChanged ? void 0 : inherited.reasoningEffort;
2888
+ this.azureConfig = { ...inherited.azureConfig };
2889
+ this.serverCompactionTokens = inherited.serverCompactionTokens;
2890
+ if (inherited.codexAuthState)
2891
+ this.codexAuthState = inherited.codexAuthState;
2892
+ if (inherited.geminiAuthState)
2893
+ this.geminiAuthState = inherited.geminiAuthState;
2894
+ this.fallbackChain = [];
2895
+ this.fallbackIndex = 0;
2896
+ if (!modelChanged) {
2897
+ this.hostedMaxOutputTokens = inherited.hostedMaxOutputTokens;
2898
+ this.hostedCatalogPromise = inherited.hostedCatalogPromise;
2899
+ }
2900
+ return;
2901
+ }
2902
+ this.applyConfiguration(config);
2903
+ }
2904
+ /**
2905
+ * Root provider/model detection. Shared by the constructor and
2906
+ * {@link reconfigure} so a live provider switch re-resolves the account,
2907
+ * endpoint, wire protocol and default model with the SAME logic the
2908
+ * constructor uses — never a second, drifting code path.
2909
+ */
2910
+ applyConfiguration(config) {
2911
+ this.config = {
2912
+ ...config,
2913
+ ...config.agentModels ? { agentModels: Object.freeze({ ...config.agentModels }) } : {}
2914
+ };
2915
+ this.env = Object.freeze({ ...process.env, ...config.env });
2916
+ this.azureConfig = parseCodexAzureConfig(this.env);
2917
+ this.fallbackChain = getFallbackChain(this.env).map((entry) => ({
2918
+ ...entry,
2919
+ credentials: resolveFailoverProvider(entry.provider, entry.model, this.env)
2920
+ }));
2921
+ this.fallbackIndex = 0;
2922
+ const detected = detectProvider(config.apiKey, config.model ?? this.env["0SEC_MODEL"], this.env, config.provider);
2923
+ this.provider = detected.provider;
2924
+ this.apiKey = detected.apiKey;
2925
+ this.baseUrl = detected.baseUrl;
2926
+ this.wireApi = detected.wireApi;
2927
+ this.codexAuthState = void 0;
2928
+ if (this.provider === "chatgpt-codex" || this.fallbackChain.some((entry) => entry.provider === "chatgpt-codex")) {
2929
+ if (readChatGptCodexEnv(this.env) || readChatGptCodexAuthFile(this.env)) {
2930
+ this.codexAuthState = resolveChatGptCodexAuthState(this.env);
2931
+ }
2932
+ }
2933
+ this.geminiAuthState = void 0;
2934
+ if (this.provider === "google" || this.fallbackChain.some((entry) => entry.provider === "google")) {
2935
+ if (readGeminiCodeAssistEnv(this.env)) {
2936
+ this.geminiAuthState = resolveGeminiCodeAssistAuthState(this.env);
2937
+ }
2938
+ }
2939
+ this.reasoningEffort = this.env["0SEC_REASONING_EFFORT"] ?? detected.reasoningEffort;
2940
+ this.serverCompactionTokens = config.serverCompactionTokens !== void 0 ? Math.max(1e3, config.serverCompactionTokens) : void 0;
2941
+ const requestedModel = config.model ?? this.env["0SEC_MODEL"];
2942
+ if (requestedModel === "free" && this.provider === "openrouter") {
2943
+ this.model = FREE_OPENROUTER_MODEL;
2944
+ } else {
2945
+ this.model = requestedModel ?? detected.defaultModel;
2946
+ }
2947
+ if (this.provider === "opencode") {
2948
+ this.model = opencodeModelId(this.model);
2949
+ }
2950
+ if (this.provider === "copilot") {
2951
+ this.model = copilotModelId(this.model);
2952
+ }
2953
+ this.applyModelWireApi();
2954
+ if (this.apiKey && !this.env["0SEC_SKIP_PROVIDER_BANNER"]) {
2955
+ void logProviderStartup(this.provider, this.providerLabel, this.baseUrl, this.model, this.wireApi, this.apiKey).catch(() => {
2956
+ });
2957
+ }
2958
+ }
2959
+ /**
2960
+ * Mutate the live selection in place so the NEXT turn (the engine reads
2961
+ * `config.runtime` per turn) and the NEXT `forkForSubagent` pick up the new
2962
+ * model / provider / role map with zero session teardown. No-op-safe:
2963
+ * undefined fields leave the corresponding state unchanged.
2964
+ */
2965
+ reconfigure(sel) {
2966
+ const providerChanged = sel.provider !== void 0 && sel.provider !== this.provider;
2967
+ if (providerChanged) {
2968
+ const merged = {
2969
+ ...this.config,
2970
+ apiKey: void 0,
2971
+ provider: sel.provider,
2972
+ ...sel.model !== void 0 ? { model: sel.model } : {},
2973
+ ...sel.agentModels !== void 0 ? { agentModels: sel.agentModels } : {},
2974
+ ...sel.singleModel !== void 0 ? { singleModel: sel.singleModel } : {},
2975
+ ...sel.env !== void 0 ? { env: sel.env } : {}
2976
+ };
2977
+ this.hostedCatalogPromise = null;
2978
+ this.hostedMaxOutputTokens = void 0;
2979
+ this.applyConfiguration(merged);
2980
+ return;
2981
+ }
2982
+ if (sel.agentModels !== void 0) {
2983
+ this.config = { ...this.config, agentModels: Object.freeze({ ...sel.agentModels }) };
2984
+ }
2985
+ if (sel.singleModel !== void 0) {
2986
+ this.config = { ...this.config, singleModel: sel.singleModel };
2987
+ }
2988
+ if (sel.model !== void 0) {
2989
+ const model = this.provider === "opencode" ? opencodeModelId(sel.model) : this.provider === "copilot" ? copilotModelId(sel.model) : sel.model;
2990
+ if (model !== this.model) {
2991
+ this.model = model;
2992
+ this.config = { ...this.config, model };
2993
+ if (this.provider === "opencode")
2994
+ this.wireApi = opencodeWireApiForModel(model);
2995
+ if (this.provider === "openai")
2996
+ this.wireApi = openAICompatibleWireApi(this.env, "OPENAI_WIRE_API");
2997
+ if (this.provider === "azure")
2998
+ this.wireApi = openAICompatibleWireApi(this.env, "AZURE_OPENAI_WIRE_API", this.azureConfig.wireApi);
2999
+ this.applyModelWireApi();
3000
+ this.reasoningEffort = void 0;
3001
+ }
3002
+ }
3003
+ }
3004
+ /** Exact deployments requiring Responses for tools, shared by roots and forks. */
3005
+ applyModelWireApi() {
3006
+ const normalizedModel = this.model.toLowerCase();
3007
+ if (this.wireApi === "chat_completions" && (this.provider === "azure" && normalizedModel === "gpt-5.6-sol" || this.provider === "openai" && normalizedModel === "gpt-5.6-luna")) {
3008
+ this.wireApi = "responses";
3009
+ }
3010
+ }
3011
+ /** Isolated child inference, bound to this runtime's resolved account and route. */
3012
+ /**
3013
+ * Providers whose credentials are present in THIS runtime's environment
3014
+ * (`this.env`, never process-global). A provider counts as accessible iff
3015
+ * `resolveFailoverProvider` — the same auth-presence check the cross-provider
3016
+ * failover chain uses — can build a connection for it, so auth-only providers
3017
+ * (chatgpt-codex OAuth) and cloud-hosted are covered by the identical rule.
3018
+ */
3019
+ accessibleProviders() {
3020
+ const out = [];
3021
+ for (const provider of Object.keys(DEFAULT_PROVIDER_MODELS)) {
3022
+ const probeModel = DEFAULT_PROVIDER_MODELS[provider] || "probe";
3023
+ try {
3024
+ if (resolveFailoverProvider(provider, probeModel, this.env))
3025
+ out.push(provider);
3026
+ } catch {
3027
+ }
3028
+ }
3029
+ return out;
3030
+ }
3031
+ /**
3032
+ * A concrete, reachable model id per accessible provider (its catalog default)
3033
+ * plus the currently-resolved model. This is the read-only roster the
3034
+ * orchestrator can be shown so it names a model under an "auto" role that is
3035
+ * actually reachable; the fork guard independently accepts any model whose
3036
+ * provider has creds, so this is a helpful starting set, not the whole bound.
3037
+ */
3038
+ accessibleModels() {
3039
+ const models = /* @__PURE__ */ new Set();
3040
+ if (this.model)
3041
+ models.add(this.model);
3042
+ for (const provider of this.accessibleProviders()) {
3043
+ const def = DEFAULT_PROVIDER_MODELS[provider];
3044
+ if (def)
3045
+ models.add(def);
3046
+ }
3047
+ return [...models];
3048
+ }
3049
+ /**
3050
+ * Whether `model` routes to a provider whose credentials are present. Uses the
3051
+ * same per-call `providerForModel` routing the runtime uses everywhere (which
3052
+ * returns a provider only when its key is present), with the concrete
3053
+ * accessible defaults as a fallback for ids the router does not pattern-match.
3054
+ */
3055
+ isModelAccessible(model) {
3056
+ if (providerForModel(model, this.env) !== void 0)
3057
+ return true;
3058
+ return this.accessibleModels().includes(model);
3059
+ }
3060
+ async forkForSubagent(timeoutMs, selection) {
3061
+ if (!Number.isFinite(timeoutMs) || timeoutMs <= 0) {
3062
+ throw new Error("Subagent timeout must be a positive finite number");
3063
+ }
3064
+ await this.ensureHostedModel();
3065
+ const roleModel = selection?.role !== void 0 && this.config.agentModels && Object.hasOwn(this.config.agentModels, selection.role) ? this.config.agentModels[selection.role] : void 0;
3066
+ const autoInEffect = roleModel === AUTO_MODEL_SENTINEL || this.config.autoRoute === true && roleModel === void 0;
3067
+ const configuredModel = roleModel === AUTO_MODEL_SENTINEL ? void 0 : roleModel;
3068
+ const selectedModel = this.config.singleModel ? this.model : selection?.model ?? configuredModel ?? this.model;
3069
+ if (typeof selectedModel !== "string" || !selectedModel.trim()) {
3070
+ throw new Error("Subagent model must be a non-empty model ID");
3071
+ }
3072
+ const stripProviderPrefix = (id) => this.provider === "opencode" ? opencodeModelId(id) : this.provider === "copilot" ? copilotModelId(id) : id;
3073
+ const model = stripProviderPrefix(selectedModel);
3074
+ const approvedByAllowlist = model === this.model || Object.values(this.config.agentModels ?? {}).some((id) => id !== AUTO_MODEL_SENTINEL && stripProviderPrefix(id) === model);
3075
+ const approved = approvedByAllowlist || autoInEffect && this.isModelAccessible(selectedModel);
3076
+ if (!approved) {
3077
+ throw new Error(autoInEffect ? `Subagent model "${selectedModel}" is not reachable: its provider has no configured credentials.` : `Subagent model "${selectedModel}" is not operator-approved. Configure agentModels before selecting it.`);
3078
+ }
3079
+ const child = new _LlmApiRuntime({ type: "api", timeout: timeoutMs, model }, this);
3080
+ await child.ensureHostedModel();
3081
+ return child;
3082
+ }
3083
+ /** The server catalog is authoritative even when a model was selected explicitly. */
3084
+ async ensureHostedModel() {
3085
+ if (this.provider !== "hosted")
3086
+ return;
3087
+ if (!this.hostedCatalogPromise) {
3088
+ this.hostedCatalogPromise = (async () => {
3089
+ const client = new CloudClient({
3090
+ host: this.baseUrl.replace(/\/api\/inference\/v1$/, ""),
3091
+ token: this.apiKey
3092
+ });
3093
+ const catalog = await client.getInferenceModels();
3094
+ const selected = this.model ? catalog.data.find((model) => model.id === this.model) : catalog.data[0];
3095
+ if (!selected) {
3096
+ throw new Error(this.model ? `Hosted model "${this.model}" is unavailable. Run \`0sec models\` for available models.` : "No hosted models are available. Run `0sec models` to check service availability.");
3097
+ }
3098
+ this.model = selected.id;
3099
+ this.wireApi = selected.wire_api;
3100
+ this.hostedMaxOutputTokens = selected.max_output_tokens;
3101
+ })();
3102
+ }
3103
+ await this.hostedCatalogPromise;
3104
+ }
3105
+ /**
3106
+ * A hard dollar ceiling needs a provider-enforced bound on the next response.
3107
+ * ChatGPT Codex OAuth rejects `max_output_tokens`, so it cannot support that
3108
+ * contract; callers must fail closed before making a metered comparison call.
3109
+ */
3110
+ get outputTokenLimit() {
3111
+ return this.provider === "chatgpt-codex" ? void 0 : this.effectiveOutputTokens;
3112
+ }
3113
+ /** Hosted requests honor both the catalog ceiling and the local hard cap. */
3114
+ get effectiveOutputTokens() {
3115
+ if (this.provider === "hosted" && this.hostedMaxOutputTokens !== void 0) {
3116
+ return Math.min(NATIVE_COMPLETION_TOKEN_LIMIT, this.hostedMaxOutputTokens);
3117
+ }
3118
+ return NATIVE_COMPLETION_TOKEN_LIMIT;
3119
+ }
3120
+ /**
3121
+ * Whether this provider uses OpenAI-compatible chat/completions format.
3122
+ *
3123
+ * DO NOT add "z-ai" or "kimi" here. They speak the Anthropic Messages wire
3124
+ * (see `isAnthropicWire`); adding them to this getter would silently route
3125
+ * them to `/chat/completions` with a Bearer header and break them. Their
3126
+ * `wireApi` field is set to "chat_completions" by detectProvider only as an
3127
+ * inert default — it is intentionally unused for these two providers.
3128
+ */
3129
+ get isOpenAICompat() {
3130
+ return this.provider === "openrouter" || this.provider === "openai" || this.provider === "azure" || this.provider === "deepseek" || this.provider === "qwen" || this.provider === "xai" || this.provider === "copilot" || this.provider === "hosted" || this.provider === "opencode" && (this.wireApi === "chat_completions" || this.wireApi === "responses") || // chatgpt-codex always speaks Responses API; treat it as
3131
+ // OpenAI-compat for body-shape branching purposes (the Responses
3132
+ // wire-API code paths below already key on `wireApi === "responses"`
3133
+ // and produce a body codex's backend accepts as-is).
3134
+ this.provider === "chatgpt-codex";
3135
+ }
3136
+ /**
3137
+ * Whether this provider speaks the Anthropic Messages wire (`/v1/messages`
3138
+ * with `x-api-key` + `anthropic-version`, Anthropic-shaped body + response).
3139
+ *
3140
+ * z-ai (GLM) and kimi (Moonshot) are Anthropic-compatible endpoints, so they
3141
+ * ride this wire alongside real Anthropic. This is the POSITIVE predicate
3142
+ * that drives buildUrl / buildHeaders and the Anthropic branches of
3143
+ * execute() / executeNative() — replacing the old implicit "everything that
3144
+ * isn't isOpenAICompat" else-fallthrough, which was a footgun: adding a
3145
+ * provider to isOpenAICompat, or trusting these two's `wireApi` field, would
3146
+ * have silently mis-routed them off the Anthropic wire.
3147
+ */
3148
+ get isAnthropicWire() {
3149
+ return this.provider === "anthropic" || this.provider === "z-ai" || this.provider === "kimi" || this.provider === "opencode" && this.wireApi === "anthropic_messages";
3150
+ }
3151
+ /** Whether this OpenCode Zen model uses the Google generateContent wire. */
3152
+ get isGoogleWire() {
3153
+ return this.provider === "opencode" && this.wireApi === "google_generate_content";
3154
+ }
3155
+ /**
3156
+ * Whether this is the Google Gemini Code Assist backend. It reuses the Google
3157
+ * generateContent request/response SHAPE (googleContents + the isGoogleWire
3158
+ * parser) but wraps the body in a `{ model, project, request }` envelope,
3159
+ * unwraps `{ response }`, authenticates with a refreshed OAuth Bearer, and
3160
+ * posts to a fixed Code Assist endpoint — so it is a distinct predicate from
3161
+ * `isGoogleWire`, checked BEFORE it wherever the body/parse branches.
3162
+ */
3163
+ get isGeminiCodeAssist() {
3164
+ return this.provider === "google";
3165
+ }
3166
+ /**
3167
+ * The resolved model id this runtime will actually call — the requested
3168
+ * model when one was picked, otherwise the provider's detected default.
3169
+ * Surfaced so the pipeline can stamp the engine-resolved model on
3170
+ * `scan_completed` (CI review scans are dispatched with no model pick, so
3171
+ * this is the only place the concrete id exists).
3172
+ */
3173
+ resolvedModel() {
3174
+ return this.model;
3175
+ }
3176
+ /** Build the appropriate headers for the configured provider. */
3177
+ buildHeaders() {
3178
+ if (this.provider === "chatgpt-codex") {
3179
+ return {
3180
+ "Content-Type": "application/json",
3181
+ originator: "0sec",
3182
+ "User-Agent": `0sec/${VERSION}`
3183
+ };
3184
+ }
3185
+ if (this.isGeminiCodeAssist) {
3186
+ return {
3187
+ "Content-Type": "application/json",
3188
+ "User-Agent": geminiUserAgent(this.model)
3189
+ };
3190
+ }
3191
+ if (this.isGoogleWire) {
3192
+ return {
3193
+ "Content-Type": "application/json",
3194
+ "x-goog-api-key": this.apiKey
3195
+ };
3196
+ }
3197
+ if (this.provider === "copilot") {
3198
+ return {
3199
+ "Content-Type": "application/json",
3200
+ Authorization: `Bearer ${this.apiKey}`,
3201
+ ...COPILOT_STATIC_HEADERS
3202
+ };
3203
+ }
3204
+ if (this.isOpenAICompat) {
3205
+ const headers = {
3206
+ "Content-Type": "application/json"
3207
+ };
3208
+ if (this.provider === "azure") {
3209
+ headers["api-key"] = this.apiKey;
3210
+ } else {
3211
+ headers["Authorization"] = `Bearer ${this.apiKey}`;
3212
+ }
3213
+ if (this.provider === "openrouter") {
3214
+ headers["HTTP-Referer"] = "https://0sec.ai";
3215
+ headers["X-Title"] = "0sec Security Scanner";
3216
+ }
3217
+ return headers;
3218
+ }
3219
+ if (this.isAnthropicWire) {
3220
+ return {
3221
+ "Content-Type": "application/json",
3222
+ "x-api-key": this.apiKey,
3223
+ "anthropic-version": "2023-06-01"
3224
+ };
3225
+ }
3226
+ throw new Error(`buildHeaders: provider ${this.provider} is not mapped to a wire`);
3227
+ }
3228
+ /**
3229
+ * For the chatgpt-codex provider, decorate the headers with the
3230
+ * freshly-refreshed OAuth bearer + `ChatGPT-Account-Id` + a
3231
+ * stable `session_id` (opencode codex.ts:614 — used by Codex
3232
+ * backend for request correlation + rate-limit attribution +
3233
+ * prompt-cache affinity). For every other provider it's a no-op
3234
+ * that returns the stock headers unchanged. Caller MUST await
3235
+ * this before the fetch — that's where token refresh actually
3236
+ * happens.
3237
+ *
3238
+ * session_id is process-stable (PROCESS_SESSION_ID, randomised
3239
+ * once at module load). A 0sec-cli invocation = one scan = one
3240
+ * session, so the process-lifetime constant is the right
3241
+ * granularity. If we ever want per-scan ids inside a long-lived
3242
+ * controller process, add a setter on the runtime; for now this
3243
+ * matches how the CLI is actually invoked.
3244
+ */
3245
+ async ensureFreshHeaders() {
3246
+ const base = this.buildHeaders();
3247
+ if (this.provider === "google") {
3248
+ if (!this.geminiAuthState)
3249
+ throw new Error("Google Gemini Code Assist auth: no credential captured for this runtime");
3250
+ const accessToken2 = await refreshGeminiCodeAssistAuthState(this.geminiAuthState);
3251
+ base["Authorization"] = `Bearer ${accessToken2}`;
3252
+ return base;
3253
+ }
3254
+ if (this.provider !== "chatgpt-codex")
3255
+ return base;
3256
+ if (!this.codexAuthState)
3257
+ throw new Error("ChatGPT Codex auth: no credential captured for this runtime");
3258
+ const { accessToken, accountId } = await refreshChatGptCodexAuthState(this.codexAuthState);
3259
+ base["Authorization"] = `Bearer ${accessToken}`;
3260
+ if (accountId)
3261
+ base["ChatGPT-Account-Id"] = accountId;
3262
+ base["session_id"] = PROCESS_SESSION_ID;
3263
+ base["Accept"] = "text/event-stream";
3264
+ return base;
3265
+ }
3266
+ /** Build the API endpoint URL. */
3267
+ buildUrl() {
3268
+ if (this.provider === "chatgpt-codex") {
3269
+ return CODEX_API_ENDPOINT;
3270
+ }
3271
+ if (this.isGeminiCodeAssist) {
3272
+ return `${CODE_ASSIST_ENDPOINT}/${CODE_ASSIST_API_VERSION}:generateContent`;
3273
+ }
3274
+ if (this.isGoogleWire) {
3275
+ return `${this.baseUrl}/models/${this.model}:generateContent`;
3276
+ }
3277
+ if (this.isOpenAICompat) {
3278
+ return `${this.baseUrl}/${this.wireApi === "responses" ? "responses" : "chat/completions"}`;
3279
+ }
3280
+ if (this.isAnthropicWire) {
3281
+ return this.provider === "opencode" ? `${this.baseUrl}/messages` : `${this.baseUrl}/v1/messages`;
3282
+ }
3283
+ throw new Error(`buildUrl: provider ${this.provider} is not mapped to a wire`);
3284
+ }
3285
+ /**
3286
+ * Chat-completions param name for the token cap. Newer OpenAI model
3287
+ * families (gpt-5/6, o1/o2/o3) reject the legacy `max_tokens` field
3288
+ * and require `max_completion_tokens`. Older models still accept the
3289
+ * legacy name, so we flip based on model prefix.
3290
+ */
3291
+ get maxTokensParamKey() {
3292
+ return /^gpt-[56](?:[-.]|$)|^o[1-3](?:[-_]|$)/i.test(this.model) ? "max_completion_tokens" : "max_tokens";
3293
+ }
3294
+ /**
3295
+ * Anthropic `thinking` body fragment. Real Anthropic Claude uses adaptive
3296
+ * thinking when retained reasoning is enabled. Z.ai GLM-5.3 requires
3297
+ * enabled thinking plus reasoning_effort; earlier GLM models use a
3298
+ * budget_tokens field. Kimi reasons natively and accepts neither field.
3299
+ */
3300
+ anthropicThinkingField() {
3301
+ if (this.provider === "anthropic") {
3302
+ return features.retainedReasoning ? { thinking: { type: "adaptive" } } : {};
3303
+ }
3304
+ if (this.provider !== "z-ai")
3305
+ return {};
3306
+ const budget = zaiThinkingBudget();
3307
+ if (this.model.startsWith("glm-5.3")) {
3308
+ const reasoningEffort = this.reasoningEffort ?? (budget <= 2048 ? "low" : budget <= 4096 ? "high" : "max");
3309
+ return {
3310
+ thinking: { type: "enabled" },
3311
+ reasoning_effort: reasoningEffort
3312
+ };
3313
+ }
3314
+ if (budget <= 0)
3315
+ return {};
3316
+ return { thinking: { type: "enabled", budget_tokens: budget } };
3317
+ }
3318
+ /** Convert the unified transcript into Gemini generateContent contents. */
3319
+ googleContents(messages) {
3320
+ const toolNames = /* @__PURE__ */ new Map();
3321
+ const upstreamCallIds = /* @__PURE__ */ new Map();
3322
+ const contents = [];
3323
+ for (const message of messages) {
3324
+ const parts = [];
3325
+ const rawParts = message.role === "assistant" && message.providerRaw?.provider === this.provider && message.providerRaw.model === this.model && message.providerRaw.wireApi === this.wireApi && Array.isArray(message.providerRaw.output) ? message.providerRaw.output : void 0;
3326
+ if (rawParts) {
3327
+ const toolUses = message.content.filter((block) => block.type === "tool_use");
3328
+ let toolUseIndex = 0;
3329
+ for (const part of rawParts) {
3330
+ const call = part.functionCall && typeof part.functionCall === "object" ? part.functionCall : void 0;
3331
+ const toolUse = call ? toolUses[toolUseIndex++] : void 0;
3332
+ const name = typeof call?.name === "string" ? call.name : toolUse?.name;
3333
+ if (toolUse && name) {
3334
+ toolNames.set(toolUse.id, name);
3335
+ if (typeof call?.id === "string")
3336
+ upstreamCallIds.set(toolUse.id, call.id);
3337
+ }
3338
+ parts.push(part);
3339
+ }
3340
+ } else {
3341
+ for (const block of message.content) {
3342
+ if (block.type === "text") {
3343
+ parts.push({ text: block.text });
3344
+ } else if (block.type === "tool_use") {
3345
+ toolNames.set(block.id, block.name);
3346
+ upstreamCallIds.set(block.id, block.id);
3347
+ parts.push({ functionCall: { id: block.id, name: block.name, args: block.input } });
3348
+ } else if (block.type === "tool_result") {
3349
+ const name = toolNames.get(block.tool_use_id);
3350
+ if (!name) {
3351
+ throw new Error(`Google tool result ${block.tool_use_id} has no matching tool call`);
3352
+ }
3353
+ const upstreamId = upstreamCallIds.get(block.tool_use_id);
3354
+ parts.push({
3355
+ functionResponse: {
3356
+ ...upstreamId ? { id: upstreamId } : {},
3357
+ name,
3358
+ response: {
3359
+ name,
3360
+ content: block.is_error ? `Error: ${block.content}` : block.content
3361
+ }
3362
+ }
3363
+ });
3364
+ }
3365
+ }
3366
+ }
3367
+ if (parts.length > 0) {
3368
+ contents.push({ role: message.role === "assistant" ? "model" : "user", parts });
3369
+ }
3370
+ }
3371
+ return contents;
3372
+ }
3373
+ /**
3374
+ * Wrap a standard Gemini generateContent request body in the Code Assist
3375
+ * envelope `{ model, project, user_prompt_id, request }`. Resolves (and
3376
+ * caches, via a singleflight) the account's project id first; the FREE tier
3377
+ * resolves to `""` and the `project` field is then omitted entirely.
3378
+ */
3379
+ async wrapGeminiCodeAssistBody(request) {
3380
+ if (!this.geminiAuthState)
3381
+ throw new Error("Google Gemini Code Assist auth: no credential captured for this runtime");
3382
+ const project = await resolveGeminiCodeAssistProject(this.geminiAuthState, this.env);
3383
+ return {
3384
+ model: this.model,
3385
+ ...project ? { project } : {},
3386
+ user_prompt_id: randomUUID(),
3387
+ request
3388
+ };
3389
+ }
3390
+ /**
3391
+ * Per-turn prompt-cache accounting line, so a run can be shown to actually
3392
+ * be hitting cache rather than assumed to be. Off unless
3393
+ * `0SEC_DEBUG_PROMPT_CACHE` is set — this fires once per agent turn, and an
3394
+ * unconditional line would interleave with the TUI on every scan.
3395
+ *
3396
+ * The same numbers reach the cloud without this flag: `cachedInputTokens`
3397
+ * flows into `ScanCostLedger` and the `scan_completed` cost breakdown, which
3398
+ * is the durable, queryable proof. This is the local fast path.
3399
+ */
3400
+ logCacheUsage(usage) {
3401
+ if (!usage || !process.env["0SEC_DEBUG_PROMPT_CACHE"])
3402
+ return;
3403
+ const read = usage.cachedInputTokens ?? 0;
3404
+ const write = usage.cacheWriteTokens ?? 0;
3405
+ const hitRate = usage.inputTokens > 0 ? Math.round(read / usage.inputTokens * 100) : 0;
3406
+ diag.info("prompt_cache_usage", `prompt-cache ${this.providerLabel}`, {
3407
+ provider: this.providerLabel,
3408
+ read,
3409
+ write,
3410
+ uncached: usage.inputTokens - read - write,
3411
+ total_in: usage.inputTokens,
3412
+ hit_pct: hitRate
3413
+ });
3414
+ }
3415
+ /** Friendly provider name for error messages. */
3416
+ get providerLabel() {
3417
+ switch (this.provider) {
3418
+ case "openrouter":
3419
+ return "OpenRouter";
3420
+ case "anthropic":
3421
+ return "Anthropic";
3422
+ case "openai":
3423
+ return "OpenAI";
3424
+ case "azure":
3425
+ return "Azure OpenAI";
3426
+ case "deepseek":
3427
+ return "DeepSeek";
3428
+ case "chatgpt-codex":
3429
+ return "ChatGPT (Codex backend)";
3430
+ case "z-ai":
3431
+ return "Z.ai (GLM)";
3432
+ case "kimi":
3433
+ return "Kimi (Moonshot)";
3434
+ case "qwen":
3435
+ return "Qwen (Alibaba Model Studio)";
3436
+ case "xai":
3437
+ return "xAI (Grok)";
3438
+ case "opencode":
3439
+ return "OpenCode Zen";
3440
+ case "copilot":
3441
+ return "GitHub Copilot";
3442
+ case "google":
3443
+ return "Google Gemini (Code Assist)";
3444
+ case "hosted":
3445
+ return "0sec Cloud";
3446
+ }
3447
+ }
3448
+ noKeyError() {
3449
+ return "No provider credential found. Set one of:\n env 0SEC_CHATGPT_OAUTH_REFRESH_TOKEN=... 0sec <command> (ChatGPT Codex subscription auth)\n export OPENROUTER_API_KEY=sk-or-... (OpenRouter \u2014 many models, one key)\n export DEEPSEEK_API_KEY=... (DeepSeek \u2014 direct Flash 0731 inference)\n export ANTHROPIC_API_KEY=sk-ant-... (Anthropic \u2014 direct Claude access)\n export AZURE_OPENAI_API_KEY=... (Azure OpenAI \u2014 reuse your Codex Azure provider)\n export OPENAI_API_KEY=sk-... (OpenAI \u2014 direct GPT access)\n export Z_AI_API_KEY=... (Z.ai GLM \u2014 flat-rate Coding Plan, Anthropic-compatible)\n export KIMI_API_KEY=... (Moonshot Kimi K3 \u2014 flat-rate coding, Anthropic-compatible)\n export QWEN_API_KEY=... (Alibaba Qwen \u2014 Token Plan sub, OpenAI-compatible)\n export XAI_API_KEY=... (xAI Grok \u2014 OpenAI-compatible)\n export OPENCODE_API_KEY=... (OpenCode Zen \u2014 multi-wire gateway)\n export 0SEC_COPILOT_GITHUB_TOKEN=... (GitHub Copilot \u2014 device-code OAuth token)\n Run `0sec login` (0sec hosted inference)";
3450
+ }
3451
+ getConfigurationDiagnostics() {
3452
+ if (!this.apiKey && this.provider !== "chatgpt-codex") {
3453
+ return {
3454
+ valid: false,
3455
+ provider: this.provider,
3456
+ providerLabel: this.providerLabel,
3457
+ reason: "missing_key",
3458
+ fatalError: this.noKeyError()
3459
+ };
3460
+ }
3461
+ if (this.provider !== "azure") {
3462
+ return {
3463
+ valid: true,
3464
+ provider: this.provider,
3465
+ providerLabel: this.providerLabel
3466
+ };
3467
+ }
3468
+ const hasConfiguredBaseUrl = !!(this.env.AZURE_OPENAI_BASE_URL || this.env.OPENAI_BASE_URL || this.azureConfig.baseUrl);
3469
+ const hasConfiguredModel = !!(this.config.model || this.env["0SEC_MODEL"] || this.env.AZURE_OPENAI_MODEL || this.azureConfig.model);
3470
+ const missing = [];
3471
+ if (!hasConfiguredBaseUrl) {
3472
+ missing.push("AZURE_OPENAI_BASE_URL (or [model_providers.azure].base_url in ~/.codex/config.toml)");
3473
+ }
3474
+ if (!hasConfiguredModel) {
3475
+ missing.push('AZURE_OPENAI_MODEL or an Azure-backed `model = "..."` in ~/.codex/config.toml');
3476
+ }
3477
+ if (missing.length > 0) {
3478
+ return {
3479
+ valid: false,
3480
+ provider: this.provider,
3481
+ providerLabel: this.providerLabel,
3482
+ reason: "invalid_config",
3483
+ fatalError: `Azure OpenAI runtime is selected, but the configuration is incomplete.
3484
+ Missing: ${missing.join("; ")}
3485
+ 0sec will not guess Azure defaults because that can silently route to the wrong endpoint or deployment.`
3486
+ };
3487
+ }
3488
+ return {
3489
+ valid: true,
3490
+ provider: this.provider,
3491
+ providerLabel: this.providerLabel
3492
+ };
3493
+ }
3494
+ /**
3495
+ * POST to the provider endpoint and, on a retryable HTTP status
3496
+ * (429 rate-limit / transient 5xx), back off and retry — honoring a
3497
+ * `Retry-After` header when present, otherwise exponential backoff with
3498
+ * full jitter (so a burst of concurrent scans desynchronises instead of
3499
+ * hammering the limit in lockstep).
3500
+ *
3501
+ * Two 429 classes are handled differently:
3502
+ * - per-minute rate limit → retry with the wider 429 budget
3503
+ * (0SEC_LLM_429_MAX_RETRIES attempts / 0SEC_LLM_429_MAX_RETRY_WAIT_MS
3504
+ * cumulative, defaults 12 / 5min) since the limiter resets every ~60s;
3505
+ * `Retry-After` / `retry-after-ms` headers are honored up to a 120s cap.
3506
+ * - plan-quota exhaustion (`usage_limit_reached`, resets in hours/days) →
3507
+ * skips retries and immediately advances `0SEC_LLM_FALLBACK`; if no
3508
+ * configured fallback has credentials, it throws QuotaExhaustedError.
3509
+ *
3510
+ * Other retryable statuses (transient 5xx) keep the generic budget:
3511
+ * 0SEC_LLM_MAX_RETRIES (attempts) and 0SEC_LLM_MAX_RETRY_WAIT_MS
3512
+ * (cumulative backoff). On exhaustion it returns the last still-failing
3513
+ * Response with its body intact, so the caller's existing `!res.ok` branch
3514
+ * surfaces the clear "API error <status>" message — a rate-limit never
3515
+ * masquerades as silent no-work.
3516
+ *
3517
+ * Headers are re-resolved per attempt (via ensureFreshHeaders → OAuth
3518
+ * refresh) so a token that rotated during the wait is picked up. The body
3519
+ * is fixed across attempts.
3520
+ */
3521
+ /**
3522
+ * Try the next fallback provider in the chain (0SEC_LLM_FALLBACK).
3523
+ * Updates `this.provider`, `this.model`, `this.apiKey`, `this.baseUrl`,
3524
+ * `this.wireApi` to match the next valid provider. Returns `true` when a
3525
+ * valid next provider was found and switched to, `false` when the chain is
3526
+ * exhausted.
3527
+ */
3528
+ _tryFailover(reason) {
3529
+ while (this.fallbackIndex < this.fallbackChain.length) {
3530
+ const entry = this.fallbackChain[this.fallbackIndex];
3531
+ this.fallbackIndex++;
3532
+ const cfg = entry.credentials;
3533
+ if (!cfg) {
3534
+ diag.warn("failover_provider_skipped", `0SEC_LLM_FALLBACK: skipping ${entry.provider} (auth env missing)`, { provider: entry.provider, model: entry.model, cause: "auth-env-missing" });
3535
+ continue;
3536
+ }
3537
+ this.provider = entry.provider;
3538
+ this.model = entry.provider === "opencode" ? opencodeModelId(entry.model) : entry.provider === "copilot" ? copilotModelId(entry.model) : entry.model;
3539
+ this.apiKey = cfg.apiKey;
3540
+ this.baseUrl = cfg.baseUrl;
3541
+ this.wireApi = cfg.wireApi;
3542
+ this.hostedCatalogPromise = null;
3543
+ diag.warn("failover_engaged", `${reason} \u2014 failover to ${entry.provider} (${entry.model})`, { reason, provider: entry.provider, model: entry.model });
3544
+ return true;
3545
+ }
3546
+ return false;
3547
+ }
3548
+ /**
3549
+ * POST to the provider endpoint and, on a retryable HTTP status
3550
+ * (429 rate-limit / transient 5xx), back off and retry — honoring a
3551
+ * `Retry-After` header when present, otherwise exponential backoff with
3552
+ * full jitter (so a burst of concurrent scans desynchronises instead of
3553
+ * hammering the limit in lockstep).
3554
+ *
3555
+ * The body factory is valid only for the current provider and wire protocol.
3556
+ * A null result signals failover: the caller resolves the hosted catalog and
3557
+ * rebuilds the complete request under its existing timeout/cancellation signal.
3558
+ *
3559
+ * Retry + failover caps documented on `retryBackoffMs` / `llm429MaxRetries`.
3560
+ *
3561
+ * Headers are re-resolved per attempt (via ensureFreshHeaders → OAuth
3562
+ * refresh) so a token that rotated during the wait is picked up.
3563
+ */
3564
+ async postWithRetry(bodyFactory, signal, abort) {
3565
+ let waited429Ms = 0;
3566
+ let waitedOtherMs = 0;
3567
+ for (let attempt = 0; ; attempt++) {
3568
+ abort?.throwIfCancelled();
3569
+ let res;
3570
+ try {
3571
+ res = await fetch(this.buildUrl(), {
3572
+ method: "POST",
3573
+ headers: await this.ensureFreshHeaders(),
3574
+ body: bodyFactory(),
3575
+ signal
3576
+ });
3577
+ } catch (error) {
3578
+ abort?.throwIfCancelled();
3579
+ if (this.provider === "hosted") {
3580
+ throw new Error("0sec hosted request outcome is unknown. Automatic replay is disabled; check your inference usage before retrying.", { cause: error });
3581
+ }
3582
+ const cause = error instanceof Error ? error.cause : void 0;
3583
+ const causeCode = cause && typeof cause === "object" && "code" in cause && typeof cause.code === "string" ? cause.code : "unknown";
3584
+ const causeHost = cause && typeof cause === "object" && "hostname" in cause && typeof cause.hostname === "string" ? `@${cause.hostname}` : "";
3585
+ const message = error instanceof Error ? error.message : String(error);
3586
+ const maxRetries2 = llmMaxRetries();
3587
+ const maxWaitMs2 = llmMaxRetryWaitMs();
3588
+ const delay2 = retryBackoffMs(attempt);
3589
+ if (isRetryableTransportCode(causeCode) && attempt < maxRetries2 && waitedOtherMs + delay2 <= maxWaitMs2) {
3590
+ diag.warn("transport_retry", `${this.providerLabel} transport ${causeCode} \u2014 backoff ${delay2}ms`, {
3591
+ provider: this.providerLabel,
3592
+ cause_code: causeCode,
3593
+ delay_ms: delay2,
3594
+ attempt: attempt + 1,
3595
+ max_retries: maxRetries2
3596
+ });
3597
+ waitedOtherMs += delay2;
3598
+ await sleepWithAbort(delay2, signal);
3599
+ continue;
3600
+ }
3601
+ throw new Error(`${this.providerLabel} transport failure [${causeCode}${causeHost}]: ${message}`, {
3602
+ cause: error
3603
+ });
3604
+ }
3605
+ if (res.ok || !isRetryableHttpStatus(res.status)) {
3606
+ return res;
3607
+ }
3608
+ if (this.provider === "hosted" && res.status === 429 && res.headers.get("x-0sec-retry-safe") !== "1") {
3609
+ return res;
3610
+ }
3611
+ if (this.provider === "hosted" && res.status >= 500) {
3612
+ await res.body?.cancel();
3613
+ throw new Error(`0sec hosted request returned HTTP ${res.status}; its outcome may be unknown. Automatic replay is disabled; check your inference usage before retrying.`);
3614
+ }
3615
+ abort?.throwIfCancelled();
3616
+ const is429 = res.status === 429;
3617
+ let bodyText;
3618
+ if (is429) {
3619
+ try {
3620
+ bodyText = await res.text?.();
3621
+ } catch {
3622
+ bodyText = void 0;
3623
+ }
3624
+ const quota = bodyText != null ? parseUsageLimitReached(bodyText) : void 0;
3625
+ if (quota) {
3626
+ const resetsAtIso = quota.resetsAtMs != null ? new Date(quota.resetsAtMs).toISOString() : "unknown";
3627
+ appendNativeTrace({
3628
+ kind: "quota-exhausted",
3629
+ provider: this.providerLabel,
3630
+ status: res.status,
3631
+ planType: quota.planType ?? null,
3632
+ resetsAtMs: quota.resetsAtMs ?? null
3633
+ });
3634
+ const quotaKind = quota.quotaKind ?? "quota_exhausted";
3635
+ diag.error("quota_exhausted", `${this.providerLabel} ${quotaKind} \u2014 plan quota exhausted; skipping retry`, {
3636
+ provider: this.providerLabel,
3637
+ quota_kind: quotaKind,
3638
+ plan: quota.planType ?? "unknown",
3639
+ resets_at: resetsAtIso,
3640
+ status: res.status
3641
+ });
3642
+ const quotaError = new QuotaExhaustedError(`${this.providerLabel} ${quotaKind}: plan quota exhausted (plan=${quota.planType ?? "unknown"}, resets_at=${resetsAtIso}) \u2014 reschedulable after reset`, quota);
3643
+ if (this._tryFailover("plan quota exhausted")) {
3644
+ return null;
3645
+ }
3646
+ throw quotaError;
3647
+ }
3648
+ }
3649
+ const maxRetries = is429 ? llm429MaxRetries() : llmMaxRetries();
3650
+ const maxWaitMs = is429 ? llm429MaxRetryWaitMs() : llmMaxRetryWaitMs();
3651
+ const waitedMs = is429 ? waited429Ms : waitedOtherMs;
3652
+ const handBack = () => is429 && bodyText != null ? new Response(bodyText, {
3653
+ status: res.status,
3654
+ statusText: res.statusText,
3655
+ headers: res.headers
3656
+ }) : res;
3657
+ if (attempt >= maxRetries) {
3658
+ if (is429 && this._tryFailover("429 retry budget exhausted")) {
3659
+ return null;
3660
+ }
3661
+ return handBack();
3662
+ }
3663
+ const retryAfter = is429 ? retryAfterMsFromHeaders(res.headers) : parseRetryAfterMs(res.headers?.get?.("retry-after"));
3664
+ const delay = retryAfter ?? retryBackoffMs(attempt, is429 ? 3e4 : 2e4);
3665
+ if (waitedMs + delay > maxWaitMs) {
3666
+ if (is429 && this._tryFailover("429 retry budget exhausted")) {
3667
+ return null;
3668
+ }
3669
+ return handBack();
3670
+ }
3671
+ if (!is429) {
3672
+ try {
3673
+ await res.text?.();
3674
+ } catch {
3675
+ }
3676
+ }
3677
+ appendNativeTrace({
3678
+ kind: "retry",
3679
+ provider: this.providerLabel,
3680
+ status: res.status,
3681
+ attempt: attempt + 1,
3682
+ delayMs: delay,
3683
+ retryAfterHonored: retryAfter != null
3684
+ });
3685
+ diag.warn("retry_backoff", `${this.providerLabel} HTTP ${res.status} \u2014 backoff ${delay}ms`, {
3686
+ provider: this.providerLabel,
3687
+ status: res.status,
3688
+ delay_ms: delay,
3689
+ attempt: attempt + 1,
3690
+ max_retries: maxRetries,
3691
+ budget_used_ms: waitedMs,
3692
+ budget_max_ms: maxWaitMs,
3693
+ retry_after_honored: retryAfter != null
3694
+ });
3695
+ if (is429) {
3696
+ waited429Ms += delay;
3697
+ } else {
3698
+ waitedOtherMs += delay;
3699
+ }
3700
+ await sleepWithAbort(delay, signal);
3701
+ }
3702
+ }
3703
+ // ── Legacy Runtime interface (single-prompt) ──
3704
+ async execute(prompt, context) {
3705
+ await this.ensureHostedModel();
3706
+ const start = Date.now();
3707
+ if (!this.apiKey && this.provider !== "chatgpt-codex" && this.provider !== "google") {
3708
+ return {
3709
+ output: "",
3710
+ exitCode: 1,
3711
+ timedOut: false,
3712
+ durationMs: Date.now() - start,
3713
+ error: this.noKeyError()
3714
+ };
3715
+ }
3716
+ const systemPrompt = context?.systemPrompt ?? "";
3717
+ const controller = new AbortController();
3718
+ const timer = setTimeout(() => controller.abort(), this.config.timeout || 12e4);
3719
+ try {
3720
+ let res;
3721
+ do {
3722
+ if (this.isOpenAICompat && this.wireApi === "chat_completions") {
3723
+ const messages = [];
3724
+ if (systemPrompt) {
3725
+ messages.push({ role: "system", content: systemPrompt });
3726
+ }
3727
+ messages.push({ role: "user", content: prompt });
3728
+ res = await this.postWithRetry(() => JSON.stringify({
3729
+ model: this.model,
3730
+ [this.maxTokensParamKey]: this.effectiveOutputTokens,
3731
+ messages,
3732
+ // See executeNative: explicit reasoning_effort passthrough only.
3733
+ ...this.reasoningEffort ? { reasoning_effort: this.reasoningEffort } : {}
3734
+ }), controller.signal);
3735
+ } else if (this.isOpenAICompat && this.wireApi === "responses") {
3736
+ const input = [];
3737
+ if (systemPrompt) {
3738
+ input.push({
3739
+ role: "system",
3740
+ content: [{ type: "input_text", text: systemPrompt }]
3741
+ });
3742
+ }
3743
+ input.push({
3744
+ role: "user",
3745
+ content: [{ type: "input_text", text: prompt }]
3746
+ });
3747
+ const isCodex = this.provider === "chatgpt-codex";
3748
+ res = await this.postWithRetry(() => JSON.stringify({
3749
+ model: this.model,
3750
+ input,
3751
+ ...isCodex ? { store: false } : { max_output_tokens: this.effectiveOutputTokens }
3752
+ }), controller.signal);
3753
+ } else if (this.isGeminiCodeAssist) {
3754
+ const request = {
3755
+ ...systemPrompt ? { systemInstruction: { parts: [{ text: systemPrompt }] } } : {},
3756
+ contents: [{ role: "user", parts: [{ text: prompt }] }],
3757
+ generationConfig: { maxOutputTokens: NATIVE_COMPLETION_TOKEN_LIMIT }
3758
+ };
3759
+ const geminiBody = await this.wrapGeminiCodeAssistBody(request);
3760
+ res = await this.postWithRetry(() => JSON.stringify(geminiBody), controller.signal);
3761
+ } else if (this.isGoogleWire) {
3762
+ res = await this.postWithRetry(() => JSON.stringify({
3763
+ ...systemPrompt ? { systemInstruction: { parts: [{ text: systemPrompt }] } } : {},
3764
+ contents: [{ role: "user", parts: [{ text: prompt }] }],
3765
+ generationConfig: { maxOutputTokens: NATIVE_COMPLETION_TOKEN_LIMIT }
3766
+ }), controller.signal);
3767
+ } else if (this.isAnthropicWire) {
3768
+ res = await this.postWithRetry(() => JSON.stringify({
3769
+ model: this.model,
3770
+ max_tokens: NATIVE_COMPLETION_TOKEN_LIMIT,
3771
+ ...this.anthropicThinkingField(),
3772
+ ...systemPrompt ? { system: systemPrompt } : {},
3773
+ messages: [{ role: "user", content: prompt }]
3774
+ }), controller.signal);
3775
+ } else {
3776
+ throw new Error(`execute: provider ${this.provider} is not mapped to a wire`);
3777
+ }
3778
+ if (!res)
3779
+ await this.ensureHostedModel();
3780
+ } while (!res);
3781
+ clearTimeout(timer);
3782
+ const body = await res.text();
3783
+ if (!res.ok) {
3784
+ appendNativeTrace({
3785
+ kind: "error-response",
3786
+ provider: this.providerLabel,
3787
+ status: res.status,
3788
+ body: body.slice(0, 2e3)
3789
+ });
3790
+ return {
3791
+ output: "",
3792
+ exitCode: 1,
3793
+ timedOut: false,
3794
+ durationMs: Date.now() - start,
3795
+ error: `${this.providerLabel} API error ${res.status}: ${body.slice(0, 500)}`
3796
+ };
3797
+ }
3798
+ const parsedBody = JSON.parse(body);
3799
+ const json = this.isGeminiCodeAssist ? parsedBody.response ?? parsedBody : parsedBody;
3800
+ let text;
3801
+ if (this.isOpenAICompat && this.wireApi === "chat_completions") {
3802
+ const msg = json.choices?.[0]?.message;
3803
+ text = msg?.content ?? msg?.reasoning ?? "";
3804
+ } else if (this.isOpenAICompat && this.wireApi === "responses") {
3805
+ text = typeof json.output_text === "string" && json.output_text.trim() ? json.output_text : Array.isArray(json.output) ? json.output.flatMap((item) => Array.isArray(item.content) ? item.content : []).filter((block) => block.type === "output_text").map((block) => String(block.text ?? "")).join("\n") : "";
3806
+ } else if (this.isGoogleWire || this.isGeminiCodeAssist) {
3807
+ text = json.candidates?.[0]?.content?.parts?.filter((part) => typeof part.text === "string").map((part) => part.text).join("\n") ?? "";
3808
+ } else {
3809
+ text = json.content?.filter((b) => b.type === "text").map((b) => b.text).join("\n") ?? "";
3810
+ }
3811
+ let usage;
3812
+ if (this.isAnthropicWire) {
3813
+ usage = readCacheUsage(json.usage);
3814
+ } else if ((this.isGoogleWire || this.isGeminiCodeAssist) && json.usageMetadata) {
3815
+ usage = {
3816
+ inputTokens: json.usageMetadata.promptTokenCount ?? 0,
3817
+ outputTokens: json.usageMetadata.candidatesTokenCount ?? 0
3818
+ };
3819
+ } else if (this.isOpenAICompat && json.usage) {
3820
+ usage = this.wireApi === "chat_completions" ? { inputTokens: json.usage.prompt_tokens ?? 0, outputTokens: json.usage.completion_tokens ?? 0 } : { inputTokens: json.usage.input_tokens ?? 0, outputTokens: json.usage.output_tokens ?? 0 };
3821
+ }
3822
+ return {
3823
+ output: text,
3824
+ exitCode: 0,
3825
+ timedOut: false,
3826
+ durationMs: Date.now() - start,
3827
+ ...usage ? { usage } : {}
3828
+ };
3829
+ } catch (err) {
3830
+ clearTimeout(timer);
3831
+ if (err instanceof QuotaExhaustedError) {
3832
+ return {
3833
+ output: "",
3834
+ exitCode: 1,
3835
+ timedOut: false,
3836
+ durationMs: Date.now() - start,
3837
+ error: err.message
3838
+ };
3839
+ }
3840
+ const msg = err instanceof Error ? err.message : String(err);
3841
+ const timedOut = msg.includes("abort") || msg.includes("timeout");
3842
+ return {
3843
+ output: "",
3844
+ exitCode: 1,
3845
+ timedOut,
3846
+ durationMs: Date.now() - start,
3847
+ error: timedOut ? `${this.providerLabel} API request timed out` : `${this.providerLabel} API error: ${msg}`
3848
+ };
3849
+ }
3850
+ }
3851
+ // ── Native Runtime interface (structured messages + tool_use) ──
3852
+ /** Terminal result for an operator cancellation — `stopReason:"error"` for compatibility, `cancelled:true` for consumers that can tell the difference. */
3853
+ cancelledResult(start) {
3854
+ return {
3855
+ content: [{ type: "text", text: "" }],
3856
+ stopReason: "error",
3857
+ cancelled: true,
3858
+ durationMs: Date.now() - start,
3859
+ error: `${this.providerLabel} request cancelled by operator`
3860
+ };
3861
+ }
3862
+ /**
3863
+ * Public entry point. Runs {@link executeNativeAttempt} and, ONLY for a
3864
+ * transient empty stream (see {@link shouldRetryNativeStream}), re-issues the
3865
+ * whole request up to {@link llmStreamMaxAttempts} times with a short backoff.
3866
+ * Every other outcome — success, a real API error, a timeout, an operator
3867
+ * cancellation — is returned from the first attempt untouched, preserving the
3868
+ * existing behaviour exactly.
3869
+ */
3870
+ async executeNative(system, messages, tools, callbacks, signal) {
3871
+ const maxAttempts = llmStreamMaxAttempts();
3872
+ let result;
3873
+ let attempt = 0;
3874
+ for (attempt = 1; attempt <= maxAttempts; attempt++) {
3875
+ result = await this.executeNativeAttempt(system, messages, tools, callbacks, signal);
3876
+ if (attempt >= maxAttempts || !shouldRetryNativeStream(result) || signal?.aborted)
3877
+ break;
3878
+ const backoff = streamRetryBackoffMs(attempt);
3879
+ diag.warn("stream_retry", `${this.providerLabel} stream ended without a final response \u2014 retrying (attempt ${attempt + 1}/${maxAttempts}) after ${backoff}ms`, { provider: this.providerLabel, attempt: attempt + 1, max_attempts: maxAttempts, backoff_ms: backoff });
3880
+ await delayWithAbort(backoff, signal);
3881
+ if (signal?.aborted)
3882
+ break;
3883
+ }
3884
+ if (attempt >= maxAttempts && maxAttempts > 1 && shouldRetryNativeStream(result)) {
3885
+ return { ...result, error: `${result.error} (retried ${maxAttempts} times)` };
3886
+ }
3887
+ return result;
3888
+ }
3889
+ async executeNativeAttempt(system, messages, tools, callbacks, signal) {
3890
+ await this.ensureHostedModel();
3891
+ const start = Date.now();
3892
+ if (!this.apiKey && this.provider !== "chatgpt-codex" && this.provider !== "google") {
3893
+ return {
3894
+ content: [{ type: "text", text: "" }],
3895
+ stopReason: "error",
3896
+ durationMs: Date.now() - start,
3897
+ error: this.noKeyError()
3898
+ };
3899
+ }
3900
+ if (signal?.aborted)
3901
+ return this.cancelledResult(start);
3902
+ const controller = new AbortController();
3903
+ const timer = setTimeout(() => controller.abort(), this.config.timeout || 12e4);
3904
+ const call = composeCallAbort(controller.signal, signal);
3905
+ try {
3906
+ let res;
3907
+ do {
3908
+ if (this.isOpenAICompat && this.wireApi === "chat_completions") {
3909
+ const chatMessages = [];
3910
+ chatMessages.push({ role: "system", content: system });
3911
+ for (const m of messages) {
3912
+ const pendingToolCalls = [];
3913
+ let pendingAssistantText = null;
3914
+ const flushAssistant = () => {
3915
+ if (pendingToolCalls.length === 0 && pendingAssistantText === null)
3916
+ return;
3917
+ const msg = { role: "assistant" };
3918
+ if (pendingAssistantText !== null)
3919
+ msg.content = pendingAssistantText;
3920
+ else
3921
+ msg.content = null;
3922
+ if (pendingToolCalls.length > 0)
3923
+ msg.tool_calls = pendingToolCalls.slice();
3924
+ chatMessages.push(msg);
3925
+ pendingToolCalls.length = 0;
3926
+ pendingAssistantText = null;
3927
+ };
3928
+ for (const block of m.content) {
3929
+ if (block.type === "text") {
3930
+ if (m.role === "assistant") {
3931
+ pendingAssistantText = (pendingAssistantText ?? "") + block.text;
3932
+ } else {
3933
+ flushAssistant();
3934
+ chatMessages.push({ role: m.role, content: block.text });
3935
+ }
3936
+ } else if (block.type === "tool_use") {
3937
+ pendingToolCalls.push({
3938
+ id: block.id,
3939
+ type: "function",
3940
+ function: { name: block.name, arguments: JSON.stringify(block.input) }
3941
+ });
3942
+ } else if (block.type === "tool_result") {
3943
+ flushAssistant();
3944
+ chatMessages.push({
3945
+ role: "tool",
3946
+ tool_call_id: block.tool_use_id,
3947
+ content: block.content
3948
+ });
3949
+ }
3950
+ }
3951
+ flushAssistant();
3952
+ }
3953
+ const body = {
3954
+ model: this.model,
3955
+ [this.maxTokensParamKey]: this.effectiveOutputTokens,
3956
+ messages: chatMessages
3957
+ };
3958
+ if (this.reasoningEffort) {
3959
+ body.reasoning_effort = this.reasoningEffort;
3960
+ }
3961
+ if (tools.length > 0) {
3962
+ body.tools = tools.map((t) => ({
3963
+ type: "function",
3964
+ function: {
3965
+ name: t.name,
3966
+ description: t.description,
3967
+ parameters: t.input_schema
3968
+ }
3969
+ }));
3970
+ }
3971
+ res = await this.postWithRetry(() => JSON.stringify({ ...body, model: this.model }), call.signal, call);
3972
+ } else if (this.isOpenAICompat && this.wireApi === "responses") {
3973
+ const isCodexProvider = this.provider === "chatgpt-codex";
3974
+ const input = isCodexProvider ? [] : [
3975
+ {
3976
+ role: "system",
3977
+ content: [{ type: "input_text", text: system }]
3978
+ }
3979
+ ];
3980
+ for (const m of messages) {
3981
+ if (features.retainedReasoning && m.role === "assistant" && m.providerRaw && m.providerRaw.provider === this.provider && m.providerRaw.model === this.model && m.providerRaw.wireApi === this.wireApi && m.providerRaw.output.length > 0) {
3982
+ input.push(...m.providerRaw.output);
3983
+ continue;
3984
+ }
3985
+ const assistantText = m.role === "assistant";
3986
+ const textType = assistantText ? "output_text" : "input_text";
3987
+ const textBlocks = [];
3988
+ for (const block of m.content) {
3989
+ if (block.type === "text") {
3990
+ textBlocks.push({ type: textType, text: block.text });
3991
+ } else if (block.type === "tool_use") {
3992
+ if (textBlocks.length > 0) {
3993
+ input.push({ role: m.role, content: [...textBlocks] });
3994
+ textBlocks.length = 0;
3995
+ }
3996
+ input.push({
3997
+ type: "function_call",
3998
+ call_id: block.id,
3999
+ name: block.name,
4000
+ arguments: JSON.stringify(block.input)
4001
+ });
4002
+ } else if (block.type === "tool_result") {
4003
+ if (textBlocks.length > 0) {
4004
+ input.push({ role: m.role, content: [...textBlocks] });
4005
+ textBlocks.length = 0;
4006
+ }
4007
+ input.push({
4008
+ type: "function_call_output",
4009
+ call_id: block.tool_use_id,
4010
+ output: block.content
4011
+ });
4012
+ }
4013
+ }
4014
+ if (textBlocks.length > 0) {
4015
+ input.push({ role: m.role, content: textBlocks });
4016
+ }
4017
+ }
4018
+ const reasoningEffort = this.reasoningEffort ?? defaultReasoningEffort(this.model);
4019
+ const isCodex = this.provider === "chatgpt-codex";
4020
+ const body = {
4021
+ model: this.model,
4022
+ input,
4023
+ ...isCodex ? { store: false, instructions: system } : { max_output_tokens: this.effectiveOutputTokens },
4024
+ ...reasoningEffort ? {
4025
+ reasoning: {
4026
+ effort: reasoningEffort,
4027
+ summary: "auto"
4028
+ },
4029
+ include: ["reasoning.encrypted_content"]
4030
+ } : {},
4031
+ // Server-side compaction, opt-in per runtime. ZDR-friendly: it works
4032
+ // with `store: false`, so nothing is retained server-side between
4033
+ // requests. Only the loops with no context strategy of their own ask
4034
+ // for it — the native loop compacts client-side and must not be
4035
+ // compacted twice.
4036
+ //
4037
+ // SHAPE IS LOAD-BEARING and was verified live against
4038
+ // chatgpt.com/backend-api/codex/responses, because this backend
4039
+ // rejects unknown and mis-typed body fields rather than ignoring
4040
+ // them (a bogus field returns
4041
+ // `400 Unsupported parameter: <name>`):
4042
+ // [{"type":"compaction","compact_threshold":N}] → 200
4043
+ // {"compaction":{"compact_threshold":N}} → 400 expected an
4044
+ // array of objects
4045
+ // [{"compaction":{...}}] → 400 missing
4046
+ // 'context_management[0].type'
4047
+ // [] → 400 minimum
4048
+ // length 1
4049
+ // The object form is what the public Responses docs show; it is not
4050
+ // what this backend takes. Never emit the key with an empty array —
4051
+ // that is a hard 400, hence the guard rather than a `.filter()`.
4052
+ //
4053
+ // Only the two stages that opt in send this, and they run on the
4054
+ // Codex backend. The shape is UNVERIFIED on plain OpenAI / Azure
4055
+ // Responses; if a caller ever enables it there, verify with a live
4056
+ // request before trusting it.
4057
+ ...this.serverCompactionTokens ? {
4058
+ context_management: [
4059
+ { type: "compaction", compact_threshold: this.serverCompactionTokens }
4060
+ ]
4061
+ } : {}
4062
+ };
4063
+ if (tools.length > 0) {
4064
+ body.tools = tools.map((t) => ({
4065
+ type: "function",
4066
+ name: t.name,
4067
+ description: t.description,
4068
+ // Codex backend's Responses API expects `strict` alongside
4069
+ // parameters. `false` keeps schema enforcement off so a model
4070
+ // that drifts on argument shape still emits the call instead
4071
+ // of failing it server-side. The public OpenAI Responses
4072
+ // schema tolerates the extra field.
4073
+ strict: false,
4074
+ parameters: t.input_schema
4075
+ }));
4076
+ if (isCodex) {
4077
+ body.tool_choice = "auto";
4078
+ body.parallel_tool_calls = true;
4079
+ }
4080
+ }
4081
+ res = await this.postWithRetry(() => JSON.stringify({ ...body, stream: true, model: this.model }), call.signal, call);
4082
+ if (!res) {
4083
+ await this.ensureHostedModel();
4084
+ continue;
4085
+ }
4086
+ if (!res.ok) {
4087
+ const responseText2 = await res.text();
4088
+ clearTimeout(timer);
4089
+ return {
4090
+ content: [{ type: "text", text: "" }],
4091
+ stopReason: "error",
4092
+ durationMs: Date.now() - start,
4093
+ error: `${this.providerLabel} API error ${res.status}: ${responseText2.slice(0, 500)}`
4094
+ };
4095
+ }
4096
+ const streamed = await this.consumeResponsesStream(res, start, callbacks, {
4097
+ idleTimeoutMs: llmStreamIdleTimeoutMs(),
4098
+ abort: call
4099
+ });
4100
+ clearTimeout(timer);
4101
+ return streamed;
4102
+ } else if (this.isGeminiCodeAssist) {
4103
+ const request = {
4104
+ ...system ? { systemInstruction: { parts: [{ text: system }] } } : {},
4105
+ contents: this.googleContents(messages),
4106
+ generationConfig: { maxOutputTokens: NATIVE_COMPLETION_TOKEN_LIMIT }
4107
+ };
4108
+ if (tools.length > 0) {
4109
+ request.tools = [{
4110
+ functionDeclarations: tools.map((tool) => ({
4111
+ name: tool.name,
4112
+ description: tool.description,
4113
+ parametersJsonSchema: tool.input_schema
4114
+ }))
4115
+ }];
4116
+ }
4117
+ const geminiBody = await this.wrapGeminiCodeAssistBody(request);
4118
+ res = await this.postWithRetry(() => JSON.stringify(geminiBody), call.signal, call);
4119
+ } else if (this.isGoogleWire) {
4120
+ const body = {
4121
+ ...system ? { systemInstruction: { parts: [{ text: system }] } } : {},
4122
+ contents: this.googleContents(messages),
4123
+ generationConfig: { maxOutputTokens: NATIVE_COMPLETION_TOKEN_LIMIT }
4124
+ };
4125
+ if (tools.length > 0) {
4126
+ body.tools = [{
4127
+ functionDeclarations: tools.map((tool) => ({
4128
+ name: tool.name,
4129
+ description: tool.description,
4130
+ parametersJsonSchema: tool.input_schema
4131
+ }))
4132
+ }];
4133
+ }
4134
+ res = await this.postWithRetry(() => JSON.stringify(body), call.signal, call);
4135
+ } else if (this.isAnthropicWire) {
4136
+ const replayedRawMessageIndexes = /* @__PURE__ */ new Set();
4137
+ const apiMessages = messages.map((m, index) => {
4138
+ if (features.retainedReasoning && m.role === "assistant" && m.providerRaw && m.providerRaw.provider === this.provider && m.providerRaw.model === this.model && m.providerRaw.wireApi === this.wireApi && m.providerRaw.output.length > 0 && isWireBlockArray(m.providerRaw.output)) {
4139
+ replayedRawMessageIndexes.add(index);
4140
+ return { role: m.role, content: m.providerRaw.output };
4141
+ }
4142
+ return {
4143
+ role: m.role,
4144
+ content: m.content.map((block) => {
4145
+ if (block.type === "text")
4146
+ return { type: "text", text: block.text };
4147
+ if (block.type === "tool_use") {
4148
+ return { type: "tool_use", id: block.id, name: block.name, input: block.input };
4149
+ }
4150
+ if (block.type === "tool_result") {
4151
+ return {
4152
+ type: "tool_result",
4153
+ tool_use_id: block.tool_use_id,
4154
+ content: block.content,
4155
+ ...block.is_error ? { is_error: true } : {}
4156
+ };
4157
+ }
4158
+ return block;
4159
+ })
4160
+ };
4161
+ });
4162
+ const cacheEnabled = features.promptCache && providerSupportsPromptCache(this.provider);
4163
+ for (const index of cacheEnabled ? planMessageBreakpoints(apiMessages, MESSAGE_CACHE_BREAKPOINTS) : []) {
4164
+ if (replayedRawMessageIndexes.has(index))
4165
+ continue;
4166
+ const blocks = apiMessages[index]?.content;
4167
+ const lastBlock = blocks?.length ? blocks[blocks.length - 1] : void 0;
4168
+ if (blocks && lastBlock)
4169
+ blocks[blocks.length - 1] = withCacheControl(lastBlock);
4170
+ }
4171
+ const body = {
4172
+ model: this.model,
4173
+ max_tokens: NATIVE_COMPLETION_TOKEN_LIMIT,
4174
+ ...this.anthropicThinkingField(),
4175
+ // The remaining breakpoint goes on the system prompt. Because the
4176
+ // wire renders `tools` → `system` → `messages`, one marker here
4177
+ // caches the tool schemas AND the system prompt together — the
4178
+ // largest, most static span in the request, and the one that never
4179
+ // changes for the lifetime of an agent session. Sent as a block array
4180
+ // (the only shape that accepts `cache_control`) when caching is on,
4181
+ // and left as a plain string otherwise so non-caching providers see a
4182
+ // byte-identical body to before this change.
4183
+ system: cacheEnabled ? [withCacheControl({ type: "text", text: system })] : system,
4184
+ messages: apiMessages
4185
+ };
4186
+ if (tools.length > 0) {
4187
+ body.tools = tools;
4188
+ }
4189
+ res = await this.postWithRetry(() => JSON.stringify({ ...body, model: this.model }), call.signal, call);
4190
+ } else {
4191
+ throw new Error(`executeNative: provider ${this.provider} is not mapped to a wire`);
4192
+ }
4193
+ if (!res)
4194
+ await this.ensureHostedModel();
4195
+ } while (!res);
4196
+ const responseText = await res.text();
4197
+ clearTimeout(timer);
4198
+ if (!res.ok) {
4199
+ return {
4200
+ content: [{ type: "text", text: "" }],
4201
+ stopReason: "error",
4202
+ durationMs: Date.now() - start,
4203
+ error: `${this.providerLabel} API error ${res.status}: ${responseText.slice(0, 500)}`
4204
+ };
4205
+ }
4206
+ const parsedResponse = JSON.parse(responseText);
4207
+ const json = this.isGeminiCodeAssist ? parsedResponse.response ?? parsedResponse : parsedResponse;
4208
+ appendNativeTrace({
4209
+ kind: "native-response",
4210
+ provider: this.providerLabel,
4211
+ wireApi: this.wireApi,
4212
+ usage: json.usage ?? null,
4213
+ outputPreview: Array.isArray(json.output) ? json.output.slice(0, 10).map((item) => ({
4214
+ type: item.type,
4215
+ summary: item.summary,
4216
+ content: item.content,
4217
+ name: item.name
4218
+ })) : null,
4219
+ topLevelKeys: Object.keys(json)
4220
+ });
4221
+ let content;
4222
+ let stopReason;
4223
+ let usage;
4224
+ let providerRaw;
4225
+ if (this.isOpenAICompat && this.wireApi === "chat_completions") {
4226
+ const choice = json.choices?.[0];
4227
+ const msg = choice?.message;
4228
+ content = [];
4229
+ const textContent = msg?.content ?? msg?.reasoning;
4230
+ if (textContent) {
4231
+ content.push({ type: "text", text: textContent });
4232
+ }
4233
+ if (msg?.tool_calls) {
4234
+ for (const tc of msg.tool_calls) {
4235
+ content.push({
4236
+ type: "tool_use",
4237
+ id: tc.id,
4238
+ name: tc.function.name,
4239
+ input: safeParseJson(tc.function.arguments)
4240
+ });
4241
+ }
4242
+ }
4243
+ const finishReason = choice?.finish_reason;
4244
+ stopReason = finishReason === "tool_calls" || finishReason === "function_call" ? "tool_use" : finishReason === "length" ? "max_tokens" : "end_turn";
4245
+ if (json.usage) {
4246
+ usage = {
4247
+ inputTokens: json.usage.prompt_tokens ?? 0,
4248
+ outputTokens: json.usage.completion_tokens ?? 0
4249
+ };
4250
+ }
4251
+ } else if (this.isOpenAICompat && this.wireApi === "responses") {
4252
+ content = [];
4253
+ const reasoningSummaries = [];
4254
+ const rawOutput = json.output ?? [];
4255
+ if (rawOutput.length > 0) {
4256
+ providerRaw = {
4257
+ provider: this.provider,
4258
+ model: this.model,
4259
+ wireApi: this.wireApi,
4260
+ output: rawOutput
4261
+ };
4262
+ }
4263
+ for (const item of json.output ?? []) {
4264
+ if (item.type === "function_call") {
4265
+ content.push({
4266
+ type: "tool_use",
4267
+ id: item.call_id,
4268
+ name: item.name,
4269
+ input: safeParseJson(item.arguments)
4270
+ });
4271
+ continue;
4272
+ }
4273
+ if (item.type === "reasoning") {
4274
+ const summaryParts = Array.isArray(item.summary) ? item.summary.map((block) => typeof block.text === "string" ? block.text : "").filter((text) => text.trim().length > 0) : [];
4275
+ const reasoningText = summaryParts.join("\n").trim();
4276
+ if (reasoningText)
4277
+ reasoningSummaries.push(reasoningText);
4278
+ continue;
4279
+ }
4280
+ for (const block of item.content ?? []) {
4281
+ if (block.type === "output_text") {
4282
+ content.push({ type: "text", text: block.text });
4283
+ } else if (block.type === "summary_text" || block.type === "reasoning_text") {
4284
+ const text = typeof block.text === "string" ? block.text : "";
4285
+ if (text.trim())
4286
+ reasoningSummaries.push(text);
4287
+ }
4288
+ }
4289
+ }
4290
+ if (callbacks?.onThinking && reasoningSummaries.length > 0) {
4291
+ callbacks.onThinking(reasoningSummaries.join("\n"));
4292
+ }
4293
+ stopReason = content.some((block) => block.type === "tool_use") ? "tool_use" : "end_turn";
4294
+ if (json.usage) {
4295
+ usage = {
4296
+ inputTokens: json.usage.input_tokens ?? 0,
4297
+ outputTokens: json.usage.output_tokens ?? 0,
4298
+ // Responses `input_tokens` already includes the cached span, so
4299
+ // this is instrumentation only — see the streaming path.
4300
+ ...readResponsesCachedTokens(json.usage)
4301
+ };
4302
+ }
4303
+ } else if (this.isGoogleWire || this.isGeminiCodeAssist) {
4304
+ const candidate = json.candidates?.[0];
4305
+ const rawParts = Array.isArray(candidate?.content?.parts) ? candidate.content.parts : [];
4306
+ content = [];
4307
+ if (rawParts.length > 0) {
4308
+ providerRaw = {
4309
+ provider: this.provider,
4310
+ model: this.model,
4311
+ wireApi: this.wireApi,
4312
+ output: rawParts
4313
+ };
4314
+ }
4315
+ for (const part of rawParts) {
4316
+ if (typeof part.text === "string") {
4317
+ content.push({ type: "text", text: part.text });
4318
+ continue;
4319
+ }
4320
+ const call2 = part.functionCall && typeof part.functionCall === "object" ? part.functionCall : void 0;
4321
+ if (call2 && typeof call2.name === "string") {
4322
+ content.push({
4323
+ type: "tool_use",
4324
+ id: typeof call2.id === "string" ? call2.id : `google-call-${++googleFunctionCallSequence}`,
4325
+ name: call2.name,
4326
+ input: call2.args && typeof call2.args === "object" && !Array.isArray(call2.args) ? call2.args : {}
4327
+ });
4328
+ }
4329
+ }
4330
+ const finishReason = String(candidate?.finishReason ?? "").toUpperCase();
4331
+ stopReason = content.some((block) => block.type === "tool_use") ? "tool_use" : finishReason === "MAX_TOKENS" ? "max_tokens" : "end_turn";
4332
+ if (json.usageMetadata) {
4333
+ usage = {
4334
+ inputTokens: json.usageMetadata.promptTokenCount ?? 0,
4335
+ outputTokens: json.usageMetadata.candidatesTokenCount ?? 0
4336
+ };
4337
+ }
4338
+ } else {
4339
+ const rawBlocks = json.content ?? [];
4340
+ const hasAnthropicThinking = rawBlocks.some((block) => block.type === "thinking" || block.type === "redacted_thinking");
4341
+ if (features.retainedReasoning && this.provider === "anthropic" && hasAnthropicThinking) {
4342
+ providerRaw = {
4343
+ provider: this.provider,
4344
+ model: this.model,
4345
+ wireApi: this.wireApi,
4346
+ output: rawBlocks
4347
+ };
4348
+ }
4349
+ if (callbacks?.onThinking) {
4350
+ const thinkingText = rawBlocks.filter((b) => b.type === "thinking").map((b) => typeof b.thinking === "string" ? b.thinking : "").join("").trim();
4351
+ if (thinkingText)
4352
+ callbacks.onThinking(thinkingText);
4353
+ }
4354
+ content = rawBlocks.filter((block) => block.type !== "thinking" && block.type !== "redacted_thinking").map((block) => {
4355
+ if (block.type === "text") {
4356
+ return { type: "text", text: block.text };
4357
+ }
4358
+ if (block.type === "tool_use") {
4359
+ return {
4360
+ type: "tool_use",
4361
+ id: block.id,
4362
+ name: block.name,
4363
+ input: block.input
4364
+ };
4365
+ }
4366
+ return { type: "text", text: JSON.stringify(block) };
4367
+ });
4368
+ stopReason = json.stop_reason === "tool_use" ? "tool_use" : json.stop_reason === "max_tokens" ? "max_tokens" : "end_turn";
4369
+ if (json.usage) {
4370
+ usage = readCacheUsage(json.usage);
4371
+ this.logCacheUsage(usage);
4372
+ }
4373
+ }
4374
+ if (usage)
4375
+ callbacks?.onUsage?.(usage);
4376
+ return {
4377
+ content,
4378
+ stopReason,
4379
+ usage,
4380
+ durationMs: Date.now() - start,
4381
+ ...providerRaw ? { providerRaw } : {}
4382
+ };
4383
+ } catch (err) {
4384
+ clearTimeout(timer);
4385
+ if (err instanceof OperatorAbortError || call.operatorAborted()) {
4386
+ return this.cancelledResult(start);
4387
+ }
4388
+ if (err instanceof QuotaExhaustedError) {
4389
+ return {
4390
+ content: [{ type: "text", text: "" }],
4391
+ stopReason: "error",
4392
+ durationMs: Date.now() - start,
4393
+ error: err.message
4394
+ };
4395
+ }
4396
+ const msg = err instanceof Error ? err.message : String(err);
4397
+ const timedOut = msg.includes("abort") || msg.includes("timeout");
4398
+ return {
4399
+ content: [{ type: "text", text: "" }],
4400
+ stopReason: "error",
4401
+ durationMs: Date.now() - start,
4402
+ error: timedOut ? `${this.providerLabel} API request timed out` : `${this.providerLabel} API error: ${msg}`
4403
+ };
4404
+ } finally {
4405
+ call.dispose();
4406
+ }
4407
+ }
4408
+ async consumeResponsesStream(res, start, callbacks, opts) {
4409
+ const reader = res.body?.getReader();
4410
+ if (!reader) {
4411
+ return {
4412
+ content: [{ type: "text", text: "" }],
4413
+ stopReason: "error",
4414
+ durationMs: Date.now() - start,
4415
+ error: `${this.providerLabel} API error: missing response body`
4416
+ };
4417
+ }
4418
+ const idleTimeoutMs = opts?.idleTimeoutMs ?? llmStreamIdleTimeoutMs();
4419
+ const operatorSignal = opts?.abort?.operator;
4420
+ let stalled = false;
4421
+ const readBounded = async () => {
4422
+ let timer;
4423
+ const detach = operatorSignal ? new AbortController() : void 0;
4424
+ try {
4425
+ return await Promise.race([
4426
+ reader.read(),
4427
+ new Promise((_resolve, reject) => {
4428
+ timer = setTimeout(() => {
4429
+ stalled = true;
4430
+ reject(new Error("stream stalled"));
4431
+ }, idleTimeoutMs);
4432
+ }),
4433
+ // A real aborted `fetch` also errors the body stream, so `read()`
4434
+ // would reject on its own — but only for a live socket. This racer
4435
+ // is what makes cancellation immediate and unconditional, including
4436
+ // for a body that is buffered, mocked, or already fully delivered.
4437
+ ...operatorSignal && detach ? [
4438
+ new Promise((_resolve, reject) => {
4439
+ if (operatorSignal.aborted) {
4440
+ reject(new OperatorAbortError());
4441
+ return;
4442
+ }
4443
+ operatorSignal.addEventListener("abort", () => reject(new OperatorAbortError()), { once: true, signal: detach.signal });
4444
+ })
4445
+ ] : []
4446
+ ]);
4447
+ } finally {
4448
+ if (timer)
4449
+ clearTimeout(timer);
4450
+ detach?.abort();
4451
+ }
4452
+ };
4453
+ const decoder = new TextDecoder();
4454
+ let buffer = "";
4455
+ let completedResponse = null;
4456
+ let openRouterStreamFailed = false;
4457
+ let openRouterUsage;
4458
+ const streamedOutputItems = [];
4459
+ let thinkingText = "";
4460
+ let lastThinkingEmit = 0;
4461
+ let lastThinkingLength = 0;
4462
+ const emitThinking = (force = false) => {
4463
+ if (!callbacks?.onThinking || !thinkingText.trim())
4464
+ return;
4465
+ if (force && lastThinkingEmit > 0 && lastThinkingLength === thinkingText.length)
4466
+ return;
4467
+ const now = Date.now();
4468
+ const nextChars = thinkingText.length - lastThinkingLength;
4469
+ const firstEmit = lastThinkingLength === 0;
4470
+ if (!force) {
4471
+ if (firstEmit && thinkingText.length < 96)
4472
+ return;
4473
+ if (nextChars < 96 && now - lastThinkingEmit < 250)
4474
+ return;
4475
+ }
4476
+ lastThinkingEmit = now;
4477
+ lastThinkingLength = thinkingText.length;
4478
+ callbacks.onThinking(thinkingText);
4479
+ };
4480
+ while (true) {
4481
+ let chunk;
4482
+ try {
4483
+ chunk = await readBounded();
4484
+ } catch (err) {
4485
+ if (err instanceof OperatorAbortError || opts?.abort?.operatorAborted()) {
4486
+ try {
4487
+ await reader.cancel();
4488
+ } catch {
4489
+ }
4490
+ throw err instanceof OperatorAbortError ? err : new OperatorAbortError();
4491
+ }
4492
+ if (stalled) {
4493
+ try {
4494
+ await reader.cancel();
4495
+ } catch {
4496
+ }
4497
+ const secs = Math.round(idleTimeoutMs / 1e3);
4498
+ diag.warn("stream_stalled", `${this.providerLabel} stream stalled \u2014 no SSE events for ${secs}s (server hold; aborting call)`, {
4499
+ provider: this.providerLabel,
4500
+ idle_timeout_ms: idleTimeoutMs,
4501
+ idle_timeout_s: secs
4502
+ });
4503
+ return {
4504
+ content: [{ type: "text", text: "" }],
4505
+ stopReason: "error",
4506
+ durationMs: Date.now() - start,
4507
+ error: `${this.providerLabel} stream stalled \u2014 no SSE events for ${secs}s (server accepted but held the stream; transient)`
4508
+ };
4509
+ }
4510
+ throw err;
4511
+ }
4512
+ const { done, value } = chunk;
4513
+ if (done)
4514
+ break;
4515
+ buffer += decoder.decode(value, { stream: true });
4516
+ let boundary = buffer.indexOf("\n\n");
4517
+ while (boundary >= 0) {
4518
+ const rawChunk = buffer.slice(0, boundary);
4519
+ buffer = buffer.slice(boundary + 2);
4520
+ boundary = buffer.indexOf("\n\n");
4521
+ const payload = rawChunk.split("\n").filter((line) => line.startsWith("data:")).map((line) => line.slice(5).trim()).join("\n");
4522
+ if (!payload || payload === "[DONE]")
4523
+ continue;
4524
+ let event;
4525
+ try {
4526
+ event = JSON.parse(payload);
4527
+ } catch {
4528
+ continue;
4529
+ }
4530
+ const type = String(event.type ?? "");
4531
+ if (this.provider === "openrouter" && (type === "response.done" || type === "response.failed" || type === "response.completed" || type === "response.incomplete")) {
4532
+ const response = event.response;
4533
+ const usage2 = response?.usage;
4534
+ if (usage2 && typeof usage2.input_tokens === "number" && Number.isFinite(usage2.input_tokens) && usage2.input_tokens >= 0 && typeof usage2.output_tokens === "number" && Number.isFinite(usage2.output_tokens) && usage2.output_tokens >= 0) {
4535
+ openRouterUsage = {
4536
+ inputTokens: usage2.input_tokens,
4537
+ outputTokens: usage2.output_tokens,
4538
+ ...readResponsesCachedTokens(usage2)
4539
+ };
4540
+ callbacks?.onUsage?.(openRouterUsage);
4541
+ }
4542
+ }
4543
+ if (this.provider === "openrouter" && (type === "error" || type === "response.failed" || type === "response.incomplete")) {
4544
+ openRouterStreamFailed = true;
4545
+ continue;
4546
+ }
4547
+ if (type === "response.output_text.delta" || this.provider === "openrouter" && type === "response.content_part.delta") {
4548
+ const delta = typeof event.delta === "string" ? event.delta : "";
4549
+ if (delta) {
4550
+ callbacks?.onDelta?.("assistant_response", delta);
4551
+ }
4552
+ continue;
4553
+ }
4554
+ if (type === "response.reasoning_summary_text.delta") {
4555
+ const delta = typeof event.delta === "string" ? event.delta : "";
4556
+ if (delta) {
4557
+ thinkingText += delta;
4558
+ callbacks?.onDelta?.("reasoning", delta);
4559
+ emitThinking(false);
4560
+ }
4561
+ continue;
4562
+ }
4563
+ if (type === "response.reasoning_summary_text.done") {
4564
+ const text = typeof event.text === "string" ? event.text : typeof event.part === "object" && event.part && typeof event.part.text === "string" ? String(event.part.text) : "";
4565
+ if (text.trim()) {
4566
+ thinkingText = text;
4567
+ emitThinking(true);
4568
+ }
4569
+ continue;
4570
+ }
4571
+ if (type === "response.output_item.done") {
4572
+ const item = event.item;
4573
+ if (item && typeof item.type === "string") {
4574
+ streamedOutputItems.push(item);
4575
+ }
4576
+ continue;
4577
+ }
4578
+ if (type === "response.completed" || type === "response.incomplete" || this.provider === "openrouter" && type === "response.done") {
4579
+ const response = event.response;
4580
+ if (response) {
4581
+ if (this.provider === "openrouter" && (openRouterStreamFailed || response.error != null || (type === "response.done" || response.status !== void 0) && response.status !== "completed")) {
4582
+ openRouterStreamFailed = true;
4583
+ continue;
4584
+ }
4585
+ completedResponse = response;
4586
+ const usage2 = response.usage;
4587
+ if (usage2 && this.provider !== "openrouter") {
4588
+ callbacks?.onUsage?.({
4589
+ inputTokens: Number(usage2.input_tokens ?? 0),
4590
+ outputTokens: Number(usage2.output_tokens ?? 0)
4591
+ });
4592
+ }
4593
+ }
4594
+ }
4595
+ }
4596
+ }
4597
+ emitThinking(true);
4598
+ if (!completedResponse || openRouterStreamFailed) {
4599
+ return {
4600
+ content: thinkingText ? [{ type: "text", text: thinkingText }] : [{ type: "text", text: "" }],
4601
+ stopReason: "error",
4602
+ durationMs: Date.now() - start,
4603
+ ...openRouterUsage ? { usage: openRouterUsage } : {},
4604
+ error: `${this.providerLabel} API error: ${openRouterStreamFailed ? "response stream failed" : "stream completed without final response"}`
4605
+ };
4606
+ }
4607
+ appendNativeTrace({
4608
+ kind: "native-response-stream",
4609
+ provider: this.providerLabel,
4610
+ wireApi: this.wireApi,
4611
+ usage: completedResponse.usage ?? null,
4612
+ outputPreview: Array.isArray(completedResponse.output) ? completedResponse.output.slice(0, 10).map((item) => ({
4613
+ type: item.type,
4614
+ summary: item.summary,
4615
+ content: item.content,
4616
+ name: item.name
4617
+ })) : null,
4618
+ streamedItems: streamedOutputItems.slice(0, 10).map((item) => ({
4619
+ type: item.type,
4620
+ name: item.name,
4621
+ call_id: item.call_id,
4622
+ argumentsPreview: typeof item.arguments === "string" ? item.arguments.slice(0, 200) : void 0
4623
+ })),
4624
+ topLevelKeys: Object.keys(completedResponse)
4625
+ });
4626
+ const completedOutput = completedResponse.output ?? [];
4627
+ const outputItems = streamedOutputItems.length > 0 ? streamedOutputItems : completedOutput;
4628
+ const content = [];
4629
+ const reasoningSummaries = [];
4630
+ for (const item of outputItems) {
4631
+ if (item.type === "function_call") {
4632
+ content.push({
4633
+ type: "tool_use",
4634
+ id: String(item.call_id),
4635
+ name: String(item.name),
4636
+ input: safeParseJson(String(item.arguments ?? "{}"))
4637
+ });
4638
+ continue;
4639
+ }
4640
+ if (item.type === "reasoning") {
4641
+ const summaryParts = Array.isArray(item.summary) ? item.summary.map((block) => typeof block.text === "string" ? block.text : "").filter((text) => text.trim().length > 0) : [];
4642
+ const reasoningText = summaryParts.join("\n").trim();
4643
+ if (reasoningText)
4644
+ reasoningSummaries.push(reasoningText);
4645
+ continue;
4646
+ }
4647
+ for (const block of item.content ?? []) {
4648
+ if (block.type === "output_text") {
4649
+ content.push({ type: "text", text: String(block.text ?? "") });
4650
+ }
4651
+ }
4652
+ }
4653
+ if (lastThinkingEmit === 0 && reasoningSummaries.length > 0) {
4654
+ thinkingText = reasoningSummaries.join("\n");
4655
+ emitThinking(true);
4656
+ }
4657
+ const usageRecord = completedResponse.usage;
4658
+ const usage = usageRecord ? {
4659
+ inputTokens: Number(usageRecord.input_tokens ?? 0),
4660
+ outputTokens: Number(usageRecord.output_tokens ?? 0),
4661
+ // Responses `input_tokens` already INCLUDES the cached span (unlike
4662
+ // Anthropic, which subtracts it), so no normalisation is needed —
4663
+ // this is purely so cache behaviour becomes observable. Without it
4664
+ // the Codex cache hit rate is unmeasurable: `prompt-cache.ts`
4665
+ // instruments the Anthropic path only.
4666
+ ...readResponsesCachedTokens(usageRecord)
4667
+ } : void 0;
4668
+ return {
4669
+ content,
4670
+ stopReason: content.some((item) => item.type === "tool_use") ? "tool_use" : "end_turn",
4671
+ usage,
4672
+ durationMs: Date.now() - start,
4673
+ // `outputItems` is the complete, correctly-ordered response array —
4674
+ // reasoning items with their `encrypted_content` still attached, each
4675
+ // immediately followed by the item it produced. Handing it back lets the
4676
+ // next turn replay it verbatim instead of re-deriving the reasoning.
4677
+ ...outputItems.length > 0 ? {
4678
+ providerRaw: {
4679
+ provider: this.provider,
4680
+ model: this.model,
4681
+ wireApi: this.wireApi,
4682
+ output: outputItems
4683
+ }
4684
+ } : {}
4685
+ };
4686
+ }
4687
+ async isAvailable() {
4688
+ if (this.provider === "chatgpt-codex") {
4689
+ return this.codexAuthState !== void 0;
4690
+ }
4691
+ return !!this.apiKey;
4692
+ }
4693
+ };
4694
+
4695
+ export {
4696
+ FEATURE_PRESETS,
4697
+ resolveFeaturePreset,
4698
+ applyFeaturePreset,
4699
+ applyFeaturePresetFromEnv,
4700
+ features,
4701
+ MAX_MESSAGE_LENGTH,
4702
+ MAX_FIELD_VALUE_LENGTH,
4703
+ MAX_CODE_LENGTH,
4704
+ MAX_FIELDS,
4705
+ MAX_BUFFERED,
4706
+ formatDiagnosticLine,
4707
+ stderrDiagnosticSink,
4708
+ diag,
4709
+ claimDiagnostics,
4710
+ subscribeDiagnostics,
4711
+ isDiagnosticsClaimed,
4712
+ recentDiagnostics,
4713
+ _resetDiagnosticsForTests,
4714
+ DEFAULT_CLOUD_HOST,
4715
+ CloudAuthMissingError,
4716
+ CloudAuthError,
4717
+ loadCloudCredentials,
4718
+ CloudError,
4719
+ CloudUnauthorizedError,
4720
+ CloudForbiddenError,
4721
+ CloudNetworkError,
4722
+ CloudClient,
4723
+ QuotaExhaustedError,
4724
+ OperatorAbortError,
4725
+ parseUsageLimitReached,
4726
+ LOOP_SERVER_COMPACTION_TOKENS,
4727
+ LlmApiRuntime
4728
+ };