@young1lin/dsh-gpt-sub 0.1.0 → 0.1.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@young1lin/dsh-gpt-sub",
3
- "version": "0.1.0",
3
+ "version": "0.1.2",
4
4
  "publishConfig": {
5
5
  "access": "public"
6
6
  },
@@ -21,7 +21,11 @@
21
21
  "default": "./lib/index.js"
22
22
  },
23
23
  "./package.json": "./package.json",
24
- "./client": "./client.js"
24
+ "./client": "./client.js",
25
+ "./compaction": {
26
+ "types": "./lib/compaction.d.ts",
27
+ "default": "./lib/compaction.js"
28
+ }
25
29
  },
26
30
  "files": [
27
31
  "lib",
@@ -55,17 +59,40 @@
55
59
  ],
56
60
  "license": "MIT",
57
61
  "dependencies": {
58
- "@deepseek-ai/schemastery": "^3.18.1",
59
- "undici": "^7.10.0"
62
+ "@deepseek-ai/schemastery": "^3.18.5-alpha.1",
63
+ "undici": "^7.10.0",
64
+ "@earendil-works/pi-ai": "^0.87.1",
65
+ "@deepseek-ai/dsh-llm-pi-ai": "^0.2.1-alpha.1",
66
+ "eventsource-parser": "^3.0.6",
67
+ "zod": "4.4.3"
60
68
  },
61
69
  "peerDependencies": {
62
- "@deepseek-ai/cordis": "^4.0.1-rc.1",
63
- "@deepseek-ai/dsh-credentials": "^0.1.0-rc.2"
70
+ "@deepseek-ai/cordis": "^4.0.5-alpha.1",
71
+ "@deepseek-ai/dsh-credentials": "^0.2.1-alpha.1",
72
+ "@deepseek-ai/dsh-agent": "^0.2.1-alpha.1",
73
+ "@deepseek-ai/dsh-commands": "^0.2.1-alpha.1",
74
+ "@deepseek-ai/dsh-compaction": "^0.2.1-alpha.1",
75
+ "@deepseek-ai/dsh-compaction-basic": "^0.2.1-alpha.1",
76
+ "@deepseek-ai/dsh-llm": "^0.2.1-alpha.1",
77
+ "@deepseek-ai/dsh-session": "^0.2.1-alpha.1",
78
+ "@deepseek-ai/dsh-session-projection": "^0.2.1-alpha.1",
79
+ "@deepseek-ai/dsh-token-meter": "^0.2.1-alpha.1"
64
80
  },
65
81
  "devDependencies": {
66
- "@deepseek-ai/cordis": "^4.0.1-rc.1",
67
- "@deepseek-ai/dsh-credentials": "^0.1.0-rc.2",
68
- "@deepseek-ai/dsh-host-webserver": "0.1.0-rc.7",
82
+ "@deepseek-ai/cordis": "^4.0.5-alpha.1",
83
+ "@deepseek-ai/schemastery": "^3.18.5-alpha.1",
84
+ "@deepseek-ai/dsh-agent-loop": "^0.2.1-alpha.1",
85
+ "@deepseek-ai/dsh-agent-loop-testkit": "^0.2.1-alpha.1",
86
+ "@deepseek-ai/dsh-credentials": "^0.2.1-alpha.1",
87
+ "@deepseek-ai/dsh-host-webserver": "^0.2.1-alpha.1",
88
+ "@deepseek-ai/dsh-agent": "^0.2.1-alpha.1",
89
+ "@deepseek-ai/dsh-commands": "^0.2.1-alpha.1",
90
+ "@deepseek-ai/dsh-compaction": "^0.2.1-alpha.1",
91
+ "@deepseek-ai/dsh-compaction-basic": "^0.2.1-alpha.1",
92
+ "@deepseek-ai/dsh-llm": "^0.2.1-alpha.1",
93
+ "@deepseek-ai/dsh-session": "^0.2.1-alpha.1",
94
+ "@deepseek-ai/dsh-session-projection": "^0.2.1-alpha.1",
95
+ "@deepseek-ai/dsh-token-meter": "^0.2.1-alpha.1",
69
96
  "@types/node": "^22.10.0",
70
97
  "typescript": "^5.9.0",
71
98
  "vitest": "^3.2.0"
@@ -0,0 +1,52 @@
1
+ /** Durable compaction locking and native token deltas for restored Sessions. */
2
+ import type { ProjectionDefinition } from '@deepseek-ai/dsh-session-projection'
3
+ import { z } from 'zod'
4
+ import { readCheckpoint } from './native-history.ts'
5
+
6
+ const stateSchema = z.object({
7
+ systemHead: z.boolean().nullable(),
8
+ openTurn: z.number().nullable(),
9
+ activeCompaction: z.string().nullable(),
10
+ checkpointSeq: z.number().nullable(),
11
+ nativeExtraTokens: z.number(),
12
+ lastAssistantExtraTokens: z.number(),
13
+ })
14
+
15
+ /** Compaction lifecycle and pressure facts folded from the Session log. */
16
+ export type NativeCompactionState = z.infer<typeof stateSchema>
17
+
18
+ declare module '@deepseek-ai/dsh-session-projection/types' {
19
+ interface SessionProjectionStateMap { gptSubCompaction: NativeCompactionState }
20
+ }
21
+
22
+ /** Pure fold used by both live compaction and Session restoration. */
23
+ export const nativeCompactionProjection: ProjectionDefinition<'gptSubCompaction'> = {
24
+ key: 'gptSubCompaction', stateSchema, stateVersion: 1,
25
+ init: () => ({ systemHead: null, openTurn: null, activeCompaction: null, checkpointSeq: null, nativeExtraTokens: 0, lastAssistantExtraTokens: 0 }),
26
+ apply(state, event) {
27
+ if (state.systemHead === null && 'surfaceOp' in event) state = { ...state, systemHead: event.type === 'system/message' }
28
+ switch (event.type) {
29
+ case 'turn/start': return { ...state, openTurn: event.data.turn }
30
+ case 'turn/end': return { ...state, openTurn: null }
31
+ case 'session/end-seed': return { ...state, openTurn: null, activeCompaction: null }
32
+ case 'compaction/start': return { ...state, activeCompaction: event.data.compactionId }
33
+ case 'compaction/end': return { ...state, activeCompaction: null }
34
+ case 'assistant/message': return { ...state, lastAssistantExtraTokens: state.nativeExtraTokens }
35
+ case 'user/message': {
36
+ const checkpoint = readCheckpoint(event.data.source)
37
+ if (checkpoint !== undefined) {
38
+ const text = event.data.content.filter(block => block.type === 'text').map(block => block.text).join('')
39
+ return {
40
+ ...state, checkpointSeq: event.seq,
41
+ nativeExtraTokens: checkpoint.contextTokens - (Math.ceil(text.length / 4) + 8),
42
+ }
43
+ }
44
+ if (state.checkpointSeq !== null && event.sourceEventSeqs?.some(seq => seq === state.checkpointSeq)) {
45
+ return { ...state, checkpointSeq: null, nativeExtraTokens: 0 }
46
+ }
47
+ return state
48
+ }
49
+ default: return state
50
+ }
51
+ },
52
+ }
@@ -0,0 +1,188 @@
1
+ /** Routes Codex Subscription compaction to native Responses checkpoints. */
2
+ import { randomUUID } from 'node:crypto'
3
+ import { isDeepStrictEqual } from 'node:util'
4
+ import type { Context } from '@deepseek-ai/cordis'
5
+ import type { Agent } from '@deepseek-ai/dsh-agent'
6
+ import type { CommandId } from '@deepseek-ai/dsh-commands/brand'
7
+ import {
8
+ CompactionId, compactCheckpointSource, ManualCompactionError,
9
+ toolPairingBalancedAfter, toolPairingBalancedBefore,
10
+ type CompactionResult, type CompactionTrigger,
11
+ } from '@deepseek-ai/dsh-compaction'
12
+ import { BasicCompactionEngine, type BasicCompactionConfig } from '@deepseek-ai/dsh-compaction-basic'
13
+ import { BlockAssembler, createUserMessage, type Message } from '@deepseek-ai/dsh-llm'
14
+ import type { Session, SessionSeq } from '@deepseek-ai/dsh-session'
15
+ import type {} from '@deepseek-ai/dsh-token-meter'
16
+ import { hasPiAiNativeStream } from './native-bridge.ts'
17
+ import { nativeCompactionProjection, type NativeCompactionState } from './compaction-state.ts'
18
+ import { nativeItemTokens, nativeResult, type CodexCheckpointSource } from './native-history.ts'
19
+
20
+ /** Compaction backend selected by the sub bundle. Other routes retain basic compaction. */
21
+ export default class CodexCompactionEngine extends BasicCompactionEngine {
22
+ static override inject = ['llm', 'tokenMeter', 'sessions', 'sessionProjections', 'gptSubNative']
23
+
24
+ constructor(ctx: Context, config: BasicCompactionConfig = {}) {
25
+ super(ctx, config)
26
+ ctx.effect(() => ctx.sessionProjections.register(nativeCompactionProjection))
27
+ }
28
+
29
+ override async compactIfNeeded(agent: Agent, trigger: CompactionTrigger, signal: AbortSignal): Promise<CompactionResult | null> {
30
+ // Without the llm-pi-ai/stream extension the native protocol cannot run;
31
+ // Codex Subscription then keeps DSH basic compaction like every other route.
32
+ if (!this.isCodex(agent) || !hasPiAiNativeStream()) return super.compactIfNeeded(agent, trigger, signal)
33
+ signal.throwIfAborted()
34
+ const header = agent.session.requestHeader()
35
+ if (header === undefined) return null
36
+ if (trigger === 'pressure') {
37
+ const info = await this.ctx.llm.resolveModelInfo(header.config.provider, header.config.model, signal)
38
+ if (info.context === undefined) throw new Error('Codex native compaction requires the model contextWindow')
39
+ const override = this.config.modelPolicies.find(value => value.provider === header.config.provider && value.model === header.config.model)
40
+ const ratio = override?.thresholdRatio ?? this.config.thresholdRatio
41
+ const headroom = override?.headroomTokens ?? this.config.headroomTokens
42
+ const reserve = header.config.maxTokens ?? info.defaultMaxTokens ?? 0
43
+ const threshold = Math.min(Math.floor(info.context.contextWindow * ratio), info.context.contextWindow - reserve - headroom)
44
+ if (threshold <= 0) throw new Error('Codex compaction headroom and output reservation exhaust contextWindow')
45
+ if (this.nativePressure(agent.session) < threshold) return null
46
+ }
47
+ const range = this.wholeRange(agent.session)
48
+ if (range === null) return null
49
+ return this.compactRegion(range.start, range.end, agent, signal)
50
+ }
51
+
52
+ override compactNow(agent: Agent, signal: AbortSignal, sourceCommandId?: CommandId): Promise<CompactionResult | null> {
53
+ if (!this.isCodex(agent) || !hasPiAiNativeStream()) return super.compactNow(agent, signal, sourceCommandId)
54
+ signal.throwIfAborted()
55
+ try {
56
+ return agent.runMaintenance(async agentSignal => {
57
+ const operationSignal = AbortSignal.any([agentSignal, signal])
58
+ const range = this.wholeRange(agent.session)
59
+ if (range === null) return null
60
+ try {
61
+ const result = await this.nativeRegion(range.start, range.end, agent, operationSignal, null, sourceCommandId)
62
+ await this.ctx.sessions.flush(agent.session)
63
+ return result
64
+ } catch (error) {
65
+ if (agentSignal.aborted) throw new ManualCompactionError('cancelled', 'Codex native compaction was cancelled', { cause: error })
66
+ operationSignal.throwIfAborted()
67
+ throw error
68
+ }
69
+ })
70
+ } catch (error) {
71
+ throw new ManualCompactionError('busy', 'manual Codex compaction requires an idle agent', { cause: error })
72
+ }
73
+ }
74
+
75
+ override compactRegion(start: SessionSeq, end: SessionSeq, agent: Agent, signal?: AbortSignal): Promise<CompactionResult> {
76
+ if (!this.isCodex(agent) || !hasPiAiNativeStream()) return super.compactRegion(start, end, agent, signal)
77
+ const owner = this.state(agent.session).openTurn
78
+ if (owner === null) return Promise.reject(new Error('automatic Codex compaction requires an open turn'))
79
+ return this.nativeRegion(start, end, agent, signal ?? new AbortController().signal, owner)
80
+ }
81
+
82
+ private isCodex(agent: Agent): boolean {
83
+ return (agent.session.requestHeader()?.config.provider ?? agent.options.provider) === 'openai-codex'
84
+ }
85
+
86
+ private state(session: Session): NativeCompactionState {
87
+ const state = this.ctx.sessionProjections.stateOf(session, 'gptSubCompaction')
88
+ if (state === undefined) throw new Error('Codex compaction projection is not registered')
89
+ return state
90
+ }
91
+
92
+ private nativePressure(session: Session): number {
93
+ const measurement = this.ctx.tokenMeter.measure(session)
94
+ const state = this.state(session)
95
+ const correction = measurement.baseline.kind === 'usage'
96
+ ? state.nativeExtraTokens - state.lastAssistantExtraTokens
97
+ : state.nativeExtraTokens
98
+ return Math.max(0, measurement.totalTokens + correction)
99
+ }
100
+
101
+ private wholeRange(session: Session): { start: SessionSeq; end: SessionSeq } | null {
102
+ const nodes = session.surface.nodes
103
+ const first = this.state(session).systemHead === true ? 1 : 0
104
+ const start = nodes[first]
105
+ const end = nodes.at(-1)
106
+ if (start === undefined || end === undefined || nodes.length - first < 2) return null
107
+ if (!toolPairingBalancedBefore(session, start) || !toolPairingBalancedAfter(session, end)) return null
108
+ return { start, end }
109
+ }
110
+
111
+ private async nativeRegion(
112
+ start: SessionSeq, end: SessionSeq, agent: Agent, signal: AbortSignal,
113
+ owner: number | null, sourceCommandId?: CommandId,
114
+ ): Promise<CompactionResult> {
115
+ signal.throwIfAborted()
116
+ const session = agent.session
117
+ const state = this.state(session)
118
+ if (state.activeCompaction !== null || state.openTurn !== owner) {
119
+ throw new ManualCompactionError('busy', 'another turn or compaction owns this Session')
120
+ }
121
+ const range = this.wholeRange(session)
122
+ if (range === null || range.start !== start || range.end !== end) {
123
+ throw new Error('native Codex compaction requires the complete balanced history after the system head')
124
+ }
125
+ const nodes = [...session.surface.nodes]
126
+ const first = nodes.indexOf(start)
127
+ const last = nodes.indexOf(end)
128
+ const shadowedSeqs = nodes.slice(first, last + 1)
129
+ const messages: Message[] = [...session.deriveMessages()]
130
+ const measurement = this.ctx.tokenMeter.measure(session)
131
+ const shadowedTokenCount = measurement.nodes.slice(first, last + 1).reduce((sum, node) => sum + node.heuristicTokens, 0)
132
+ const previousTokens = measurement.nodes.slice(first, last + 1).reduce((sum, node) => sum + node.tokens, 0)
133
+ + state.nativeExtraTokens
134
+ const header = session.requestHeader()
135
+ const provider = header?.config.provider ?? agent.options.provider
136
+ const model = header?.config.model ?? agent.options.model
137
+ if (provider !== 'openai-codex' || model === undefined) throw new Error('native Codex compaction requires a routed subscription model')
138
+ const compactionId = CompactionId(randomUUID())
139
+ const lifecycle = { compactionId, turn: owner, ...sourceCommandId === undefined ? {} : { sourceCommandId } }
140
+ const startEvent = session.append('compaction/start', lifecycle)
141
+ let closing = false
142
+ try {
143
+ const assembler = new BlockAssembler()
144
+ for await (const chunk of this.ctx.llm.stream({
145
+ provider, model, messages, purpose: 'compaction', sessionId: session.id, signal,
146
+ toolHistory: session.toolHistory(),
147
+ ...header?.tools === undefined ? {} : { tools: [...header.tools] },
148
+ ...header?.config.reasoningEffort === undefined ? {} : { reasoningEffort: header.config.reasoningEffort },
149
+ })) assembler.push(chunk)
150
+ signal.throwIfAborted()
151
+ if (assembler.finish.kind === 'error' || assembler.finish.kind === 'aborted') {
152
+ throw new Error(assembler.finish.failure.message)
153
+ }
154
+ const result = nativeResult(assembler.blocks())
155
+ const contextTokens = result.summaryTokens + result.retained.reduce((sum, item) => sum + nativeItemTokens(item), 0)
156
+ if (contextTokens >= previousTokens) throw new Error('Codex native compaction did not reduce context tokens')
157
+ const currentNodes = session.surface.nodes
158
+ if (!isDeepStrictEqual(currentNodes.slice(first, last + 1), shadowedSeqs)
159
+ || !isDeepStrictEqual(session.deriveMessages().slice(0, messages.length), messages)) {
160
+ throw new ManualCompactionError('changed', 'Session history changed during Codex native compaction')
161
+ }
162
+ const source: CodexCheckpointSource = {
163
+ ...compactCheckpointSource(compactionId, sourceCommandId),
164
+ nativeCodex: { version: 1, provider, model, items: [...result.retained, result.item], contextTokens },
165
+ }
166
+ const summary = [{ type: 'text' as const, text: `[Codex native checkpoint ${compactionId}]` }]
167
+ const summaryEvent = session.append('compaction/summary', {
168
+ compactionId, ...sourceCommandId === undefined ? {} : { sourceCommandId },
169
+ summary, rawOutput: assembler.blocks(), llmStreamCall: true,
170
+ shadowedRange: { start, end }, shadowedSeqs, shadowedTokenCount, provider, model,
171
+ ...assembler.usage === undefined ? {} : { usage: assembler.usage },
172
+ })
173
+ session.append('user/message', createUserMessage({ content: summary, source }), {
174
+ surfaceOp: { op: 'replace', startSeq: start, endSeq: end },
175
+ sourceEventSeqs: [startEvent.seq, summaryEvent.seq, ...shadowedSeqs],
176
+ })
177
+ closing = true
178
+ const endEvent = session.append('compaction/end', lifecycle)
179
+ this.ctx.logger.info('Codex native compaction (%s): %s, summary %d tokens, retained %d messages', model, compactionId, result.summaryTokens, result.retained.length)
180
+ return { compactionId, ...sourceCommandId === undefined ? {} : { sourceCommandId },
181
+ startSeq: startEvent.seq, summarySeq: summaryEvent.seq, endSeq: endEvent.seq,
182
+ summary, shadowedRange: { start, end }, shadowedSeqs, shadowedTokenCount }
183
+ } catch (error) {
184
+ if (!closing) session.append('compaction/end', { ...lifecycle, error: error instanceof Error ? error.message : String(error) })
185
+ throw error
186
+ }
187
+ }
188
+ }
package/src/index.ts CHANGED
@@ -1,15 +1,9 @@
1
1
  /**
2
2
  * Direct ChatGPT/Codex subscription access for DeepSeek Harness.
3
3
  *
4
- * This plugin owns no transport. pi-ai already ships an `openai-codex`
5
- * provider that knows the Codex wire format -- it sets `store`, derives
6
- * `chatgpt-account-id` from the access token's own JWT claim, and speaks the
7
- * codex-responses protocol -- but it authenticates only through OAuth, and
8
- * `dsh-llm-pi-ai` runs no login flow and holds no OAuth store. Naming a
9
- * credential on the route grafts an api-key method beside the provider's own,
10
- * which is the seam this plugin fills: it keeps a live Codex access token in
11
- * the harness credential store, refreshing it from `~/.codex/auth.json` before
12
- * it expires.
4
+ * The plugin refreshes the Codex CLI subscription token, routes Codex egress,
5
+ * and executes native context compaction. pi-ai serves normal conversation
6
+ * requests and replays the encrypted checkpoints stored in Session history.
13
7
  *
14
8
  * The result needs no loopback port, no shared local key, and no external
15
9
  * proxy binary. The `codex` CLI still owns the login.
@@ -41,6 +35,7 @@ import { consumeResetCredit, listResetCredits } from './reset-credits.ts'
41
35
  import { installProxyRouting, type ProxyRouting } from './proxy-routing.ts'
42
36
  import { QuotaSource } from './quota-route.ts'
43
37
  import { TokenStore } from './token-store.ts'
38
+ import { CodexNativeBridge } from './native-bridge.ts'
44
39
  import type { Config as ConfigShape } from './types.ts'
45
40
  import { fetchUsage, reportedWindows } from './usage.ts'
46
41
 
@@ -126,7 +121,10 @@ export const Config: z<Partial<ConfigShape>, ConfigShape> = z.object({
126
121
  // exists, its proxyUrl wins over the config's -- a page edit survives
127
122
  // restarts without touching cordis config layers.
128
123
  stateFile: z.string().default('~/.dsh/gpt-sub.json'),
129
- })
124
+ nativeCompactionTimeoutMs: z.number().step(1).min(1).default(300_000),
125
+ nativeRetainTokens: z.number().step(1).min(0).default(64_000),
126
+ nativeResponseMaxBytes: z.number().step(1).min(1).default(16 * 1024 * 1024),
127
+ }) as z<Partial<ConfigShape>, ConfigShape>
130
128
 
131
129
  /** Expand a leading `~` against the current user's home directory. */
132
130
  function expandHome(path: string): string {
@@ -236,6 +234,8 @@ export async function apply(ctx: Context, config: ConfigShape): Promise<() => Pr
236
234
  throw error
237
235
  }
238
236
  ctx.logger.info('gpt-sub: published Codex access token to %s', config.tokenRef)
237
+ const nativeBridge = new CodexNativeBridge(ctx, config)
238
+ ctx.effect(() => () => nativeBridge.stop())
239
239
 
240
240
  // Reachability probe. An unproxied egress answers 403 with a Cloudflare
241
241
  // block page, and that failure is otherwise invisible until the first model
@@ -677,6 +677,7 @@ export async function apply(ctx: Context, config: ConfigShape): Promise<() => Pr
677
677
 
678
678
  return async () => {
679
679
  clearInterval(timer)
680
+ await nativeBridge.stop()
680
681
  // Restores the previous global dispatcher before closing the agent, so a
681
682
  // fiber restart never leaves the host dispatching into a closing proxy.
682
683
  await switching.catch(() => undefined)
@@ -0,0 +1,288 @@
1
+ /** Codex native compaction and checkpoint replay through resolved pi-ai requests. */
2
+ import { Service, type Context } from '@deepseek-ai/cordis'
3
+ import { LlmError, type Message, type StreamChunk, type TokenUsage } from '@deepseek-ai/dsh-llm'
4
+ import * as PiAi from '@deepseek-ai/dsh-llm-pi-ai'
5
+ import { normalizeContext } from '@earendil-works/pi-ai/utils/transcript'
6
+ import { convertResponsesMessages, convertResponsesTools } from '@earendil-works/pi-ai/api/openai-responses-shared'
7
+ import { createParser } from 'eventsource-parser'
8
+ import { z } from 'zod'
9
+ import type { ModelThinkingLevel } from '@earendil-works/pi-ai'
10
+ import {
11
+ expandCheckpoints, nativeItemsSchema, readCheckpoint, retainUserItems,
12
+ type NativeCompactionResultBlock,
13
+ } from './native-history.ts'
14
+
15
+ /** Native compaction request bounds configured by the plugin. */
16
+ export interface NativeBridgeConfig {
17
+ nativeCompactionTimeoutMs: number
18
+ nativeRetainTokens: number
19
+ nativeResponseMaxBytes: number
20
+ }
21
+
22
+ /** Options the pi-ai SDK accepts for one native request stream. */
23
+ export interface PiAiNativeStreamOptions {
24
+ apiKey?: string
25
+ reasoning?: ModelThinkingLevel
26
+ maxRetries?: number
27
+ signal?: AbortSignal
28
+ headers?: Record<string, string | null | undefined>
29
+ transport?: string
30
+ fetch?: typeof fetch
31
+ onPayload?: (payload: unknown, model: unknown) => unknown | Promise<unknown>
32
+ }
33
+
34
+ /**
35
+ * Native request the llm-pi-ai/stream extension hands to listeners.
36
+ * Mirrors the extension's own shape; declared locally so this plugin also
37
+ * typechecks against adapters that do not ship the extension yet.
38
+ */
39
+ export interface PiAiNativeRequest {
40
+ options: {
41
+ provider: string
42
+ model: string
43
+ purpose?: string
44
+ messages: Message[]
45
+ sessionId?: string | number
46
+ reasoningEffort?: string
47
+ }
48
+ model: Parameters<typeof convertResponsesMessages>[0]
49
+ context: Parameters<typeof normalizeContext>[0]
50
+ streamOptions: PiAiNativeStreamOptions
51
+ stream: (options: PiAiNativeStreamOptions) => AsyncIterable<StreamChunk>
52
+ }
53
+
54
+ /**
55
+ * Event name published by the llm-pi-ai/stream extension. Falls back to the
56
+ * wire name so the plugin loads — inert — on hosts whose adapter does not
57
+ * emit it yet; nothing fires and ordinary SDK dispatch is unchanged.
58
+ */
59
+ const PI_AI_NATIVE_STREAM_EVENT = (PiAi as { PI_AI_NATIVE_STREAM_EVENT?: string }).PI_AI_NATIVE_STREAM_EVENT ?? 'llm-pi-ai/stream'
60
+
61
+ /** Whether the host's llm-pi-ai adapter emits native requests through the stream extension. */
62
+ export function hasPiAiNativeStream(): boolean {
63
+ return (PiAi as { PI_AI_NATIVE_STREAM_EVENT?: string }).PI_AI_NATIVE_STREAM_EVENT !== undefined
64
+ }
65
+
66
+ /** Listener for one native request; call next() to defer to the SDK dispatch. */
67
+ export type PiAiNativeStreamListener = (request: PiAiNativeRequest, next: () => AsyncIterable<StreamChunk>) => AsyncIterable<StreamChunk>
68
+
69
+ /**
70
+ * Register a native stream listener. The event key is provided by the
71
+ * extension's typings when present; typed loosely here so registration also
72
+ * compiles — and stays dormant — where the extension is absent.
73
+ */
74
+ export function onPiAiNativeStream(ctx: Context, listener: PiAiNativeStreamListener): () => boolean {
75
+ return (ctx.on as unknown as (name: string, listener: PiAiNativeStreamListener) => () => boolean)(PI_AI_NATIVE_STREAM_EVENT, listener)
76
+ }
77
+
78
+ /** Drive one native request through the registered listeners, as the extension does. */
79
+ export function waterfallPiAiNativeStream(ctx: Context, request: PiAiNativeRequest, next: () => AsyncIterable<StreamChunk>): AsyncIterable<StreamChunk> {
80
+ return (ctx.waterfall as unknown as (name: string, request: PiAiNativeRequest, next: () => AsyncIterable<StreamChunk>) => AsyncIterable<StreamChunk>)(PI_AI_NATIVE_STREAM_EVENT, request, next)
81
+ }
82
+
83
+ declare module '@deepseek-ai/cordis' {
84
+ interface Context { gptSubNative: CodexNativeBridge }
85
+ }
86
+
87
+ const jsonObject = z.record(z.string(), z.unknown())
88
+ const claimsSchema = z.object({
89
+ 'https://api.openai.com/auth': z.object({ chatgpt_account_id: z.string().min(1) }),
90
+ })
91
+ const usageSchema = z.object({
92
+ input_tokens: z.number().int().nonnegative(),
93
+ output_tokens: z.number().int().nonnegative(),
94
+ input_tokens_details: z.object({ cached_tokens: z.number().int().nonnegative().optional() }).optional(),
95
+ })
96
+
97
+ /**
98
+ * Expand persisted checkpoints immediately before the SDK sends its payload.
99
+ * @param request - prepared pi-ai request.
100
+ * @param payload - provider-native request body.
101
+ * @returns body with native checkpoint items restored.
102
+ */
103
+ export function replayPayload(request: PiAiNativeRequest, payload: unknown): unknown {
104
+ const body = jsonObject.parse(payload)
105
+ const input = nativeItemsSchema.parse(body['input'])
106
+ return { ...body, input: expandCheckpoints(input, request.options.messages, request.options.provider, request.options.model) }
107
+ }
108
+
109
+ /**
110
+ * Resolve the Responses endpoint from the captured model descriptor.
111
+ * @param baseUrl - configured native model endpoint.
112
+ * @returns Codex Responses URL.
113
+ */
114
+ export function codexResponsesUrl(baseUrl: string): string {
115
+ const base = baseUrl.replace(/\/+$/, '')
116
+ if (base.endsWith('/codex/responses')) return base
117
+ if (base.endsWith('/codex')) return `${base}/responses`
118
+ return `${base}/codex/responses`
119
+ }
120
+
121
+ /** Plugin-owned native request execution, cancelled before its proxy is closed. */
122
+ export class CodexNativeBridge extends Service {
123
+ private readonly stopping = new AbortController()
124
+ private readonly active = new Set<Promise<void>>()
125
+
126
+ constructor(ctx: Context, private readonly config: NativeBridgeConfig, private readonly fetchImpl: typeof fetch = fetch) {
127
+ super(ctx, 'gptSubNative')
128
+ onPiAiNativeStream(ctx, (request, next) => {
129
+ if (request.options.provider !== 'openai-codex') return next()
130
+ if (request.model.api !== 'openai-codex-responses') {
131
+ throw new LlmError('Codex native compaction requires the openai-codex-responses protocol', 'INVALID_CONFIG')
132
+ }
133
+ if (request.options.purpose === 'compaction') return this.compact(request)
134
+ const checkpoints = request.options.messages.some(message => 'source' in message && readCheckpoint(message.source) !== undefined)
135
+ if (!checkpoints) return next()
136
+ return request.stream({
137
+ ...request.streamOptions,
138
+ onPayload: payload => replayPayload(request, payload),
139
+ })
140
+ })
141
+ ctx.on('llm/stream', (options, next) => {
142
+ for (const message of options.messages) {
143
+ if (!('source' in message)) continue
144
+ const checkpoint = readCheckpoint(message.source)
145
+ if (checkpoint !== undefined && (checkpoint.provider !== options.provider || checkpoint.model !== options.model)) {
146
+ throw new LlmError(
147
+ `This Session contains a Codex checkpoint for ${checkpoint.model}; use the same model or start a new Session`,
148
+ 'UNSUPPORTED_CONTENT',
149
+ )
150
+ }
151
+ }
152
+ return next()
153
+ })
154
+ }
155
+
156
+ /**
157
+ * Abort and await all native compaction requests before transport teardown.
158
+ * @returns completion of every active request.
159
+ */
160
+ async stop(): Promise<void> {
161
+ this.stopping.abort(new Error('Codex subscription plugin unloaded'))
162
+ await Promise.all(this.active)
163
+ }
164
+
165
+ private async *compact(request: PiAiNativeRequest): AsyncGenerator<StreamChunk> {
166
+ let settle: () => void = () => undefined
167
+ const done = new Promise<void>(resolve => { settle = resolve })
168
+ this.active.add(done)
169
+ const signal = AbortSignal.any([
170
+ this.stopping.signal,
171
+ AbortSignal.timeout(this.config.nativeCompactionTimeoutMs),
172
+ ...request.streamOptions.signal === undefined ? [] : [request.streamOptions.signal],
173
+ ])
174
+ let reader: ReadableStreamDefaultReader<Uint8Array> | undefined
175
+ try {
176
+ signal.throwIfAborted()
177
+ const token = request.streamOptions.apiKey
178
+ if (token === undefined) throw new LlmError('Codex native compaction requires a resolved subscription token', 'MISSING_CREDENTIAL')
179
+ const payloadPart = token.split('.')[1]
180
+ if (payloadPart === undefined) throw new LlmError('Codex subscription token is not a JWT', 'MISSING_CREDENTIAL')
181
+ const account = claimsSchema.parse(JSON.parse(Buffer.from(payloadPart, 'base64url').toString('utf8')))
182
+ const serialized = convertResponsesMessages(request.model, normalizeContext(request.context), new Set(['openai', 'openai-codex', 'azure-openai-responses']), {
183
+ includeSystemPrompt: false,
184
+ })
185
+ const input = expandCheckpoints(
186
+ nativeItemsSchema.parse(JSON.parse(JSON.stringify(serialized))),
187
+ request.options.messages, request.options.provider, request.options.model,
188
+ )
189
+ const body: Record<string, unknown> = {
190
+ model: request.model.id,
191
+ store: false,
192
+ stream: true,
193
+ instructions: request.context.systemPrompt ?? '',
194
+ input: [...input, { type: 'compaction_trigger' }],
195
+ parallel_tool_calls: true,
196
+ include: ['reasoning.encrypted_content'],
197
+ ...request.options.sessionId === undefined ? {} : { prompt_cache_key: String(request.options.sessionId) },
198
+ }
199
+ if (request.context.tools !== undefined) body['tools'] = convertResponsesTools(request.context.tools, { strict: null })
200
+ const reasoning = request.streamOptions.reasoning
201
+ if (reasoning !== undefined) {
202
+ const effort = request.model.thinkingLevelMap?.[reasoning] ?? reasoning
203
+ body['reasoning'] = { effort, summary: 'auto' }
204
+ }
205
+ const headers = new Headers()
206
+ for (const [name, value] of Object.entries({ ...request.model.headers, ...request.streamOptions.headers })) {
207
+ if (value !== null && value !== undefined) headers.set(name, value)
208
+ }
209
+ headers.set('authorization', `Bearer ${token}`)
210
+ headers.set('chatgpt-account-id', account['https://api.openai.com/auth'].chatgpt_account_id)
211
+ headers.set('originator', 'codex_cli_rs')
212
+ headers.set('OpenAI-Beta', 'responses=experimental')
213
+ headers.set('content-type', 'application/json')
214
+ headers.set('accept', 'text/event-stream')
215
+ if (request.options.sessionId !== undefined) headers.set('session-id', String(request.options.sessionId))
216
+ const response = await this.fetchImpl(codexResponsesUrl(request.model.baseUrl), {
217
+ method: 'POST', headers, body: JSON.stringify(body), signal,
218
+ })
219
+ if (!response.ok) throw new LlmError(`Codex native compaction HTTP ${response.status}: ${await response.text()}`, 'CODEX_COMPACTION_HTTP_ERROR')
220
+ if (response.body === null) throw new LlmError('Codex native compaction returned no response stream', 'STREAM_CLOSED')
221
+ reader = response.body.getReader()
222
+ const events: string[] = []
223
+ let parseError: Error | undefined
224
+ const parser = createParser({ onEvent: event => events.push(event.data), onError: error => { parseError = error } })
225
+ const decoder = new TextDecoder()
226
+ let receivedBytes = 0
227
+ let compaction: NativeCompactionResultBlock['item'] | undefined
228
+ let compactionCount = 0
229
+ let completed: Record<string, unknown> | undefined
230
+ while (completed === undefined) {
231
+ const chunk = await reader.read()
232
+ if (chunk.done) {
233
+ parser.feed(decoder.decode())
234
+ if (events.length === 0) throw new LlmError('Codex compaction stream closed before response.completed', 'STREAM_CLOSED')
235
+ } else {
236
+ receivedBytes += chunk.value.byteLength
237
+ if (receivedBytes > this.config.nativeResponseMaxBytes) throw new LlmError('Codex compaction response exceeded nativeResponseMaxBytes', 'CODEX_COMPACTION_RESPONSE_LIMIT')
238
+ parser.feed(decoder.decode(chunk.value, { stream: true }))
239
+ }
240
+ if (parseError !== undefined) throw parseError
241
+ for (const data of events.splice(0)) {
242
+ if (data === '[DONE]') continue
243
+ const event = jsonObject.parse(JSON.parse(data))
244
+ if (event['type'] === 'response.output_item.done') {
245
+ const item = nativeItemsSchema.element.parse(event['item'])
246
+ if (item['type'] === 'compaction') {
247
+ compaction = item
248
+ compactionCount += 1
249
+ }
250
+ } else if (event['type'] === 'response.completed') completed = jsonObject.parse(event['response'])
251
+ else if (event['type'] === 'error' || event['type'] === 'response.failed' || event['type'] === 'response.incomplete') {
252
+ throw new LlmError(`Codex native compaction failed: ${data}`, 'CODEX_COMPACTION_ERROR')
253
+ }
254
+ }
255
+ signal.throwIfAborted()
256
+ if (chunk.done && completed === undefined) throw new LlmError('Codex compaction stream closed before response.completed', 'STREAM_CLOSED')
257
+ }
258
+ if (compactionCount !== 1 || compaction === undefined || typeof compaction['encrypted_content'] !== 'string' || compaction['encrypted_content'].length === 0) {
259
+ throw new LlmError(`Codex native compaction expected one encrypted compaction item, received ${compactionCount}`, 'INVALID_REPLAY_STATE')
260
+ }
261
+ const responseId = z.string().min(1).parse(completed['id'])
262
+ const usage = usageSchema.parse(completed['usage'])
263
+ const cached = usage.input_tokens_details?.cached_tokens ?? 0
264
+ const tokenUsage: TokenUsage = {
265
+ inputTokens: Math.max(0, usage.input_tokens - cached),
266
+ outputTokens: usage.output_tokens,
267
+ cacheReadTokens: cached,
268
+ totalTokens: usage.input_tokens + usage.output_tokens,
269
+ }
270
+ yield {
271
+ type: 'block-end', index: 0,
272
+ block: {
273
+ type: 'codex-compaction-result', item: compaction, responseId,
274
+ summaryTokens: usage.output_tokens,
275
+ retained: retainUserItems(input, this.config.nativeRetainTokens),
276
+ },
277
+ }
278
+ yield { type: 'usage', usage: tokenUsage }
279
+ yield { type: 'finish', reason: { kind: 'stop' } }
280
+ } finally {
281
+ try { await reader?.cancel() } finally {
282
+ reader?.releaseLock()
283
+ this.active.delete(done)
284
+ settle()
285
+ }
286
+ }
287
+ }
288
+ }