codex-grok-bridge 1.7.1 → 1.7.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +9 -0
- package/README.md +84 -20
- package/package.json +1 -1
- package/src/bridge.mjs +106 -7
- package/src/imagine.mjs +150 -0
- package/src/proxy.mjs +87 -2
- package/src/tools.mjs +256 -27
- package/src/videogen.mjs +196 -0
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,14 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 1.7.2 — 2026-09-30
|
|
4
|
+
|
|
5
|
+
- Readable compaction summaries are passed through as user messages. A compaction item that only carries encrypted or internal fields is still dropped.
|
|
6
|
+
- A Grok thread can generate a video without an API key. The bridge declares `grok_bridge_generate_video` on the upstream request and, when the model calls it, posts to `https://api.x.ai/v1/videos/generations` with the grok login bearer, polls `GET /videos/{request_id}`, and returns the video URL to the model as the tool result. A refusal from that API stays a tool error and does not fail the turn.
|
|
7
|
+
- Codex input items Grok does not accept (`local_shell_call`, `web_search_call`, `tool_search_call`, `tool_search_output`, `additional_tools`, `image_generation_call`, and unmapped custom tool items) keep their readable text as user `input_text` messages, the same shape as a compaction summary. That text includes shell commands, search queries, tool-search text, an image `revised_prompt`, and custom tool text. `encrypted_content`, `encrypted_function_args`, `internal_*` fields, and image-byte `result` values stay dropped.
|
|
8
|
+
- `POST /v1/images/generations` and `POST /v1/images/edits` are forwarded to the Imagine API with the grok login bearer. Codex's `gpt-image-2` body is rewritten to `grok-imagine-image-quality` and `b64_json`. OpenAI `file_id` edits are refused. No `XAI_API_KEY`.
|
|
9
|
+
- Codex `web_search` is kept as one server-side `{type:"web_search"}` tool. `web_search_call` results are passed through to Codex.
|
|
10
|
+
- Voice WebRTC (`POST /v1/realtime/calls`) and Codex cloud tasks are not in this package.
|
|
11
|
+
|
|
3
12
|
## 1.7.1 — 2026-09-30
|
|
4
13
|
|
|
5
14
|
- The bridge holds the upstream reply and sends it to Codex only after `response.completed`. A reset before that (`ECONNRESET`, `EPIPE`, `UND_ERR_SOCKET`, or a close before the reply finishes) is retried up to two more times, and only if Codex has not been sent a byte. This provider's Codex `request_max_retries` and `stream_max_retries` are 0. A retry sends the prompt again, so that attempt's input tokens can be billed again.
|
package/README.md
CHANGED
|
@@ -30,13 +30,45 @@ The bridge does **not** execute Grok-native tools. It translates Codex tools int
|
|
|
30
30
|
function calling, streams the upstream Responses events, and rewrites names back
|
|
31
31
|
so Codex still recognizes them.
|
|
32
32
|
|
|
33
|
-
|
|
33
|
+
Eighteen files under `src/`. Zero runtime dependencies. Node.js ≥ 22.
|
|
34
34
|
|
|
35
35
|
This is not a second-opinion review product and not a Grok-native search
|
|
36
36
|
product. The picker shows `grok-4.7` (Grok 4.7 / xAI) and keeps `grok-4.6`.
|
|
37
37
|
Any other `grok-*` id uses the same route. `GROK_BRIDGE_MODELS` adds further
|
|
38
|
-
`grok-*` ids.
|
|
39
|
-
|
|
38
|
+
`grok-*` ids. Codex `web_search` is xAI's server-side tool, not a function
|
|
39
|
+
the bridge runs. See Release history.
|
|
40
|
+
|
|
41
|
+
## Release history
|
|
42
|
+
|
|
43
|
+
What each recent version added. Older cuts are in `CHANGELOG.md`.
|
|
44
|
+
|
|
45
|
+
### 1.7.2 — 2026-09-30
|
|
46
|
+
|
|
47
|
+
- Readable compaction summaries are passed through as user messages. A compaction item that only carries encrypted or internal fields is still dropped.
|
|
48
|
+
- `grok_bridge_generate_video`. The bridge declares this tool and, when the model calls it, posts to `https://api.x.ai/v1/videos/generations` with the grok login bearer, polls `GET /videos/{request_id}`, and returns the video URL as the tool result. A refusal from that API stays a tool error and does not fail the turn.
|
|
49
|
+
- Readable text from Codex input items that used to be dropped is forwarded as user `input_text`: shell, search, tool-search, a revised image prompt, and custom tool text. Encrypted blobs and image-byte results stay dropped.
|
|
50
|
+
- `POST /v1/images/generations` and `POST /v1/images/edits` are forwarded to the Imagine API with the grok login bearer. Codex's `gpt-image-2` body is rewritten to `grok-imagine-image-quality` and `b64_json`. OpenAI `file_id` edits are refused. No `XAI_API_KEY`.
|
|
51
|
+
- Codex `web_search` is kept as one server-side `{type:"web_search"}` tool. `web_search_call` results are passed through to Codex.
|
|
52
|
+
- Voice WebRTC (`POST /v1/realtime/calls`) and Codex cloud tasks are not in this package.
|
|
53
|
+
|
|
54
|
+
### 1.7.1 — 2026-09-30
|
|
55
|
+
|
|
56
|
+
- The bridge holds the upstream reply and sends it to Codex only after `response.completed`. A reset before that (`ECONNRESET`, `EPIPE`, `UND_ERR_SOCKET`, or a close before the reply finishes) is retried up to two more times, and only if Codex has not been sent a byte. This provider's Codex `request_max_retries` and `stream_max_retries` are 0. A retry sends the prompt again, so that attempt's input tokens can be billed again.
|
|
57
|
+
- When `grok --version` fails or does not report a version, the client version is `unknown` instead of the stale `1.0.24`.
|
|
58
|
+
|
|
59
|
+
### 1.7.0 — 2026-09-29
|
|
60
|
+
|
|
61
|
+
- One `x-grok-conv-id` per Codex thread, and the forwarded transcript prefix stays byte-stable so the prompt cache can hit. The full transcript is still sent.
|
|
62
|
+
- Upstream `cached_prompt_tokens` and `cache_read_input_tokens` are copied onto `response.completed` usage and the diagnostics log when the proxy sends them, including `0`. Missing counters are not invented.
|
|
63
|
+
|
|
64
|
+
### 1.6.1 — 2026-09-28
|
|
65
|
+
|
|
66
|
+
- Darwin bundled CLI resolves `Codex.app/Contents/Resources/codex-cli/bin/codex` when that file exists (Codex 26.924). The legacy `Contents/Resources/codex` path remains the fallback. `CODEX_BINARY` still wins. Linux and Windows layouts are unchanged.
|
|
67
|
+
|
|
68
|
+
### 1.6.0 — 2026-09-27
|
|
69
|
+
|
|
70
|
+
- Default catalog model is `grok-4.7` (`Grok 4.7 / xAI`). `grok-4.6` stays listed so existing threads still resolve.
|
|
71
|
+
- `grok-*` still routes to `grok_build_cli`. No other model ids were added. `GROK_BRIDGE_MODELS` still appends extra `grok-*` ids.
|
|
40
72
|
|
|
41
73
|
## Requirements
|
|
42
74
|
|
|
@@ -72,7 +104,7 @@ starts, and tears the provider down with that process.
|
|
|
72
104
|
```sh
|
|
73
105
|
git clone https://github.com/deximple/codex-grok-bridge.git
|
|
74
106
|
cd codex-grok-bridge
|
|
75
|
-
npm test #
|
|
107
|
+
npm test # 191 tests, no network, no inference
|
|
76
108
|
node scripts/codex-grok.mjs
|
|
77
109
|
```
|
|
78
110
|
|
|
@@ -107,8 +139,8 @@ launches from its usual icon.
|
|
|
107
139
|
2. The wrapper adds Grok to the model catalog and sets the provider of a new
|
|
108
140
|
Grok thread to `grok_build_cli`.
|
|
109
141
|
3. Codex `/v1/responses` requests go to the localhost bridge.
|
|
110
|
-
4. The bridge flattens Codex tools
|
|
111
|
-
|
|
142
|
+
4. The bridge flattens Codex function and custom tools into function tools,
|
|
143
|
+
forwards `web_search` as xAI's server-side tool, then pipes the
|
|
112
144
|
`cli-chat-proxy.grok.com` Responses stream through.
|
|
113
145
|
5. Tool results return as the next Codex request `input`.
|
|
114
146
|
|
|
@@ -139,8 +171,8 @@ before any of it is sent.
|
|
|
139
171
|
Codex keeps the conversation and sends the whole turn each time. Grok can
|
|
140
172
|
treat the unchanged beginning as a cache. If the connection drops before
|
|
141
173
|
Codex has been sent any bytes, the bridge tries again. Codex does not send
|
|
142
|
-
that same request again. Voice
|
|
143
|
-
|
|
174
|
+
that same request again. Voice WebRTC (`POST /v1/realtime/calls`) and
|
|
175
|
+
Codex cloud tasks are not in this package.
|
|
144
176
|
|
|
145
177
|
If the Responses path misbehaves, `GROK_BRIDGE_INFERENCE=cli` falls back to the
|
|
146
178
|
older CLI envelope. That path pastes the whole JSON into a prompt each turn, so
|
|
@@ -258,7 +290,7 @@ starting a new thread, before the first turn is saved.
|
|
|
258
290
|
|
|
259
291
|
### Not in this release
|
|
260
292
|
|
|
261
|
-
Voice
|
|
293
|
+
Voice WebRTC (`POST /v1/realtime/calls`) and Codex cloud tasks are not in this package.
|
|
262
294
|
|
|
263
295
|
Upstream sometimes resets the connection mid-response (three measured cases:
|
|
264
296
|
25 s / 27 s / 253 s, 726 KB–22 MB). The bridge holds the reply and, if that
|
|
@@ -308,7 +340,7 @@ Start here when something breaks. Do not open `~/.grok/auth.json` or
|
|
|
308
340
|
## Verify
|
|
309
341
|
|
|
310
342
|
```sh
|
|
311
|
-
npm test #
|
|
343
|
+
npm test # 191 tests, no remote inference
|
|
312
344
|
npm run test:coverage # 80% line / branch / function gate
|
|
313
345
|
npm run verify:app-server # real app-server routing; also runs against an installed bundle
|
|
314
346
|
npm audit --omit=dev
|
|
@@ -377,14 +409,46 @@ Grok 4.7을 Codex 모델 목록에 넣고, Codex의 `/v1/responses`를
|
|
|
377
409
|
calling으로 옮기고, 상류 Responses 스트림을 전달한 뒤, Codex가 알아보는
|
|
378
410
|
이름으로 되돌립니다.
|
|
379
411
|
|
|
380
|
-
`src/` 아래 파일
|
|
412
|
+
`src/` 아래 파일 18개. 런타임 의존성 없음. Node.js 22 이상. 1.5.0부터 npm
|
|
381
413
|
`"os"`는 `darwin` / `linux` / `win32`입니다.
|
|
382
414
|
|
|
383
415
|
코드 리뷰 전용 제품이 아니고 Grok 네이티브 검색 제품도 아닙니다. 피커에는
|
|
384
416
|
`grok-4.7`(Grok 4.7 / xAI)이 기본으로 보이고 `grok-4.6`도 남습니다. 그 외
|
|
385
417
|
`grok-*` id도 같은 경로로 붙고, `GROK_BRIDGE_MODELS`로 카탈로그에 더합니다.
|
|
386
|
-
Codex
|
|
387
|
-
|
|
418
|
+
Codex `web_search`는 브리지가 실행하는 함수가 아니라 xAI 서버 도구입니다.
|
|
419
|
+
릴리스 기록을 보세요.
|
|
420
|
+
|
|
421
|
+
## 릴리스 기록
|
|
422
|
+
|
|
423
|
+
최근 버전이 더한 것입니다. 그 이전은 `CHANGELOG.md`에 있습니다.
|
|
424
|
+
|
|
425
|
+
### 1.7.2 — 2026-09-30
|
|
426
|
+
|
|
427
|
+
- 읽을 수 있는 compaction 요약을 사용자 메시지로 넘깁니다. 암호화된 내용이나 내부 필드만 있는 compaction 항목은 그대로 버립니다.
|
|
428
|
+
- `grok_bridge_generate_video`. 브리지가 이 도구를 선언하고, 모델이 호출하면 grok 로그인 bearer로 `https://api.x.ai/v1/videos/generations`에 보낸 뒤 `GET /videos/{request_id}`를 폴링하고, 영상 URL을 도구 결과로 돌려줍니다. 그 API의 거절은 도구 오류로 남고 턴을 실패시키지 않습니다.
|
|
429
|
+
- 예전에는 버리던 Codex 입력 항목의 읽을 수 있는 글을 사용자 `input_text`로 넘깁니다. 셸, 검색, tool-search, 수정된 이미지 프롬프트, 커스텀 도구 텍스트입니다. 암호화된 blob과 이미지 바이트 결과는 그대로 버립니다.
|
|
430
|
+
- `POST /v1/images/generations`와 `POST /v1/images/edits`를 grok 로그인 bearer로 Imagine API에 넘깁니다. Codex의 `gpt-image-2` 본문은 `grok-imagine-image-quality`와 `b64_json`으로 바꿉니다. OpenAI `file_id` 편집은 거절합니다. `XAI_API_KEY`는 쓰지 않습니다.
|
|
431
|
+
- Codex `web_search`는 서버 측 `{type:"web_search"}` 도구 하나로 유지합니다. `web_search_call` 결과는 Codex로 통과합니다.
|
|
432
|
+
- 음성 WebRTC(`POST /v1/realtime/calls`)와 Codex 클라우드 작업은 이 패키지에 없습니다.
|
|
433
|
+
|
|
434
|
+
### 1.7.1 — 2026-09-30
|
|
435
|
+
|
|
436
|
+
- 브리지는 상류 응답을 `response.completed` 이후에만 Codex로 보냅니다. 그 전의 리셋(`ECONNRESET`, `EPIPE`, `UND_ERR_SOCKET`, 또는 응답이 끝나기 전의 닫힘)은 최대 두 번 더 재시도하며, Codex에 바이트를 아직 보내지 않았을 때만 합니다. 이 provider의 Codex `request_max_retries`와 `stream_max_retries`는 0입니다. 재시도는 프롬프트를 다시 보내므로 그 시도의 입력 토큰이 다시 과금될 수 있습니다.
|
|
437
|
+
- `grok --version`이 실패하거나 버전을 보고하지 않으면 클라이언트 버전은 오래된 `1.0.24` 대신 `unknown`입니다.
|
|
438
|
+
|
|
439
|
+
### 1.7.0 — 2026-09-29
|
|
440
|
+
|
|
441
|
+
- Codex 스레드마다 `x-grok-conv-id`는 하나이고, 전달하는 트랜스크립트 접두는 바이트 그대로라 프롬프트 캐시가 맞을 수 있습니다. 전체 트랜스크립트는 그대로 보냅니다.
|
|
442
|
+
- 상류가 보내면 `cached_prompt_tokens`와 `cache_read_input_tokens`를 `response.completed` usage와 진단 로그에 복사합니다. `0`도 포함합니다. 없는 카운터는 만들지 않습니다.
|
|
443
|
+
|
|
444
|
+
### 1.6.1 — 2026-09-28
|
|
445
|
+
|
|
446
|
+
- Darwin 번들 CLI는 파일이 있으면 `Codex.app/Contents/Resources/codex-cli/bin/codex`를 씁니다(Codex 26.924). 예전 `Contents/Resources/codex`는 폴백으로 남습니다. `CODEX_BINARY`가 우선합니다. Linux와 Windows 배치는 그대로입니다.
|
|
447
|
+
|
|
448
|
+
### 1.6.0 — 2026-09-27
|
|
449
|
+
|
|
450
|
+
- 기본 카탈로그 모델은 `grok-4.7`(Grok 4.7 / xAI)입니다. `grok-4.6`은 기존 스레드가 계속 맞게 목록에 남습니다.
|
|
451
|
+
- `grok-*`는 여전히 `grok_build_cli`로 갑니다. 다른 모델 id는 추가하지 않았습니다. `GROK_BRIDGE_MODELS`는 그 외 `grok-*` id를 카탈로그에 더합니다.
|
|
388
452
|
|
|
389
453
|
## 필요한 것
|
|
390
454
|
|
|
@@ -419,7 +483,7 @@ codex-grok exec --skip-git-repo-check --sandbox workspace-write '작업 내용'
|
|
|
419
483
|
```sh
|
|
420
484
|
git clone https://github.com/deximple/codex-grok-bridge.git
|
|
421
485
|
cd codex-grok-bridge
|
|
422
|
-
npm test #
|
|
486
|
+
npm test # 191건, 네트워크·추론 없음
|
|
423
487
|
node scripts/codex-grok.mjs
|
|
424
488
|
```
|
|
425
489
|
|
|
@@ -453,8 +517,8 @@ Codex 창에도 보일 수 있습니다.
|
|
|
453
517
|
2. 래퍼가 모델 목록에 Grok를 추가하고, 새 Grok 작업의 제공자를
|
|
454
518
|
`grok_build_cli`로 설정합니다.
|
|
455
519
|
3. Codex의 `/v1/responses` 요청은 localhost 브리지로 갑니다.
|
|
456
|
-
4. 브리지는 Codex
|
|
457
|
-
|
|
520
|
+
4. 브리지는 Codex 함수·커스텀 도구를 function tool로 펼치고, `web_search`는
|
|
521
|
+
xAI 서버 도구로 넘긴 뒤 `cli-chat-proxy.grok.com` Responses 스트림을 이어줍니다.
|
|
458
522
|
5. 도구 결과는 다음 Codex 요청의 `input`으로 돌아갑니다.
|
|
459
523
|
|
|
460
524
|
인증은 `grok login` 세션입니다. `XAI_API_KEY`는 쓰지 않습니다.
|
|
@@ -481,8 +545,8 @@ Codex 창에도 보일 수 있습니다.
|
|
|
481
545
|
|
|
482
546
|
Codex가 대화를 갖고 있고, 턴마다 그 턴 전체를 보냅니다. 앞부분이 그대로면
|
|
483
547
|
Grok는 그 부분을 캐시로 볼 수 있습니다. Codex에 바이트를 보내기 전에 연결이
|
|
484
|
-
끊기면 브리지가 다시 시도합니다. Codex는 그 요청을 또 보내지 않습니다.
|
|
485
|
-
클라우드
|
|
548
|
+
끊기면 브리지가 다시 시도합니다. Codex는 그 요청을 또 보내지 않습니다. 음성
|
|
549
|
+
WebRTC(`POST /v1/realtime/calls`)와 Codex 클라우드 작업은 이 패키지에 없습니다.
|
|
486
550
|
|
|
487
551
|
Responses 경로가 이상하면 `GROK_BRIDGE_INFERENCE=cli`로 이전 CLI 봉투 경로를
|
|
488
552
|
씁니다. 매 턴 전체 JSON을 프롬프트로 넣으므로 더 느리고 비싸며, 토큰 단위
|
|
@@ -592,7 +656,7 @@ OpenAI 경로가 맞습니다.
|
|
|
592
656
|
|
|
593
657
|
### 이번 릴리스에 없는 것
|
|
594
658
|
|
|
595
|
-
|
|
659
|
+
음성 WebRTC(`POST /v1/realtime/calls`)와 Codex 클라우드 작업은 이 패키지에 없습니다.
|
|
596
660
|
|
|
597
661
|
상류가 응답 중간에 연결을 리셋하는 경우가 있습니다(실측 3건: 25초 / 27초 /
|
|
598
662
|
253초, 726 KB–22 MB). 브리지는 응답을 들고 있다가, Codex에 그 응답을 보내기
|
|
@@ -640,7 +704,7 @@ OpenAI 경로가 맞습니다.
|
|
|
640
704
|
## 검증
|
|
641
705
|
|
|
642
706
|
```sh
|
|
643
|
-
npm test #
|
|
707
|
+
npm test # 191건, 외부 추론 없음
|
|
644
708
|
npm run test:coverage # line/branch/function 80% 게이트
|
|
645
709
|
npm run verify:app-server # 실제 app-server 라우팅. 설치된 앱 번들에서도 실행
|
|
646
710
|
npm audit --omit=dev
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "codex-grok-bridge",
|
|
3
|
-
"version": "1.7.
|
|
3
|
+
"version": "1.7.2",
|
|
4
4
|
"description": "Run Grok 4.7 as the model inside Codex, with Codex still owning tools, permissions, history and MCP. Uses the grok login session, not an API key.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"license": "MIT",
|
package/src/bridge.mjs
CHANGED
|
@@ -8,7 +8,9 @@ import {
|
|
|
8
8
|
parseGrokResult,
|
|
9
9
|
runGrok,
|
|
10
10
|
} from "./cli-inference.mjs";
|
|
11
|
-
import {
|
|
11
|
+
import { emitProxySse, readRelayProxySse } from "./proxy.mjs";
|
|
12
|
+
import { videoCallsFromParts, videoToolOutput } from "./videogen.mjs";
|
|
13
|
+
import { forwardImagine } from "./imagine.mjs";
|
|
12
14
|
import {
|
|
13
15
|
applyCacheUsage,
|
|
14
16
|
createPrefixMemory,
|
|
@@ -23,6 +25,89 @@ import { catalogModelInfos, isGrokModel, MODEL_INFO } from "./models.mjs";
|
|
|
23
25
|
|
|
24
26
|
export { MODEL_INFO };
|
|
25
27
|
|
|
28
|
+
const VIDEO_FOLLOW_UPS = 2;
|
|
29
|
+
|
|
30
|
+
async function relayWithVideo(options, output, map, usageBox, onClientByte) {
|
|
31
|
+
let body = options.body;
|
|
32
|
+
for (let step = 0; step <= VIDEO_FOLLOW_UPS; step += 1) {
|
|
33
|
+
const parts = await readRelayProxySse({ ...options, body }, output);
|
|
34
|
+
const calls =
|
|
35
|
+
process.env.GROK_BRIDGE_VIDEO_GEN === "off" ? [] : videoCallsFromParts(parts);
|
|
36
|
+
if (!calls.length || step === VIDEO_FOLLOW_UPS) {
|
|
37
|
+
await emitProxySse(parts, output, map, usageBox, onClientByte);
|
|
38
|
+
return;
|
|
39
|
+
}
|
|
40
|
+
// The videos API poll can sit for minutes. Keepalive is safe here: this
|
|
41
|
+
// upstream reply is finished, and nothing has been written to Codex yet.
|
|
42
|
+
onClientByte();
|
|
43
|
+
const additions = [];
|
|
44
|
+
for (const call of calls) {
|
|
45
|
+
const toolOutput = await videoToolOutput(call, options.video);
|
|
46
|
+
additions.push(
|
|
47
|
+
{
|
|
48
|
+
type: "function_call",
|
|
49
|
+
name: call.name,
|
|
50
|
+
call_id: call.call_id,
|
|
51
|
+
arguments: call.arguments,
|
|
52
|
+
},
|
|
53
|
+
{
|
|
54
|
+
type: "function_call_output",
|
|
55
|
+
name: call.name,
|
|
56
|
+
call_id: call.call_id,
|
|
57
|
+
output: toolOutput,
|
|
58
|
+
},
|
|
59
|
+
);
|
|
60
|
+
}
|
|
61
|
+
body = { ...body, input: [...body.input, ...additions] };
|
|
62
|
+
}
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
async function relayImagine(req, res, json, kind, options) {
|
|
66
|
+
const controller = new AbortController();
|
|
67
|
+
res.on("close", () => controller.abort());
|
|
68
|
+
let body;
|
|
69
|
+
try {
|
|
70
|
+
const chunks = [];
|
|
71
|
+
let size = 0;
|
|
72
|
+
const limit = options.maxBodyBytes ?? 40 * 1024 * 1024;
|
|
73
|
+
for await (const chunk of req) {
|
|
74
|
+
size += chunk.length;
|
|
75
|
+
if (size > limit) return json(413, { error: "Request too large" });
|
|
76
|
+
chunks.push(chunk);
|
|
77
|
+
}
|
|
78
|
+
body = JSON.parse(Buffer.concat(chunks).toString("utf8"));
|
|
79
|
+
} catch {
|
|
80
|
+
return json(400, { error: "Invalid request" });
|
|
81
|
+
}
|
|
82
|
+
try {
|
|
83
|
+
const session = options.grokSession ?? readGrokBearerToken(options.grokHome);
|
|
84
|
+
const response = await forwardImagine({
|
|
85
|
+
kind,
|
|
86
|
+
body,
|
|
87
|
+
token: session.token,
|
|
88
|
+
fetchImpl: options.imagineFetch,
|
|
89
|
+
baseUrl: options.imagineBaseUrl,
|
|
90
|
+
signal: controller.signal,
|
|
91
|
+
});
|
|
92
|
+
if (!res.writableEnded && !res.destroyed) json(200, response);
|
|
93
|
+
} catch (error) {
|
|
94
|
+
if (res.writableEnded || res.destroyed) return;
|
|
95
|
+
if (error instanceof GrokAuthError) return json(401, { error: error.message });
|
|
96
|
+
const status = Number(error?.status);
|
|
97
|
+
const message = String(error?.message ?? "");
|
|
98
|
+
if (
|
|
99
|
+
status >= 400 &&
|
|
100
|
+
status < 600 &&
|
|
101
|
+
(message.startsWith("Image generation") ||
|
|
102
|
+
message.startsWith("Image edit") ||
|
|
103
|
+
message.startsWith("OpenAI file") ||
|
|
104
|
+
message.startsWith("Grok login"))
|
|
105
|
+
)
|
|
106
|
+
return json(status, { error: message });
|
|
107
|
+
return json(502, { error: "Image generation failed." });
|
|
108
|
+
}
|
|
109
|
+
}
|
|
110
|
+
|
|
26
111
|
export function publicBridgeError(error, context = {}) {
|
|
27
112
|
if (error instanceof GrokAuthError) return error.message;
|
|
28
113
|
const message = String(error?.message ?? "");
|
|
@@ -51,13 +136,20 @@ export function createBridgeServer(options = {}) {
|
|
|
51
136
|
const route = new URL(req.url, "http://127.0.0.1").pathname;
|
|
52
137
|
if (req.method === "GET" && route === "/v1/models")
|
|
53
138
|
return json(200, { models: catalogModelInfos() });
|
|
54
|
-
|
|
139
|
+
const imagineKind =
|
|
140
|
+
route === "/v1/images/generations"
|
|
141
|
+
? "generations"
|
|
142
|
+
: route === "/v1/images/edits"
|
|
143
|
+
? "edits"
|
|
144
|
+
: null;
|
|
145
|
+
if (req.method !== "POST" || (!imagineKind && route !== "/v1/responses"))
|
|
55
146
|
return json(404, { error: "Not found" });
|
|
56
147
|
const auth = Buffer.from(req.headers.authorization ?? "");
|
|
57
148
|
if (auth.length !== token.length || !timingSafeEqual(auth, token))
|
|
58
149
|
return json(401, { error: "Unauthorized" });
|
|
59
150
|
if (req.headers.origin)
|
|
60
151
|
return json(403, { error: "Browser requests are not accepted" });
|
|
152
|
+
if (imagineKind) return relayImagine(req, res, json, imagineKind, options);
|
|
61
153
|
let body;
|
|
62
154
|
let requestBytes = 0;
|
|
63
155
|
try {
|
|
@@ -165,7 +257,7 @@ export function createBridgeServer(options = {}) {
|
|
|
165
257
|
},
|
|
166
258
|
});
|
|
167
259
|
} else {
|
|
168
|
-
const session = readGrokBearerToken(options.grokHome);
|
|
260
|
+
const session = options.grokSession ?? readGrokBearerToken(options.grokHome);
|
|
169
261
|
// Grok takes input_image blocks natively; only unusable attachments
|
|
170
262
|
// are swapped for an explanation, so one bad image cannot make the
|
|
171
263
|
// upstream reject the whole conversation.
|
|
@@ -179,7 +271,7 @@ export function createBridgeServer(options = {}) {
|
|
|
179
271
|
request.input = prefixes.reuse(convId, projected);
|
|
180
272
|
const usageBox = {};
|
|
181
273
|
try {
|
|
182
|
-
await
|
|
274
|
+
await relayWithVideo(
|
|
183
275
|
{
|
|
184
276
|
token: session.token,
|
|
185
277
|
userId: session.userId,
|
|
@@ -189,9 +281,6 @@ export function createBridgeServer(options = {}) {
|
|
|
189
281
|
baseUrl: options.proxyBaseUrl,
|
|
190
282
|
convId,
|
|
191
283
|
sessionId: Array.isArray(threadId) ? threadId[0] : threadId,
|
|
192
|
-
onClientByte: () => {
|
|
193
|
-
responseStarted = true;
|
|
194
|
-
},
|
|
195
284
|
onRetry: ({ attempt, kind }) =>
|
|
196
285
|
diagnostics.record({
|
|
197
286
|
event: "turn_retried",
|
|
@@ -199,10 +288,20 @@ export function createBridgeServer(options = {}) {
|
|
|
199
288
|
attempt,
|
|
200
289
|
elapsedMs: Date.now() - startedAt,
|
|
201
290
|
}),
|
|
291
|
+
video: {
|
|
292
|
+
token: session.token,
|
|
293
|
+
fetchImpl: options.videoFetch,
|
|
294
|
+
baseUrl: options.videoBaseUrl,
|
|
295
|
+
pause: options.videoPause,
|
|
296
|
+
signal: controller.signal,
|
|
297
|
+
},
|
|
202
298
|
},
|
|
203
299
|
res,
|
|
204
300
|
map,
|
|
205
301
|
usageBox,
|
|
302
|
+
() => {
|
|
303
|
+
responseStarted = true;
|
|
304
|
+
},
|
|
206
305
|
);
|
|
207
306
|
} finally {
|
|
208
307
|
cacheUsage = usageBox.cacheUsage ?? null;
|
package/src/imagine.mjs
ADDED
|
@@ -0,0 +1,150 @@
|
|
|
1
|
+
// Codex posts OpenAI Images requests on the provider base
|
|
2
|
+
// (`POST /v1/images/generations`, `POST /v1/images/edits`). grok-build sends
|
|
3
|
+
// the same two paths to the Imagine API with the login bearer, not an
|
|
4
|
+
// XAI_API_KEY. The bodies are not the same: Codex hardcodes `gpt-image-2`
|
|
5
|
+
// plus `size` / `quality` / `background`, and edits use `{image_url}` or
|
|
6
|
+
// `{file_id}`. Imagine wants `grok-imagine-image-quality`, `response_format:
|
|
7
|
+
// b64_json`, and `{url}` references.
|
|
8
|
+
|
|
9
|
+
export const DEFAULT_IMAGINE_API_BASE = "https://api.x.ai/v1";
|
|
10
|
+
export const IMAGINE_MODEL = "grok-imagine-image-quality";
|
|
11
|
+
|
|
12
|
+
const IMAGINE_TIMEOUT_MS = 300_000;
|
|
13
|
+
|
|
14
|
+
function promptOf(input) {
|
|
15
|
+
const prompt = typeof input?.prompt === "string" ? input.prompt.trim() : "";
|
|
16
|
+
if (!prompt) {
|
|
17
|
+
const error = new Error("Image generation requires a prompt.");
|
|
18
|
+
error.status = 400;
|
|
19
|
+
throw error;
|
|
20
|
+
}
|
|
21
|
+
return prompt;
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
function basePayload(prompt) {
|
|
25
|
+
return {
|
|
26
|
+
model: IMAGINE_MODEL,
|
|
27
|
+
prompt,
|
|
28
|
+
n: 1,
|
|
29
|
+
resolution: "1k",
|
|
30
|
+
response_format: "b64_json",
|
|
31
|
+
};
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
export function imagineGenerationBody(input) {
|
|
35
|
+
return basePayload(promptOf(input));
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
export function imagineEditBody(input) {
|
|
39
|
+
const prompt = promptOf(input);
|
|
40
|
+
const images = Array.isArray(input?.images) ? input.images : [];
|
|
41
|
+
const urls = [];
|
|
42
|
+
for (const image of images) {
|
|
43
|
+
if (!image || typeof image !== "object") continue;
|
|
44
|
+
if (typeof image.image_url === "string" && image.image_url.trim()) {
|
|
45
|
+
urls.push(image.image_url.trim());
|
|
46
|
+
continue;
|
|
47
|
+
}
|
|
48
|
+
if (typeof image.file_id === "string" && image.file_id.trim()) {
|
|
49
|
+
const error = new Error("OpenAI file ids cannot be edited with Grok.");
|
|
50
|
+
error.status = 400;
|
|
51
|
+
throw error;
|
|
52
|
+
}
|
|
53
|
+
}
|
|
54
|
+
if (!urls.length) {
|
|
55
|
+
const error = new Error("Image edit requires a reference image.");
|
|
56
|
+
error.status = 400;
|
|
57
|
+
throw error;
|
|
58
|
+
}
|
|
59
|
+
const payload = basePayload(prompt);
|
|
60
|
+
if (urls.length === 1) payload.image = { url: urls[0] };
|
|
61
|
+
else {
|
|
62
|
+
payload.images = urls.map((url) => ({ url }));
|
|
63
|
+
payload.aspect_ratio = "auto";
|
|
64
|
+
}
|
|
65
|
+
return payload;
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
export function codexImageResponse(payload) {
|
|
69
|
+
const data = [];
|
|
70
|
+
if (Array.isArray(payload?.data)) {
|
|
71
|
+
for (const item of payload.data) {
|
|
72
|
+
if (item && typeof item.b64_json === "string" && item.b64_json.length)
|
|
73
|
+
data.push({ b64_json: item.b64_json });
|
|
74
|
+
}
|
|
75
|
+
}
|
|
76
|
+
if (!data.length) {
|
|
77
|
+
const error = new Error("Image generation returned no image.");
|
|
78
|
+
error.status = 502;
|
|
79
|
+
throw error;
|
|
80
|
+
}
|
|
81
|
+
const created = Number.isInteger(payload?.created)
|
|
82
|
+
? payload.created
|
|
83
|
+
: Math.floor(Date.now() / 1000);
|
|
84
|
+
return { created, data };
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
function scrub(text, token) {
|
|
88
|
+
let out = String(text ?? "");
|
|
89
|
+
out = out.replace(/Bearer\s+\S+/gi, "Bearer [redacted]");
|
|
90
|
+
if (token) out = out.split(token).join("[redacted]");
|
|
91
|
+
return out;
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
async function imagineHttp(fetchImpl, url, init) {
|
|
95
|
+
const response = await fetchImpl(url, { ...init, redirect: "manual" });
|
|
96
|
+
const text = await response.text().catch(() => "");
|
|
97
|
+
if (response.status >= 300 && response.status < 400) {
|
|
98
|
+
const error = new Error("Image generation failed (redirect).");
|
|
99
|
+
error.status = 502;
|
|
100
|
+
throw error;
|
|
101
|
+
}
|
|
102
|
+
let body = null;
|
|
103
|
+
if (text) {
|
|
104
|
+
try {
|
|
105
|
+
body = JSON.parse(text);
|
|
106
|
+
} catch {
|
|
107
|
+
body = null;
|
|
108
|
+
}
|
|
109
|
+
}
|
|
110
|
+
return { status: response.status, body };
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
export async function forwardImagine(options) {
|
|
114
|
+
const token = options.token;
|
|
115
|
+
const kind = options.kind === "edits" ? "edits" : "generations";
|
|
116
|
+
const payload =
|
|
117
|
+
kind === "edits" ? imagineEditBody(options.body) : imagineGenerationBody(options.body);
|
|
118
|
+
const fetchImpl = options.fetchImpl ?? fetch;
|
|
119
|
+
const base = (options.baseUrl ?? DEFAULT_IMAGINE_API_BASE).replace(/\/$/, "");
|
|
120
|
+
const timeout = AbortSignal.timeout(IMAGINE_TIMEOUT_MS);
|
|
121
|
+
const signal = options.signal ? AbortSignal.any([options.signal, timeout]) : timeout;
|
|
122
|
+
let result;
|
|
123
|
+
try {
|
|
124
|
+
result = await imagineHttp(fetchImpl, `${base}/images/${kind}`, {
|
|
125
|
+
method: "POST",
|
|
126
|
+
headers: {
|
|
127
|
+
authorization: `Bearer ${token}`,
|
|
128
|
+
"content-type": "application/json",
|
|
129
|
+
},
|
|
130
|
+
body: JSON.stringify(payload),
|
|
131
|
+
signal,
|
|
132
|
+
});
|
|
133
|
+
} catch (error) {
|
|
134
|
+
if (error?.status) throw error;
|
|
135
|
+
const wrapped = new Error(scrub(error?.message || "Image generation failed.", token));
|
|
136
|
+
wrapped.status = 502;
|
|
137
|
+
throw wrapped;
|
|
138
|
+
}
|
|
139
|
+
if (result.status === 401 || result.status === 403) {
|
|
140
|
+
const error = new Error("Grok login expired. Run grok login.");
|
|
141
|
+
error.status = 401;
|
|
142
|
+
throw error;
|
|
143
|
+
}
|
|
144
|
+
if (result.status < 200 || result.status >= 300) {
|
|
145
|
+
const error = new Error(`Image generation failed (${result.status}).`);
|
|
146
|
+
error.status = result.status >= 400 && result.status < 500 ? result.status : 502;
|
|
147
|
+
throw error;
|
|
148
|
+
}
|
|
149
|
+
return codexImageResponse(result.body);
|
|
150
|
+
}
|
package/src/proxy.mjs
CHANGED
|
@@ -261,13 +261,13 @@ export async function pipeProxySse(stream, output, map, usageBox, onClientByte)
|
|
|
261
261
|
|
|
262
262
|
// Open and read one attempt. If the socket dies before Codex has a byte, send
|
|
263
263
|
// the same request again. Once a byte has been written, never resubmit.
|
|
264
|
-
|
|
264
|
+
async function relayAttempts(options, consume, attempts = PROXY_ATTEMPTS) {
|
|
265
265
|
const rounds = Math.max(1, attempts);
|
|
266
266
|
let lastError;
|
|
267
267
|
for (let attempt = 1; attempt <= rounds; attempt += 1) {
|
|
268
268
|
try {
|
|
269
269
|
const response = await openProxyStream(options);
|
|
270
|
-
await
|
|
270
|
+
await consume(response);
|
|
271
271
|
return response;
|
|
272
272
|
} catch (error) {
|
|
273
273
|
lastError = error;
|
|
@@ -284,3 +284,88 @@ export async function relayProxySse(options, output, map, usageBox, attempts = P
|
|
|
284
284
|
}
|
|
285
285
|
throw lastError;
|
|
286
286
|
}
|
|
287
|
+
|
|
288
|
+
export async function relayProxySse(options, output, map, usageBox, attempts = PROXY_ATTEMPTS) {
|
|
289
|
+
return relayAttempts(
|
|
290
|
+
options,
|
|
291
|
+
(response) =>
|
|
292
|
+
pipeProxySse(response.body, output, map, usageBox, options.onClientByte),
|
|
293
|
+
attempts,
|
|
294
|
+
);
|
|
295
|
+
}
|
|
296
|
+
|
|
297
|
+
// Read until response.completed (or the other terminal events). Do not write
|
|
298
|
+
// those bytes yet: a video tool call has to be answered before Codex sees the
|
|
299
|
+
// turn, and a reset before that write is still safe to retry.
|
|
300
|
+
export async function readUntilTerminal(stream, output) {
|
|
301
|
+
const reader = stream.getReader();
|
|
302
|
+
const decoder = new TextDecoder();
|
|
303
|
+
let buffer = "";
|
|
304
|
+
const collected = [];
|
|
305
|
+
let released = false;
|
|
306
|
+
const finish = async (error) => {
|
|
307
|
+
if (released) return;
|
|
308
|
+
released = true;
|
|
309
|
+
await reader.cancel(error).catch(() => {});
|
|
310
|
+
};
|
|
311
|
+
try {
|
|
312
|
+
while (true) {
|
|
313
|
+
if (output?.destroyed || output?.writableEnded) throw clientClosed();
|
|
314
|
+
const { done, value } = await reader.read();
|
|
315
|
+
if (done) break;
|
|
316
|
+
buffer += decoder.decode(value, { stream: true });
|
|
317
|
+
const split = completeBlocks(buffer);
|
|
318
|
+
buffer = split.rest;
|
|
319
|
+
collected.push(...split.parts);
|
|
320
|
+
if (collected.some((part) => isTerminalSseBlock(part))) {
|
|
321
|
+
await finish();
|
|
322
|
+
return collected;
|
|
323
|
+
}
|
|
324
|
+
}
|
|
325
|
+
buffer += decoder.decode();
|
|
326
|
+
if (buffer.trim()) collected.push(buffer);
|
|
327
|
+
if (!collected.some((part) => isTerminalSseBlock(part)))
|
|
328
|
+
throw prematureUpstreamClose();
|
|
329
|
+
await finish();
|
|
330
|
+
return collected;
|
|
331
|
+
} catch (error) {
|
|
332
|
+
await finish(error);
|
|
333
|
+
if (error && typeof error === "object" && error.clientBytes == null)
|
|
334
|
+
error.clientBytes = 0;
|
|
335
|
+
throw error;
|
|
336
|
+
}
|
|
337
|
+
}
|
|
338
|
+
|
|
339
|
+
export async function readRelayProxySse(options, output, attempts = PROXY_ATTEMPTS) {
|
|
340
|
+
let parts;
|
|
341
|
+
await relayAttempts(
|
|
342
|
+
options,
|
|
343
|
+
async (response) => {
|
|
344
|
+
parts = await readUntilTerminal(response.body, output);
|
|
345
|
+
},
|
|
346
|
+
attempts,
|
|
347
|
+
);
|
|
348
|
+
return parts;
|
|
349
|
+
}
|
|
350
|
+
|
|
351
|
+
export async function emitProxySse(parts, output, map, usageBox, onClientByte) {
|
|
352
|
+
const rewrite = createSseRewriter(map);
|
|
353
|
+
let clientBytes = 0;
|
|
354
|
+
try {
|
|
355
|
+
for (const part of parts) {
|
|
356
|
+
const rewritten = rewrite(part);
|
|
357
|
+
if (rewritten === null) continue;
|
|
358
|
+
const block = `${rewritten}\n\n`;
|
|
359
|
+
if (output.destroyed || output.writableEnded) throw clientClosed();
|
|
360
|
+
const accepted = output.write(block);
|
|
361
|
+
clientBytes += Buffer.byteLength(block);
|
|
362
|
+
onClientByte?.();
|
|
363
|
+
if (!accepted) await drained(output);
|
|
364
|
+
}
|
|
365
|
+
} catch (error) {
|
|
366
|
+
if (error && typeof error === "object") error.clientBytes = clientBytes;
|
|
367
|
+
throw error;
|
|
368
|
+
} finally {
|
|
369
|
+
if (usageBox) usageBox.cacheUsage = rewrite.cacheUsage ?? null;
|
|
370
|
+
}
|
|
371
|
+
}
|
package/src/tools.mjs
CHANGED
|
@@ -4,6 +4,7 @@ import {
|
|
|
4
4
|
isImageGenerationItem,
|
|
5
5
|
saveGeneratedImage,
|
|
6
6
|
} from "./imagegen.mjs";
|
|
7
|
+
import { GROK_VIDEO_TOOL, GROK_VIDEO_TOOL_NAME } from "./videogen.mjs";
|
|
7
8
|
import { DEFAULT_GROK_MODEL, resolveGrokModel } from "./models.mjs";
|
|
8
9
|
import {
|
|
9
10
|
applyCacheUsage,
|
|
@@ -27,14 +28,6 @@ const CUSTOM_INPUT_SCHEMA = {
|
|
|
27
28
|
additionalProperties: false,
|
|
28
29
|
};
|
|
29
30
|
|
|
30
|
-
const WEB_SEARCH_SCHEMA = {
|
|
31
|
-
type: "object",
|
|
32
|
-
properties: {
|
|
33
|
-
query: { type: "string", description: "Search query" },
|
|
34
|
-
},
|
|
35
|
-
required: ["query"],
|
|
36
|
-
};
|
|
37
|
-
|
|
38
31
|
const EFFORT = {
|
|
39
32
|
ultra: "xhigh",
|
|
40
33
|
max: "xhigh",
|
|
@@ -50,12 +43,14 @@ const DROP = Symbol("drop");
|
|
|
50
43
|
// Items whose payload only the originating provider can read. "reasoning" is
|
|
51
44
|
// deliberately absent: Codex reasoning items carry a plain-text summary that is
|
|
52
45
|
// useful to Grok across a multi-call turn, and only their encrypted_content is
|
|
53
|
-
// opaque - that field is stripped by the key filter below.
|
|
54
|
-
|
|
46
|
+
// opaque - that field is stripped by the key filter below. Compaction items are
|
|
47
|
+
// likewise absent: a stored summary is rewritten into a user message, and an
|
|
48
|
+
// item with no readable text is dropped.
|
|
49
|
+
const PROVIDER_OPAQUE_TYPES = new Set(["encrypted_content"]);
|
|
50
|
+
const COMPACTION_ITEM_TYPES = new Set([
|
|
55
51
|
"compaction",
|
|
56
52
|
"compaction_summary",
|
|
57
53
|
"context_compaction",
|
|
58
|
-
"encrypted_content",
|
|
59
54
|
]);
|
|
60
55
|
const GROK_INPUT_ITEM_TYPES = new Set([
|
|
61
56
|
"message",
|
|
@@ -233,6 +228,21 @@ function decodeCustomInput(argumentsValue) {
|
|
|
233
228
|
return argumentsValue;
|
|
234
229
|
}
|
|
235
230
|
|
|
231
|
+
function plainObject(value) {
|
|
232
|
+
return Boolean(value) && typeof value === "object" && !Array.isArray(value);
|
|
233
|
+
}
|
|
234
|
+
|
|
235
|
+
// xAI runs `{type:"web_search"}` on the server. A hashed function is never
|
|
236
|
+
// called, so Codex's hosted tool stays one server tool. Keep filters and
|
|
237
|
+
// allowed_domains; every other field on that tool is dropped.
|
|
238
|
+
function keptWebSearchFields(tool) {
|
|
239
|
+
const next = {};
|
|
240
|
+
if (plainObject(tool.filters)) next.filters = tool.filters;
|
|
241
|
+
if (Array.isArray(tool.allowed_domains))
|
|
242
|
+
next.allowed_domains = tool.allowed_domains;
|
|
243
|
+
return next;
|
|
244
|
+
}
|
|
245
|
+
|
|
236
246
|
function addFunction(flattened, map, spec) {
|
|
237
247
|
// Index names change when Codex reorders tools, which rewrites every
|
|
238
248
|
// earlier function_call and breaks the prompt-cache prefix.
|
|
@@ -264,8 +274,25 @@ function addFunction(flattened, map, spec) {
|
|
|
264
274
|
export function flattenCodexTools(tools = []) {
|
|
265
275
|
const flattened = [];
|
|
266
276
|
const map = new Map();
|
|
277
|
+
let webSearch = null;
|
|
267
278
|
for (const tool of tools) {
|
|
268
279
|
if (!tool || typeof tool !== "object") continue;
|
|
280
|
+
if (tool.type === "web_search") {
|
|
281
|
+
const kept = keptWebSearchFields(tool);
|
|
282
|
+
if (!webSearch) {
|
|
283
|
+
webSearch = { type: "web_search", ...kept };
|
|
284
|
+
flattened.push(webSearch);
|
|
285
|
+
} else {
|
|
286
|
+
if (webSearch.filters === undefined && kept.filters !== undefined)
|
|
287
|
+
webSearch.filters = kept.filters;
|
|
288
|
+
if (
|
|
289
|
+
webSearch.allowed_domains === undefined &&
|
|
290
|
+
kept.allowed_domains !== undefined
|
|
291
|
+
)
|
|
292
|
+
webSearch.allowed_domains = kept.allowed_domains;
|
|
293
|
+
}
|
|
294
|
+
continue;
|
|
295
|
+
}
|
|
269
296
|
if (tool.type === "function" || tool.type === "custom") {
|
|
270
297
|
addFunction(flattened, map, {
|
|
271
298
|
kind: tool.type === "custom" ? "custom" : "function",
|
|
@@ -276,16 +303,6 @@ export function flattenCodexTools(tools = []) {
|
|
|
276
303
|
});
|
|
277
304
|
continue;
|
|
278
305
|
}
|
|
279
|
-
if (tool.type === "web_search") {
|
|
280
|
-
addFunction(flattened, map, {
|
|
281
|
-
kind: "web_search",
|
|
282
|
-
namespace: null,
|
|
283
|
-
name: "web_search",
|
|
284
|
-
description: tool.description || "Search the web",
|
|
285
|
-
parameters: tool.parameters || WEB_SEARCH_SCHEMA,
|
|
286
|
-
});
|
|
287
|
-
continue;
|
|
288
|
-
}
|
|
289
306
|
if (tool.type === "namespace" && Array.isArray(tool.tools)) {
|
|
290
307
|
for (const nested of tool.tools) {
|
|
291
308
|
if (!nested || typeof nested !== "object") continue;
|
|
@@ -301,6 +318,10 @@ export function flattenCodexTools(tools = []) {
|
|
|
301
318
|
}
|
|
302
319
|
flattened.push(stableJsonValue(structuredClone(tool)));
|
|
303
320
|
}
|
|
321
|
+
if (webSearch) {
|
|
322
|
+
const index = flattened.indexOf(webSearch);
|
|
323
|
+
flattened[index] = stableJsonValue(structuredClone(webSearch));
|
|
324
|
+
}
|
|
304
325
|
return { tools: flattened, map };
|
|
305
326
|
}
|
|
306
327
|
|
|
@@ -316,7 +337,178 @@ export function proxyToolChoice(choice, map) {
|
|
|
316
337
|
return "auto";
|
|
317
338
|
}
|
|
318
339
|
|
|
319
|
-
function
|
|
340
|
+
function readableString(value) {
|
|
341
|
+
return typeof value === "string" && value.trim() ? value : null;
|
|
342
|
+
}
|
|
343
|
+
|
|
344
|
+
function textsFromValue(value) {
|
|
345
|
+
const direct = readableString(value);
|
|
346
|
+
if (direct) return [direct];
|
|
347
|
+
if (!Array.isArray(value)) return [];
|
|
348
|
+
const texts = [];
|
|
349
|
+
for (const part of value) {
|
|
350
|
+
if (!part || typeof part !== "object" || part.type === "encrypted_content")
|
|
351
|
+
continue;
|
|
352
|
+
const text = readableString(part.text);
|
|
353
|
+
if (text) texts.push(text);
|
|
354
|
+
}
|
|
355
|
+
return texts;
|
|
356
|
+
}
|
|
357
|
+
|
|
358
|
+
// Codex replaces older history with a compaction item. The encrypted blob is
|
|
359
|
+
// provider-private, but the item may already carry the summary as plain text.
|
|
360
|
+
// That text is what Grok can read; everything else on the item is dropped.
|
|
361
|
+
function compactionInputTexts(node) {
|
|
362
|
+
for (const key of ["summary", "content", "text", "message"]) {
|
|
363
|
+
const texts = textsFromValue(node[key]);
|
|
364
|
+
if (texts.length) return texts;
|
|
365
|
+
}
|
|
366
|
+
return [];
|
|
367
|
+
}
|
|
368
|
+
|
|
369
|
+
function rememberText(texts, seen, value) {
|
|
370
|
+
const text = readableString(value);
|
|
371
|
+
if (!text || seen.has(text)) return;
|
|
372
|
+
seen.add(text);
|
|
373
|
+
texts.push(text);
|
|
374
|
+
}
|
|
375
|
+
|
|
376
|
+
// Content parts, stdout/stderr rows, and bare string lists. Encrypted parts
|
|
377
|
+
// are skipped. This does not walk arbitrary keys, so ids, status, env, and
|
|
378
|
+
// image bytes are not treated as text.
|
|
379
|
+
function rememberTextValue(texts, seen, value) {
|
|
380
|
+
if (typeof value === "string") {
|
|
381
|
+
rememberText(texts, seen, value);
|
|
382
|
+
return;
|
|
383
|
+
}
|
|
384
|
+
if (Array.isArray(value)) {
|
|
385
|
+
for (const part of value) {
|
|
386
|
+
if (typeof part === "string") {
|
|
387
|
+
rememberText(texts, seen, part);
|
|
388
|
+
continue;
|
|
389
|
+
}
|
|
390
|
+
if (!part || typeof part !== "object" || part.type === "encrypted_content")
|
|
391
|
+
continue;
|
|
392
|
+
rememberText(texts, seen, part.text);
|
|
393
|
+
rememberText(texts, seen, part.stdout);
|
|
394
|
+
rememberText(texts, seen, part.stderr);
|
|
395
|
+
}
|
|
396
|
+
return;
|
|
397
|
+
}
|
|
398
|
+
if (!value || typeof value !== "object") return;
|
|
399
|
+
rememberText(texts, seen, value.text);
|
|
400
|
+
rememberText(texts, seen, value.stdout);
|
|
401
|
+
rememberText(texts, seen, value.stderr);
|
|
402
|
+
if (typeof value.content === "string" || Array.isArray(value.content))
|
|
403
|
+
rememberTextValue(texts, seen, value.content);
|
|
404
|
+
}
|
|
405
|
+
|
|
406
|
+
function commandLines(action) {
|
|
407
|
+
const lines = [];
|
|
408
|
+
for (const list of [action.command, action.commands]) {
|
|
409
|
+
if (Array.isArray(list)) {
|
|
410
|
+
const line = list.filter((part) => typeof part === "string").join(" ").trim();
|
|
411
|
+
if (line) lines.push(line);
|
|
412
|
+
} else if (typeof list === "string" && list.trim()) {
|
|
413
|
+
lines.push(list.trim());
|
|
414
|
+
}
|
|
415
|
+
}
|
|
416
|
+
return lines;
|
|
417
|
+
}
|
|
418
|
+
|
|
419
|
+
function rememberArguments(texts, seen, value) {
|
|
420
|
+
if (typeof value === "string") {
|
|
421
|
+
try {
|
|
422
|
+
const parsed = JSON.parse(value);
|
|
423
|
+
if (parsed && typeof parsed === "object" && !Array.isArray(parsed))
|
|
424
|
+
value = parsed;
|
|
425
|
+
else {
|
|
426
|
+
rememberText(texts, seen, value);
|
|
427
|
+
return;
|
|
428
|
+
}
|
|
429
|
+
} catch {
|
|
430
|
+
rememberText(texts, seen, value);
|
|
431
|
+
return;
|
|
432
|
+
}
|
|
433
|
+
}
|
|
434
|
+
if (!value || typeof value !== "object" || Array.isArray(value)) return;
|
|
435
|
+
rememberText(texts, seen, value.query);
|
|
436
|
+
rememberTextValue(texts, seen, value.queries);
|
|
437
|
+
}
|
|
438
|
+
|
|
439
|
+
function stripOpaqueFields(value) {
|
|
440
|
+
if (Array.isArray(value)) return value.map(stripOpaqueFields);
|
|
441
|
+
if (!value || typeof value !== "object") return value;
|
|
442
|
+
const next = {};
|
|
443
|
+
for (const [key, child] of Object.entries(value)) {
|
|
444
|
+
if (
|
|
445
|
+
key === "encrypted_content" ||
|
|
446
|
+
key === "encrypted_function_args" ||
|
|
447
|
+
key.startsWith("internal_")
|
|
448
|
+
)
|
|
449
|
+
continue;
|
|
450
|
+
next[key] = stripOpaqueFields(child);
|
|
451
|
+
}
|
|
452
|
+
return next;
|
|
453
|
+
}
|
|
454
|
+
|
|
455
|
+
function containsReadableText(value) {
|
|
456
|
+
if (typeof value === "string") return Boolean(value.trim());
|
|
457
|
+
if (Array.isArray(value)) return value.some(containsReadableText);
|
|
458
|
+
if (value && typeof value === "object")
|
|
459
|
+
return Object.values(value).some(containsReadableText);
|
|
460
|
+
return false;
|
|
461
|
+
}
|
|
462
|
+
|
|
463
|
+
function rememberTools(texts, seen, tools) {
|
|
464
|
+
if (!Array.isArray(tools) || !tools.length) return;
|
|
465
|
+
const stripped = stripOpaqueFields(tools);
|
|
466
|
+
if (!containsReadableText(stripped)) return;
|
|
467
|
+
rememberText(texts, seen, JSON.stringify(stripped));
|
|
468
|
+
}
|
|
469
|
+
|
|
470
|
+
// Text Grok can read off an item type it will not accept. The caller turns
|
|
471
|
+
// these strings into a user message. Fields that 422 when forwarded
|
|
472
|
+
// (encrypted blobs, internal metadata, image bytes) are not read.
|
|
473
|
+
function droppedItemInputTexts(node) {
|
|
474
|
+
const texts = [];
|
|
475
|
+
const seen = new Set();
|
|
476
|
+
for (const key of [
|
|
477
|
+
"summary",
|
|
478
|
+
"content",
|
|
479
|
+
"text",
|
|
480
|
+
"message",
|
|
481
|
+
"input",
|
|
482
|
+
"output",
|
|
483
|
+
"query",
|
|
484
|
+
"revised_prompt",
|
|
485
|
+
]) {
|
|
486
|
+
rememberTextValue(texts, seen, node[key]);
|
|
487
|
+
}
|
|
488
|
+
const action = node.action;
|
|
489
|
+
if (action && typeof action === "object" && !Array.isArray(action)) {
|
|
490
|
+
const lines = commandLines(action);
|
|
491
|
+
for (const line of lines) rememberText(texts, seen, line);
|
|
492
|
+
if (lines.length) rememberText(texts, seen, action.working_directory);
|
|
493
|
+
rememberText(texts, seen, action.query);
|
|
494
|
+
rememberTextValue(texts, seen, action.queries);
|
|
495
|
+
rememberText(texts, seen, action.url);
|
|
496
|
+
rememberText(texts, seen, action.pattern);
|
|
497
|
+
}
|
|
498
|
+
rememberArguments(texts, seen, node.arguments);
|
|
499
|
+
rememberTools(texts, seen, node.tools);
|
|
500
|
+
return texts;
|
|
501
|
+
}
|
|
502
|
+
|
|
503
|
+
function userInputTextMessage(texts) {
|
|
504
|
+
return whitelistInputNode({
|
|
505
|
+
type: "message",
|
|
506
|
+
role: "user",
|
|
507
|
+
content: texts.map((text) => ({ type: "input_text", text })),
|
|
508
|
+
});
|
|
509
|
+
}
|
|
510
|
+
|
|
511
|
+
function toProxyInputNode(node, map, state, salvage = false) {
|
|
320
512
|
if (!node || typeof node !== "object") return node;
|
|
321
513
|
if (Array.isArray(node)) {
|
|
322
514
|
const items = [];
|
|
@@ -342,6 +534,16 @@ function toProxyInputNode(node, map, state) {
|
|
|
342
534
|
if (converted !== DROP) next[key] = converted;
|
|
343
535
|
}
|
|
344
536
|
|
|
537
|
+
if (COMPACTION_ITEM_TYPES.has(next.type)) {
|
|
538
|
+
const texts = compactionInputTexts(next);
|
|
539
|
+
if (!texts.length) return DROP;
|
|
540
|
+
return whitelistInputNode({
|
|
541
|
+
type: "message",
|
|
542
|
+
role: "user",
|
|
543
|
+
content: texts.map((text) => ({ type: "input_text", text })),
|
|
544
|
+
});
|
|
545
|
+
}
|
|
546
|
+
|
|
345
547
|
if (next.type === "reasoning") {
|
|
346
548
|
// Forward the summary and nothing else. Codex's own item id and null
|
|
347
549
|
// content fields mean nothing to the upstream and only widen the surface
|
|
@@ -404,7 +606,14 @@ function toProxyInputNode(node, map, state) {
|
|
|
404
606
|
}
|
|
405
607
|
}
|
|
406
608
|
|
|
407
|
-
|
|
609
|
+
const whitelisted = whitelistInputNode(next);
|
|
610
|
+
if (!salvage || isForwardedItem(whitelisted)) return whitelisted;
|
|
611
|
+
// Grok rejects these Codex item types. Keep the readable text as a user
|
|
612
|
+
// message, the same shape as a compaction summary. An item with nothing
|
|
613
|
+
// left but an encrypted blob, image bytes, or ids is dropped.
|
|
614
|
+
const texts = droppedItemInputTexts(next);
|
|
615
|
+
if (!texts.length) return DROP;
|
|
616
|
+
return userInputTextMessage(texts);
|
|
408
617
|
}
|
|
409
618
|
|
|
410
619
|
function isForwardedItem(item) {
|
|
@@ -419,14 +628,14 @@ function isForwardedItem(item) {
|
|
|
419
628
|
|
|
420
629
|
function projectInput(input, map) {
|
|
421
630
|
if (!Array.isArray(input)) {
|
|
422
|
-
const items = toProxyInputNode(input, map, { callIds: new Map() });
|
|
631
|
+
const items = toProxyInputNode(input, map, { callIds: new Map() }, true);
|
|
423
632
|
return { items, pairs: [] };
|
|
424
633
|
}
|
|
425
634
|
const state = { callIds: new Map() };
|
|
426
635
|
const pairs = [];
|
|
427
636
|
for (const item of input) {
|
|
428
637
|
const fingerprint = logicalFingerprint(item);
|
|
429
|
-
const next = toProxyInputNode(item, map, state);
|
|
638
|
+
const next = toProxyInputNode(item, map, state, true);
|
|
430
639
|
if (!isForwardedItem(next)) continue;
|
|
431
640
|
pairs.push({ fingerprint, item: next });
|
|
432
641
|
}
|
|
@@ -463,6 +672,14 @@ export const IMAGE_GENERATION_PROVENANCE =
|
|
|
463
672
|
"and do not use OpenAI, image_gen, or a Python image CLI — those send the " +
|
|
464
673
|
"picture to another vendor.";
|
|
465
674
|
|
|
675
|
+
// Codex does not offer a video tool to this provider, and cli-chat-proxy has
|
|
676
|
+
// no video_generation tool type. The function below is the bridge's own.
|
|
677
|
+
export const VIDEO_GENERATION_PROVENANCE =
|
|
678
|
+
"Videos: clips on this transport are generated by the grok_bridge_generate_video " +
|
|
679
|
+
"function, which is already on the request. The bridge calls Grok's videos API " +
|
|
680
|
+
"with the grok login and returns the video URL. Do not ask for an API key and " +
|
|
681
|
+
"do not send the clip to another vendor.";
|
|
682
|
+
|
|
466
683
|
export function toProxyRequest(body) {
|
|
467
684
|
const { tools, map } = flattenCodexTools(body.tools ?? []);
|
|
468
685
|
const effort = EFFORT[body.reasoning?.effort] ?? "high";
|
|
@@ -484,13 +701,25 @@ export function toProxyRequest(body) {
|
|
|
484
701
|
const imageGenOn = process.env.GROK_BRIDGE_IMAGE_GEN !== "off";
|
|
485
702
|
if (imageGenOn && !declaresImageTool)
|
|
486
703
|
request.tools = [...request.tools, GROK_IMAGE_TOOL];
|
|
704
|
+
// Codex tool names are rewritten to codex_<hash>_…, so this literal name
|
|
705
|
+
// cannot collide with one. A second copy is not added if it is already there.
|
|
706
|
+
const videoGenOn = process.env.GROK_BRIDGE_VIDEO_GEN !== "off";
|
|
707
|
+
const declaresVideoTool = request.tools.some(
|
|
708
|
+
(tool) => tool?.name === GROK_VIDEO_TOOL_NAME,
|
|
709
|
+
);
|
|
710
|
+
if (videoGenOn && !declaresVideoTool)
|
|
711
|
+
request.tools = [...request.tools, GROK_VIDEO_TOOL];
|
|
487
712
|
if (tools.length) {
|
|
488
713
|
request.tool_choice = proxyToolChoice(body.tool_choice, map);
|
|
489
714
|
request.parallel_tool_calls = body.parallel_tool_calls !== false;
|
|
490
715
|
}
|
|
491
716
|
const provenanceLine = transportProvenance(request.model);
|
|
492
|
-
const
|
|
493
|
-
?
|
|
717
|
+
const notes = [
|
|
718
|
+
imageGenOn ? IMAGE_GENERATION_PROVENANCE : "",
|
|
719
|
+
videoGenOn ? VIDEO_GENERATION_PROVENANCE : "",
|
|
720
|
+
].filter(Boolean);
|
|
721
|
+
const provenance = notes.length
|
|
722
|
+
? `${provenanceLine}\n\n${notes.join("\n\n")}`
|
|
494
723
|
: provenanceLine;
|
|
495
724
|
request.instructions =
|
|
496
725
|
typeof body.instructions === "string" && body.instructions
|
package/src/videogen.mjs
ADDED
|
@@ -0,0 +1,196 @@
|
|
|
1
|
+
// cli-chat-proxy has no Responses `video_generation` tool. grok-build generates
|
|
2
|
+
// video by calling the videos API itself when the model invokes a function
|
|
3
|
+
// (`image_to_video` / `reference_to_video`): POST {xai_api_base}/videos/generations,
|
|
4
|
+
// then GET /videos/{request_id}. The bridge does the same with the grok login
|
|
5
|
+
// bearer it already sends to cli-chat-proxy — the `key` from that session, not
|
|
6
|
+
// an XAI_API_KEY.
|
|
7
|
+
|
|
8
|
+
export const GROK_VIDEO_TOOL_NAME = "grok_bridge_generate_video";
|
|
9
|
+
|
|
10
|
+
export const DEFAULT_VIDEO_API_BASE = "https://api.x.ai/v1";
|
|
11
|
+
|
|
12
|
+
const VIDEO_MODEL = "grok-imagine-video-1.5";
|
|
13
|
+
const POLL_LIMIT = 60;
|
|
14
|
+
|
|
15
|
+
export const GROK_VIDEO_TOOL = Object.freeze({
|
|
16
|
+
type: "function",
|
|
17
|
+
name: GROK_VIDEO_TOOL_NAME,
|
|
18
|
+
description:
|
|
19
|
+
"Generate a short video with Grok. The bridge runs this tool itself; Codex does not. " +
|
|
20
|
+
"Provide a prompt. Provide image_url only when the clip should start from that still image.",
|
|
21
|
+
parameters: {
|
|
22
|
+
type: "object",
|
|
23
|
+
properties: {
|
|
24
|
+
prompt: { type: "string", description: "What the video should show." },
|
|
25
|
+
image_url: {
|
|
26
|
+
type: "string",
|
|
27
|
+
description: "Optional http(s) or data URL of a still image to animate.",
|
|
28
|
+
},
|
|
29
|
+
duration: {
|
|
30
|
+
type: "integer",
|
|
31
|
+
description: "Length in seconds, from 1 to 15.",
|
|
32
|
+
},
|
|
33
|
+
aspect_ratio: {
|
|
34
|
+
type: "string",
|
|
35
|
+
description: "One of 1:1, 16:9, 9:16, 4:3, 3:4, 3:2, 2:3.",
|
|
36
|
+
},
|
|
37
|
+
},
|
|
38
|
+
required: ["prompt"],
|
|
39
|
+
additionalProperties: false,
|
|
40
|
+
},
|
|
41
|
+
});
|
|
42
|
+
|
|
43
|
+
function scrub(text, token) {
|
|
44
|
+
let out = String(text ?? "");
|
|
45
|
+
out = out.replace(/Bearer\s+\S+/gi, "Bearer [redacted]");
|
|
46
|
+
if (token) out = out.split(token).join("[redacted]");
|
|
47
|
+
return out;
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
function parseArgs(value) {
|
|
51
|
+
if (value && typeof value === "object") return value;
|
|
52
|
+
if (typeof value !== "string" || !value.trim()) return {};
|
|
53
|
+
try {
|
|
54
|
+
const parsed = JSON.parse(value);
|
|
55
|
+
return parsed && typeof parsed === "object" ? parsed : {};
|
|
56
|
+
} catch {
|
|
57
|
+
return {};
|
|
58
|
+
}
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
function videoPayload(args) {
|
|
62
|
+
const payload = { model: VIDEO_MODEL, prompt: args.prompt };
|
|
63
|
+
if (typeof args.image_url === "string" && args.image_url.trim())
|
|
64
|
+
payload.image = { url: args.image_url.trim() };
|
|
65
|
+
if (Number.isInteger(args.duration)) payload.duration = args.duration;
|
|
66
|
+
else if (typeof args.duration === "string" && /^\d+$/.test(args.duration))
|
|
67
|
+
payload.duration = Number(args.duration);
|
|
68
|
+
if (typeof args.aspect_ratio === "string" && args.aspect_ratio.trim())
|
|
69
|
+
payload.aspect_ratio = args.aspect_ratio.trim();
|
|
70
|
+
return payload;
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
async function videoHttp(fetchImpl, url, init) {
|
|
74
|
+
const response = await fetchImpl(url, { ...init, redirect: "manual" });
|
|
75
|
+
const text = await response.text().catch(() => "");
|
|
76
|
+
if (response.status >= 300 && response.status < 400) {
|
|
77
|
+
const error = new Error("Video generation failed (redirect).");
|
|
78
|
+
error.status = response.status;
|
|
79
|
+
throw error;
|
|
80
|
+
}
|
|
81
|
+
let body = null;
|
|
82
|
+
if (text) {
|
|
83
|
+
try {
|
|
84
|
+
body = JSON.parse(text);
|
|
85
|
+
} catch {
|
|
86
|
+
body = null;
|
|
87
|
+
}
|
|
88
|
+
}
|
|
89
|
+
return { status: response.status, body };
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
function httpFailure(status) {
|
|
93
|
+
const error = new Error(`Video generation failed (${status}).`);
|
|
94
|
+
error.status = status;
|
|
95
|
+
return error;
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
export async function generateVideo(options) {
|
|
99
|
+
const fetchImpl = options.fetchImpl ?? fetch;
|
|
100
|
+
const base = (options.baseUrl ?? DEFAULT_VIDEO_API_BASE).replace(/\/$/, "");
|
|
101
|
+
const headers = {
|
|
102
|
+
authorization: `Bearer ${options.token}`,
|
|
103
|
+
"content-type": "application/json",
|
|
104
|
+
};
|
|
105
|
+
const started = await videoHttp(fetchImpl, `${base}/videos/generations`, {
|
|
106
|
+
method: "POST",
|
|
107
|
+
headers,
|
|
108
|
+
body: JSON.stringify(videoPayload(options)),
|
|
109
|
+
signal: options.signal,
|
|
110
|
+
});
|
|
111
|
+
if (started.status < 200 || started.status >= 300)
|
|
112
|
+
throw httpFailure(started.status);
|
|
113
|
+
const requestId = started.body?.request_id;
|
|
114
|
+
if (typeof requestId !== "string" || !requestId)
|
|
115
|
+
throw httpFailure(started.status);
|
|
116
|
+
const pause = options.pause ?? ((ms) => new Promise((resolve) => setTimeout(resolve, ms)));
|
|
117
|
+
for (let attempt = 0; attempt < POLL_LIMIT; attempt += 1) {
|
|
118
|
+
if (attempt > 0) await pause(5000);
|
|
119
|
+
const poll = await videoHttp(
|
|
120
|
+
fetchImpl,
|
|
121
|
+
`${base}/videos/${encodeURIComponent(requestId)}`,
|
|
122
|
+
{ method: "GET", headers, signal: options.signal },
|
|
123
|
+
);
|
|
124
|
+
if (poll.status === 202) continue;
|
|
125
|
+
if (poll.status < 200 || poll.status >= 300) throw httpFailure(poll.status);
|
|
126
|
+
const status = poll.body?.status;
|
|
127
|
+
if (status === "done") {
|
|
128
|
+
const url = poll.body?.video?.url;
|
|
129
|
+
if (typeof url !== "string" || !url)
|
|
130
|
+
throw new Error("Video generation finished without a URL.");
|
|
131
|
+
return url;
|
|
132
|
+
}
|
|
133
|
+
if (status === "failed" || status === "expired")
|
|
134
|
+
throw new Error(`Video generation ${status}.`);
|
|
135
|
+
}
|
|
136
|
+
throw new Error("Video generation timed out.");
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
export async function videoToolOutput(call, options) {
|
|
140
|
+
const token = options?.token;
|
|
141
|
+
try {
|
|
142
|
+
const args = parseArgs(call?.arguments);
|
|
143
|
+
const prompt = typeof args.prompt === "string" ? args.prompt.trim() : "";
|
|
144
|
+
if (!prompt) return "Video generation failed: a prompt is required.";
|
|
145
|
+
const url = await generateVideo({
|
|
146
|
+
token: options.token,
|
|
147
|
+
fetchImpl: options.fetchImpl,
|
|
148
|
+
baseUrl: options.baseUrl,
|
|
149
|
+
pause: options.pause,
|
|
150
|
+
signal: options.signal,
|
|
151
|
+
prompt,
|
|
152
|
+
image_url: args.image_url,
|
|
153
|
+
duration: args.duration,
|
|
154
|
+
aspect_ratio: args.aspect_ratio,
|
|
155
|
+
});
|
|
156
|
+
return scrub(`Video ready at ${url}`, token);
|
|
157
|
+
} catch (error) {
|
|
158
|
+
const status = Number(error?.status) || 0;
|
|
159
|
+
if (status === 401 || status === 403)
|
|
160
|
+
return `Video generation failed (${status}).`;
|
|
161
|
+
return "Video generation failed.";
|
|
162
|
+
}
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
export function videoCallsFromParts(parts) {
|
|
166
|
+
const calls = [];
|
|
167
|
+
const seen = new Set();
|
|
168
|
+
for (const part of parts ?? []) {
|
|
169
|
+
for (const line of String(part).split("\n")) {
|
|
170
|
+
if (!line.startsWith("data:")) continue;
|
|
171
|
+
const payload = line.slice(5).trim();
|
|
172
|
+
if (!payload || payload === "[DONE]") continue;
|
|
173
|
+
let value;
|
|
174
|
+
try {
|
|
175
|
+
value = JSON.parse(payload);
|
|
176
|
+
} catch {
|
|
177
|
+
continue;
|
|
178
|
+
}
|
|
179
|
+
const item = value?.item;
|
|
180
|
+
if (value?.type !== "response.output_item.done") continue;
|
|
181
|
+
if (item?.type !== "function_call" || item.name !== GROK_VIDEO_TOOL_NAME)
|
|
182
|
+
continue;
|
|
183
|
+
if (typeof item.call_id !== "string" || seen.has(item.call_id)) continue;
|
|
184
|
+
seen.add(item.call_id);
|
|
185
|
+
calls.push({
|
|
186
|
+
name: item.name,
|
|
187
|
+
call_id: item.call_id,
|
|
188
|
+
arguments:
|
|
189
|
+
typeof item.arguments === "string"
|
|
190
|
+
? item.arguments
|
|
191
|
+
: JSON.stringify(item.arguments ?? {}),
|
|
192
|
+
});
|
|
193
|
+
}
|
|
194
|
+
}
|
|
195
|
+
return calls;
|
|
196
|
+
}
|