@agentex/agent 0.0.39 → 0.0.40
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +16 -0
- package/README.md +2 -0
- package/dist/history/fs.d.ts.map +1 -1
- package/dist/history/fs.js +6 -12
- package/dist/history/fs.js.map +1 -1
- package/dist/providers/claude/attach.d.ts.map +1 -1
- package/dist/providers/claude/attach.js +4 -6
- package/dist/providers/claude/attach.js.map +1 -1
- package/dist/providers/claude/transcript.d.ts.map +1 -1
- package/dist/providers/claude/transcript.js +9 -20
- package/dist/providers/claude/transcript.js.map +1 -1
- package/dist/providers/codex/attach.d.ts.map +1 -1
- package/dist/providers/codex/attach.js +4 -0
- package/dist/providers/codex/attach.js.map +1 -1
- package/dist/providers/codex/history.d.ts.map +1 -1
- package/dist/providers/codex/history.js +11 -0
- package/dist/providers/codex/history.js.map +1 -1
- package/dist/providers/codex/transcript-normalize.d.ts +12 -3
- package/dist/providers/codex/transcript-normalize.d.ts.map +1 -1
- package/dist/providers/codex/transcript-normalize.js +87 -10
- package/dist/providers/codex/transcript-normalize.js.map +1 -1
- package/dist/providers/codex/transcript.d.ts.map +1 -1
- package/dist/providers/codex/transcript.js +5 -10
- package/dist/providers/codex/transcript.js.map +1 -1
- package/dist/providers/codex/usage-scanner.d.ts.map +1 -1
- package/dist/providers/codex/usage-scanner.js +4 -6
- package/dist/providers/codex/usage-scanner.js.map +1 -1
- package/dist/utils/jsonl-lines.d.ts +35 -0
- package/dist/utils/jsonl-lines.d.ts.map +1 -0
- package/dist/utils/jsonl-lines.js +61 -0
- package/dist/utils/jsonl-lines.js.map +1 -0
- package/package.json +1 -1
- package/src/history/fs.ts +6 -12
- package/src/providers/claude/attach.ts +4 -6
- package/src/providers/claude/transcript.ts +9 -22
- package/src/providers/codex/attach.ts +4 -0
- package/src/providers/codex/history.ts +11 -0
- package/src/providers/codex/transcript-normalize.ts +85 -10
- package/src/providers/codex/transcript.ts +5 -11
- package/src/providers/codex/usage-scanner.ts +4 -7
- package/src/utils/jsonl-lines.ts +80 -0
|
@@ -23,8 +23,8 @@ import { createReadStream } from "node:fs";
|
|
|
23
23
|
import { readdir, realpath, stat, open as fsOpen } from "node:fs/promises";
|
|
24
24
|
import * as os from "node:os";
|
|
25
25
|
import * as path from "node:path";
|
|
26
|
-
import * as readline from "node:readline";
|
|
27
26
|
|
|
27
|
+
import { readJsonlLines } from "../../utils/jsonl-lines.js";
|
|
28
28
|
import { getDefaultRuntimeHome, getRuntimeHomeEnvVar } from "../../utils/runtime-homes.js";
|
|
29
29
|
import type { FoundTranscript, StreamEvent, TranscriptOps } from "../../types.js";
|
|
30
30
|
import { parseStreamLine } from "./parse.js";
|
|
@@ -273,13 +273,12 @@ async function readCwdFromTranscript(filePath: string): Promise<string | null> {
|
|
|
273
273
|
if (!fh) return null;
|
|
274
274
|
|
|
275
275
|
try {
|
|
276
|
-
const stream = fh.createReadStream({
|
|
277
|
-
const rl = readline.createInterface({ input: stream, crlfDelay: Infinity });
|
|
276
|
+
const stream = fh.createReadStream({ autoClose: false });
|
|
278
277
|
let count = 0;
|
|
279
278
|
try {
|
|
280
|
-
for await (const
|
|
279
|
+
for await (const line of readJsonlLines(stream)) {
|
|
281
280
|
if (++count > 50) break;
|
|
282
|
-
const trimmed =
|
|
281
|
+
const trimmed = line.text.trim();
|
|
283
282
|
if (!trimmed) continue;
|
|
284
283
|
try {
|
|
285
284
|
const obj = JSON.parse(trimmed) as Record<string, unknown>;
|
|
@@ -300,7 +299,6 @@ async function readCwdFromTranscript(filePath: string): Promise<string | null> {
|
|
|
300
299
|
}
|
|
301
300
|
}
|
|
302
301
|
} finally {
|
|
303
|
-
rl.close();
|
|
304
302
|
stream.destroy();
|
|
305
303
|
}
|
|
306
304
|
} finally {
|
|
@@ -377,26 +375,16 @@ export async function* readClaudeTranscript(
|
|
|
377
375
|
|
|
378
376
|
const stream = createReadStream(filePath, { start: fromOffset, encoding: undefined });
|
|
379
377
|
|
|
380
|
-
// Suppress stray ENOENT (file deleted between stat and open) —
|
|
378
|
+
// Suppress stray ENOENT (file deleted between stat and open) — the stream
|
|
381
379
|
// raises the same condition through its own iterator, which we catch below.
|
|
382
380
|
stream.on("error", () => {});
|
|
383
381
|
|
|
384
|
-
const rl = readline.createInterface({ input: stream, crlfDelay: Infinity });
|
|
385
|
-
|
|
386
|
-
let pos = fromOffset;
|
|
387
382
|
let stillSkippingPastSince = !!sinceEventId;
|
|
388
383
|
|
|
389
384
|
try {
|
|
390
|
-
for await (const line of
|
|
391
|
-
|
|
392
|
-
|
|
393
|
-
// would yield offsets 1 byte short per line — accepted as a corner
|
|
394
|
-
// case; resume from such an offset would skip one stray `\r`.
|
|
395
|
-
const lineByteLen = Buffer.byteLength(line, "utf8");
|
|
396
|
-
pos += lineByteLen + 1;
|
|
397
|
-
|
|
398
|
-
if (!line) continue;
|
|
399
|
-
const trimmed = line.trim();
|
|
385
|
+
for await (const line of readJsonlLines(stream, fromOffset)) {
|
|
386
|
+
if (!line.text) continue;
|
|
387
|
+
const trimmed = line.text.trim();
|
|
400
388
|
if (!trimmed) continue;
|
|
401
389
|
|
|
402
390
|
if (looksLikeSkippedType(trimmed)) continue;
|
|
@@ -413,7 +401,7 @@ export async function* readClaudeTranscript(
|
|
|
413
401
|
}
|
|
414
402
|
|
|
415
403
|
for (const event of events) {
|
|
416
|
-
yield { event, offset:
|
|
404
|
+
yield { event, offset: line.end };
|
|
417
405
|
}
|
|
418
406
|
}
|
|
419
407
|
} catch (err) {
|
|
@@ -421,7 +409,6 @@ export async function* readClaudeTranscript(
|
|
|
421
409
|
const e = err as NodeJS.ErrnoException;
|
|
422
410
|
if (e?.code !== "ENOENT") throw err;
|
|
423
411
|
} finally {
|
|
424
|
-
rl.close();
|
|
425
412
|
stream.destroy();
|
|
426
413
|
}
|
|
427
414
|
}
|
|
@@ -132,6 +132,10 @@ export async function attachCodexSession(
|
|
|
132
132
|
if (transcript) {
|
|
133
133
|
const lastEvent = await latestTurnBoundary(transcript.filePath, sessionId);
|
|
134
134
|
if (lastEvent === null) lastTurn = "unknown";
|
|
135
|
+
// turn_aborted replays as an interrupted result, but the turn never
|
|
136
|
+
// finished, so it stays "interrupted" for hosts deciding whether to re-send.
|
|
137
|
+
else if (lastEvent.type === "event_msg" && lastEvent.payload?.["type"] === "turn_aborted")
|
|
138
|
+
lastTurn = "interrupted";
|
|
135
139
|
else if (codexLineToStreamEvents(lastEvent, { sessionId }).some((event) => event.type === "result"))
|
|
136
140
|
lastTurn = "completed";
|
|
137
141
|
else lastTurn = "interrupted";
|
|
@@ -57,6 +57,17 @@ function codexUserText(line: CodexTranscriptLine): string | null {
|
|
|
57
57
|
if (line.type === "event_msg" && line.payload?.["type"] === "user_message") {
|
|
58
58
|
return meaningfulHumanText(asString(line.payload["message"]));
|
|
59
59
|
}
|
|
60
|
+
// Paginated rollouts (`session_meta.history_mode: "paginated"`, written by
|
|
61
|
+
// Codex 0.142 and later) never write `user_message`. What the person typed
|
|
62
|
+
// is a completed `UserMessage` item instead, and legacy rollouts never
|
|
63
|
+
// persist that item, so each message is read once. The `response_item` user
|
|
64
|
+
// message beside it also carries injected context, so it is not the one read.
|
|
65
|
+
if (line.type === "event_msg" && line.payload?.["type"] === "item_completed") {
|
|
66
|
+
const item = asRecord(line.payload["item"]);
|
|
67
|
+
if (item?.["type"] === "UserMessage") {
|
|
68
|
+
return meaningfulHumanText(textFromContent(item["content"], new Set(["text", "input_text"])));
|
|
69
|
+
}
|
|
70
|
+
}
|
|
60
71
|
// Older unwrapped rollouts have no event_msg mirror.
|
|
61
72
|
if (line.type === "message" && line.raw["role"] === "user") {
|
|
62
73
|
return meaningfulHumanText(textFromContent(
|
|
@@ -12,11 +12,20 @@ import type { CodexTranscriptLine } from "./transcript.js";
|
|
|
12
12
|
* response_item / reasoning → thinking
|
|
13
13
|
* response_item / function_call → tool_call
|
|
14
14
|
* response_item / function_call_output → tool_result
|
|
15
|
-
*
|
|
15
|
+
* response_item / custom_tool_call → tool_call
|
|
16
|
+
* response_item / custom_tool_call_output → tool_result
|
|
17
|
+
* event_msg / task_complete → result (completed, or failed with `error`)
|
|
18
|
+
* event_msg / turn_aborted → result (interrupted)
|
|
19
|
+
*
|
|
20
|
+
* Current Codex runs shell work through freeform tools: `exec` takes a script
|
|
21
|
+
* and `apply_patch` takes a patch, so most tool activity in a current rollout
|
|
22
|
+
* is a `custom_tool_call`. Its `input` is the raw string the model wrote.
|
|
16
23
|
*
|
|
17
24
|
* Everything else — `session_meta`, `turn_context`, `task_started`,
|
|
18
|
-
* `token_count`, `agent_message`/`agent_reasoning` duplicates,
|
|
19
|
-
*
|
|
25
|
+
* `token_count`, `agent_message`/`agent_reasoning` duplicates, the
|
|
26
|
+
* `item_completed` mirrors of response items, inter-agent `agent_message`
|
|
27
|
+
* items, user/developer messages, unwrapped legacy lines, unknown types —
|
|
28
|
+
* yields `[]`.
|
|
20
29
|
*
|
|
21
30
|
* The on-disk vocabulary is Codex-internal and version-shifting, so every field
|
|
22
31
|
* is read defensively and a weird line NEVER throws — it returns `[]`. Codex
|
|
@@ -89,7 +98,19 @@ function mapLine(line: CodexTranscriptLine, sessionId: string | null): StreamEve
|
|
|
89
98
|
];
|
|
90
99
|
}
|
|
91
100
|
|
|
92
|
-
if (innerType === "
|
|
101
|
+
if (innerType === "custom_tool_call") {
|
|
102
|
+
return [
|
|
103
|
+
{
|
|
104
|
+
type: "tool_call",
|
|
105
|
+
toolCallId: str(payload["call_id"]) ?? str(payload["id"]),
|
|
106
|
+
name: str(payload["name"]) ?? "custom_tool_call",
|
|
107
|
+
input: typeof payload["input"] === "string" ? payload["input"] : null,
|
|
108
|
+
...base,
|
|
109
|
+
},
|
|
110
|
+
];
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
if (innerType === "function_call_output" || innerType === "custom_tool_call_output") {
|
|
93
114
|
return [
|
|
94
115
|
{
|
|
95
116
|
type: "tool_result",
|
|
@@ -110,16 +131,36 @@ function mapLine(line: CodexTranscriptLine, sessionId: string | null): StreamEve
|
|
|
110
131
|
|
|
111
132
|
if (line.type === "event_msg") {
|
|
112
133
|
if (innerType === "task_complete") {
|
|
134
|
+
// A turn that failed (for example a model the account cannot use) still
|
|
135
|
+
// ends in task_complete, with `error` set and no agent message.
|
|
136
|
+
const error = turnErrorMessage(payload["error"]);
|
|
137
|
+
return [
|
|
138
|
+
{
|
|
139
|
+
type: "result",
|
|
140
|
+
text: error ?? str(payload["last_agent_message"]) ?? "",
|
|
141
|
+
costUsd: null,
|
|
142
|
+
isError: error !== null,
|
|
143
|
+
stopReason: null,
|
|
144
|
+
terminalReason: error !== null ? "failed" : "completed",
|
|
145
|
+
numTurns: null,
|
|
146
|
+
durationMs: num(payload["duration_ms"]),
|
|
147
|
+
...base,
|
|
148
|
+
},
|
|
149
|
+
];
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
if (innerType === "turn_aborted") {
|
|
153
|
+
// Same shape as a live `turn/completed` with status "interrupted".
|
|
113
154
|
return [
|
|
114
155
|
{
|
|
115
156
|
type: "result",
|
|
116
|
-
text:
|
|
157
|
+
text: "",
|
|
117
158
|
costUsd: null,
|
|
118
159
|
isError: false,
|
|
119
160
|
stopReason: null,
|
|
120
|
-
terminalReason: "
|
|
161
|
+
terminalReason: str(payload["reason"]) ?? "interrupted",
|
|
121
162
|
numTurns: null,
|
|
122
|
-
durationMs:
|
|
163
|
+
durationMs: num(payload["duration_ms"]),
|
|
123
164
|
...base,
|
|
124
165
|
},
|
|
125
166
|
];
|
|
@@ -135,6 +176,28 @@ function str(v: unknown): string | null {
|
|
|
135
176
|
return typeof v === "string" && v.length > 0 ? v : null;
|
|
136
177
|
}
|
|
137
178
|
|
|
179
|
+
/** Finite number or null. */
|
|
180
|
+
function num(v: unknown): number | null {
|
|
181
|
+
return typeof v === "number" && Number.isFinite(v) ? v : null;
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
/**
|
|
185
|
+
* `task_complete.error` is `{ message, codex_error_info }`. The message is
|
|
186
|
+
* often the API's JSON error body, so prefer its inner `error.message`.
|
|
187
|
+
*/
|
|
188
|
+
function turnErrorMessage(error: unknown): string | null {
|
|
189
|
+
if (typeof error === "string") return str(error);
|
|
190
|
+
if (typeof error !== "object" || error === null) return null;
|
|
191
|
+
const message = str((error as Record<string, unknown>)["message"]);
|
|
192
|
+
if (!message) return "Turn failed";
|
|
193
|
+
try {
|
|
194
|
+
const body = JSON.parse(message) as { error?: { message?: unknown }; message?: unknown };
|
|
195
|
+
return str(body?.error?.message) ?? str(body?.message) ?? message;
|
|
196
|
+
} catch {
|
|
197
|
+
return message;
|
|
198
|
+
}
|
|
199
|
+
}
|
|
200
|
+
|
|
138
201
|
function messagePhase(v: unknown): "commentary" | "final_answer" | undefined {
|
|
139
202
|
return v === "commentary" || v === "final_answer" ? v : undefined;
|
|
140
203
|
}
|
|
@@ -171,13 +234,25 @@ function extractReasoningSummary(summary: unknown): string | null {
|
|
|
171
234
|
}
|
|
172
235
|
|
|
173
236
|
/**
|
|
174
|
-
*
|
|
175
|
-
*
|
|
176
|
-
*
|
|
237
|
+
* Tool output is usually a string. Some versions wrap it as
|
|
238
|
+
* `{ output: "...", metadata: {...} }`, and Codex 0.142+ also writes a list of
|
|
239
|
+
* content parts (`input_text`, `input_image`). Extract the readable text, one
|
|
240
|
+
* part per line, with images left in `raw`. Fall back to a JSON dump
|
|
241
|
+
* so nothing is silently lost. Returns "" when unreadable
|
|
177
242
|
* (`tool_result.content` is a required string).
|
|
178
243
|
*/
|
|
179
244
|
function extractOutputText(output: unknown): string {
|
|
180
245
|
if (typeof output === "string") return output;
|
|
246
|
+
if (Array.isArray(output)) {
|
|
247
|
+
let text = "";
|
|
248
|
+
for (const part of output) {
|
|
249
|
+
if (typeof part !== "object" || part === null) continue;
|
|
250
|
+
const partText = (part as { text?: unknown }).text;
|
|
251
|
+
if (typeof partText !== "string" || partText.length === 0) continue;
|
|
252
|
+
text += text && !text.endsWith("\n") ? `\n${partText}` : partText;
|
|
253
|
+
}
|
|
254
|
+
return text;
|
|
255
|
+
}
|
|
181
256
|
if (output && typeof output === "object" && !Array.isArray(output)) {
|
|
182
257
|
const inner = (output as Record<string, unknown>)["output"];
|
|
183
258
|
if (typeof inner === "string") return inner;
|
|
@@ -28,10 +28,10 @@ import { createReadStream } from "node:fs";
|
|
|
28
28
|
import { readdir, stat, open as fsOpen } from "node:fs/promises";
|
|
29
29
|
import * as os from "node:os";
|
|
30
30
|
import * as path from "node:path";
|
|
31
|
-
import * as readline from "node:readline";
|
|
32
31
|
|
|
33
32
|
import { getDefaultRuntimeHome, getRuntimeHomeEnvVar } from "../../utils/runtime-homes.js";
|
|
34
33
|
import type { FoundTranscript, TranscriptOps } from "../../types.js";
|
|
34
|
+
import { readJsonlLines } from "../../utils/jsonl-lines.js";
|
|
35
35
|
|
|
36
36
|
/** Bytes scanned from the tail of the file in {@link peekCodexTranscript}. */
|
|
37
37
|
const PEEK_TAIL_BYTES = 16 * 1024;
|
|
@@ -316,16 +316,11 @@ export async function* readCodexTranscript(
|
|
|
316
316
|
|
|
317
317
|
const stream = createReadStream(filePath, { start: fromOffset, encoding: undefined });
|
|
318
318
|
stream.on("error", () => {});
|
|
319
|
-
const rl = readline.createInterface({ input: stream, crlfDelay: Infinity });
|
|
320
319
|
|
|
321
320
|
const fileIdentity = rolloutIdentityFromPath(filePath);
|
|
322
|
-
let pos = fromOffset;
|
|
323
321
|
try {
|
|
324
|
-
for await (const line of
|
|
325
|
-
const
|
|
326
|
-
pos += Buffer.byteLength(line, "utf8") + 1;
|
|
327
|
-
|
|
328
|
-
const trimmed = line.trim();
|
|
322
|
+
for await (const line of readJsonlLines(stream, fromOffset)) {
|
|
323
|
+
const trimmed = line.text.trim();
|
|
329
324
|
if (!trimmed) continue;
|
|
330
325
|
|
|
331
326
|
const parsed = parseCodexLine(trimmed);
|
|
@@ -334,15 +329,14 @@ export async function* readCodexTranscript(
|
|
|
334
329
|
// Replay-stable synthetic identity: (rollout identity, line start offset).
|
|
335
330
|
// Codex emits no native per-event uuid, so this is the idempotency key
|
|
336
331
|
// hosts use to dedup transcript replays. Deterministic across reads.
|
|
337
|
-
parsed.eventId = `codex:${fileIdentity}:${
|
|
332
|
+
parsed.eventId = `codex:${fileIdentity}:${line.start}`;
|
|
338
333
|
|
|
339
|
-
yield { event: parsed, offset:
|
|
334
|
+
yield { event: parsed, offset: line.end };
|
|
340
335
|
}
|
|
341
336
|
} catch (err) {
|
|
342
337
|
const e = err as NodeJS.ErrnoException;
|
|
343
338
|
if (e?.code !== "ENOENT") throw err;
|
|
344
339
|
} finally {
|
|
345
|
-
rl.close();
|
|
346
340
|
stream.destroy();
|
|
347
341
|
}
|
|
348
342
|
}
|
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
import * as fs from "node:fs/promises";
|
|
2
2
|
import * as path from "node:path";
|
|
3
3
|
import * as os from "node:os";
|
|
4
|
-
import * as readline from "node:readline";
|
|
5
4
|
import { createReadStream } from "node:fs";
|
|
6
5
|
import type { TokenUsage } from "../../types.js";
|
|
6
|
+
import { readJsonlLines } from "../../utils/jsonl-lines.js";
|
|
7
7
|
|
|
8
8
|
// ---------------------------------------------------------------------------
|
|
9
9
|
// Log path resolution
|
|
@@ -104,16 +104,14 @@ async function scanFile(
|
|
|
104
104
|
): Promise<void> {
|
|
105
105
|
let stream: ReturnType<typeof createReadStream>;
|
|
106
106
|
try {
|
|
107
|
-
stream = createReadStream(filePath
|
|
107
|
+
stream = createReadStream(filePath);
|
|
108
108
|
} catch {
|
|
109
109
|
return;
|
|
110
110
|
}
|
|
111
111
|
|
|
112
|
-
const rl = readline.createInterface({ input: stream, crlfDelay: Infinity });
|
|
113
|
-
|
|
114
112
|
try {
|
|
115
|
-
for await (const line of
|
|
116
|
-
const trimmed = line.trim();
|
|
113
|
+
for await (const line of readJsonlLines(stream)) {
|
|
114
|
+
const trimmed = line.text.trim();
|
|
117
115
|
if (!trimmed) continue;
|
|
118
116
|
|
|
119
117
|
let parsed: Record<string, unknown>;
|
|
@@ -172,7 +170,6 @@ async function scanFile(
|
|
|
172
170
|
usageByModel.set(model, existing);
|
|
173
171
|
}
|
|
174
172
|
} finally {
|
|
175
|
-
rl.close();
|
|
176
173
|
stream.destroy();
|
|
177
174
|
}
|
|
178
175
|
}
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Line splitting for JSONL files on disk.
|
|
3
|
+
*
|
|
4
|
+
* `node:readline` also breaks lines at U+2028 (LINE SEPARATOR), U+2029
|
|
5
|
+
* (PARAGRAPH SEPARATOR) and a lone `\r`. JSON allows U+2028/U+2029 unescaped
|
|
6
|
+
* inside strings, and both `JSON.stringify` (Claude Code) and serde_json
|
|
7
|
+
* (Codex) write them raw, typically in text copied from web pages. readline
|
|
8
|
+
* cuts such a record in two, both halves fail to parse, and the record
|
|
9
|
+
* silently disappears. Every byte offset after it also drifts by two bytes
|
|
10
|
+
* per separator, so a resume offset can land mid-record.
|
|
11
|
+
*
|
|
12
|
+
* {@link readJsonlLines} splits on `\n` only, drops the `\r` of a `\r\n`
|
|
13
|
+
* ending, and reports exact byte offsets.
|
|
14
|
+
*/
|
|
15
|
+
|
|
16
|
+
export interface JsonlLine {
|
|
17
|
+
/** Line text without its `\n` or `\r\n` terminator. */
|
|
18
|
+
text: string;
|
|
19
|
+
/** Byte offset of the line's first byte. */
|
|
20
|
+
start: number;
|
|
21
|
+
/**
|
|
22
|
+
* Byte offset just past the line's `\n`, which is where the next line
|
|
23
|
+
* starts. For a final line with no `\n`, the end of the file.
|
|
24
|
+
*/
|
|
25
|
+
end: number;
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
const NEWLINE = 0x0a;
|
|
29
|
+
const CARRIAGE_RETURN = 0x0d;
|
|
30
|
+
|
|
31
|
+
function decodeLine(bytes: Buffer): string {
|
|
32
|
+
const length = bytes.length > 0 && bytes[bytes.length - 1] === CARRIAGE_RETURN
|
|
33
|
+
? bytes.length - 1
|
|
34
|
+
: bytes.length;
|
|
35
|
+
return bytes.toString("utf8", 0, length);
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
/**
|
|
39
|
+
* Yield the lines of a byte stream, typically `createReadStream(path, { start })`
|
|
40
|
+
* opened without an encoding. `startOffset` is the stream's starting byte
|
|
41
|
+
* offset in the file, so reported offsets are file offsets.
|
|
42
|
+
*
|
|
43
|
+
* Splitting happens on bytes before decoding. A `\n` byte never occurs inside
|
|
44
|
+
* a multi-byte UTF-8 sequence, so no character is split across lines.
|
|
45
|
+
*/
|
|
46
|
+
export async function* readJsonlLines(
|
|
47
|
+
input: AsyncIterable<Buffer | string>,
|
|
48
|
+
startOffset = 0,
|
|
49
|
+
): AsyncGenerator<JsonlLine> {
|
|
50
|
+
let pending: Buffer[] = [];
|
|
51
|
+
let pendingBytes = 0;
|
|
52
|
+
let lineStart = startOffset;
|
|
53
|
+
|
|
54
|
+
for await (const value of input) {
|
|
55
|
+
const chunk = typeof value === "string" ? Buffer.from(value, "utf8") : value;
|
|
56
|
+
let from = 0;
|
|
57
|
+
let newline = chunk.indexOf(NEWLINE, from);
|
|
58
|
+
while (newline !== -1) {
|
|
59
|
+
const piece = chunk.subarray(from, newline);
|
|
60
|
+
const bytes = pendingBytes > 0 ? Buffer.concat([...pending, piece], pendingBytes + piece.length) : piece;
|
|
61
|
+
const end = lineStart + bytes.length + 1;
|
|
62
|
+
yield { text: decodeLine(bytes), start: lineStart, end };
|
|
63
|
+
lineStart = end;
|
|
64
|
+
pending = [];
|
|
65
|
+
pendingBytes = 0;
|
|
66
|
+
from = newline + 1;
|
|
67
|
+
newline = chunk.indexOf(NEWLINE, from);
|
|
68
|
+
}
|
|
69
|
+
if (from < chunk.length) {
|
|
70
|
+
const tail = chunk.subarray(from);
|
|
71
|
+
pending.push(tail);
|
|
72
|
+
pendingBytes += tail.length;
|
|
73
|
+
}
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
if (pendingBytes > 0) {
|
|
77
|
+
const bytes = Buffer.concat(pending, pendingBytes);
|
|
78
|
+
yield { text: decodeLine(bytes), start: lineStart, end: lineStart + bytes.length };
|
|
79
|
+
}
|
|
80
|
+
}
|