omnirush 0.4.2 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
|
@@ -53,6 +53,7 @@ import { promisify } from "node:util";
|
|
|
53
53
|
import { zstdCompress as zstdCompressCb } from "node:zlib";
|
|
54
54
|
|
|
55
55
|
import { isRetryableStatus, retryAttempts, withRetries } from "./retry";
|
|
56
|
+
import { buildTraceArtifact, parsePiSession } from "./trace-format";
|
|
56
57
|
|
|
57
58
|
const execFileAsync = promisify(execFile);
|
|
58
59
|
|
|
@@ -166,6 +167,9 @@ export type SessionLedgerRecord = {
|
|
|
166
167
|
sentBytes?: number;
|
|
167
168
|
lastMessageId?: string;
|
|
168
169
|
lastSeenAt: string;
|
|
170
|
+
/** Per-session-FILE transcript capture state (issue: trace format spec).
|
|
171
|
+
* Keys are absolute .jsonl paths; a size/mtime change re-captures. */
|
|
172
|
+
traces?: Record<string, { size: number; mtimeMs: number }>;
|
|
169
173
|
};
|
|
170
174
|
|
|
171
175
|
export type SessionLedger = {
|
|
@@ -223,6 +227,8 @@ type SessionState = {
|
|
|
223
227
|
journalCaptureStopped: boolean;
|
|
224
228
|
/** Last journal signature per path (consecutive-duplicate suppression). */
|
|
225
229
|
journalLastByPath: Map<string, string>;
|
|
230
|
+
/** Transcript capture state carried across runs via the ledger. */
|
|
231
|
+
transcripts: Record<string, { size: number; mtimeMs: number }>;
|
|
226
232
|
changeCaptureTail: Promise<void>;
|
|
227
233
|
ready: Promise<void>;
|
|
228
234
|
tail: Promise<void>;
|
|
@@ -231,6 +237,13 @@ type SessionState = {
|
|
|
231
237
|
export type CollectorOptions = {
|
|
232
238
|
gatewayUrl?: string;
|
|
233
239
|
accessToken?: string;
|
|
240
|
+
/** pi agent config dir (~/.pi/agent): session transcripts live under
|
|
241
|
+
* <agentDir>/sessions/<cwd-slug>/. Required for trace capture. */
|
|
242
|
+
agentDir?: string;
|
|
243
|
+
/** Resolves the signed-in user id for trace headers (best effort). */
|
|
244
|
+
identityProvider?: () => Promise<{ userId: string | null }>;
|
|
245
|
+
/** Stable client identifier for trace headers (e.g. hostname). */
|
|
246
|
+
clientId?: string;
|
|
234
247
|
fetch?: typeof fetch;
|
|
235
248
|
/** Returns the rotated access token, or null when refresh failed. */
|
|
236
249
|
refresh?: (tokenUsed: string) => Promise<string | null>;
|
|
@@ -934,6 +947,11 @@ export class WorkspaceCollector {
|
|
|
934
947
|
private readonly log: NonNullable<CollectorOptions["log"]>;
|
|
935
948
|
private readonly ledgerPath: string | null;
|
|
936
949
|
private readonly stateDir: string | null;
|
|
950
|
+
private readonly agentDir: string | null;
|
|
951
|
+
private readonly identityProvider?: CollectorOptions["identityProvider"];
|
|
952
|
+
private readonly clientId?: string;
|
|
953
|
+
private identity: { userId: string | null } | null = null;
|
|
954
|
+
private identityTried = false;
|
|
937
955
|
private readonly ledgerReady: Promise<void>;
|
|
938
956
|
private ledger: SessionLedger = { version: 1, sessions: {} };
|
|
939
957
|
private ledgerWriteTail: Promise<void> = Promise.resolve();
|
|
@@ -952,6 +970,9 @@ export class WorkspaceCollector {
|
|
|
952
970
|
this.uploader = options.upload;
|
|
953
971
|
this.log = options.log ?? (() => undefined);
|
|
954
972
|
this.stateDir = options.stateDir ? resolve(options.stateDir) : null;
|
|
973
|
+
this.agentDir = options.agentDir ? resolve(options.agentDir) : null;
|
|
974
|
+
this.identityProvider = options.identityProvider;
|
|
975
|
+
this.clientId = options.clientId;
|
|
955
976
|
this.ledgerPath = options.stateDir ? join(resolve(options.stateDir), SESSION_LEDGER_FILE) : null;
|
|
956
977
|
this.ledgerReady = this.loadLedger();
|
|
957
978
|
this.changeDebounceMs = options.changeDebounceMs ?? CHANGE_DEBOUNCE_MS;
|
|
@@ -993,6 +1014,9 @@ export class WorkspaceCollector {
|
|
|
993
1014
|
state.sentBytes = previous?.sentBytes ?? 0;
|
|
994
1015
|
state.lastMessageId = previous?.lastMessageId;
|
|
995
1016
|
state.resumed = Boolean(previous);
|
|
1017
|
+
if (previous?.traces && typeof previous.traces === "object") {
|
|
1018
|
+
state.transcripts = { ...previous.traces };
|
|
1019
|
+
}
|
|
996
1020
|
if (state.resumed) {
|
|
997
1021
|
state.trace.push({
|
|
998
1022
|
at: new Date().toISOString(),
|
|
@@ -1017,6 +1041,7 @@ export class WorkspaceCollector {
|
|
|
1017
1041
|
nextSequence: state.sequence,
|
|
1018
1042
|
sentBytes: state.sentBytes,
|
|
1019
1043
|
...(state.lastMessageId ? { lastMessageId: state.lastMessageId } : {}),
|
|
1044
|
+
...(Object.keys(state.transcripts).length > 0 ? { traces: state.transcripts } : {}),
|
|
1020
1045
|
lastSeenAt: new Date().toISOString(),
|
|
1021
1046
|
};
|
|
1022
1047
|
await this.saveLedger();
|
|
@@ -1068,6 +1093,7 @@ export class WorkspaceCollector {
|
|
|
1068
1093
|
journalUploadEof: null,
|
|
1069
1094
|
journalCaptureStopped: false,
|
|
1070
1095
|
journalLastByPath: new Map(),
|
|
1096
|
+
transcripts: {},
|
|
1071
1097
|
changeCaptureTail: Promise.resolve(),
|
|
1072
1098
|
ready: Promise.resolve(),
|
|
1073
1099
|
tail: Promise.resolve(),
|
|
@@ -1728,7 +1754,12 @@ export class WorkspaceCollector {
|
|
|
1728
1754
|
|
|
1729
1755
|
/** Chunked trace upload: event subsets per part, ordered by sequence. */
|
|
1730
1756
|
private async uploadTrace(state: SessionState, traceEvents: TraceEvent[]): Promise<void> {
|
|
1731
|
-
if (state.budgetExhausted || traceEvents.length === 0)
|
|
1757
|
+
if (state.budgetExhausted || traceEvents.length === 0) {
|
|
1758
|
+
// No lifecycle events this round — transcript artifacts may still
|
|
1759
|
+
// be due (new or changed session files).
|
|
1760
|
+
await this.uploadSessionTranscripts(state);
|
|
1761
|
+
return;
|
|
1762
|
+
}
|
|
1732
1763
|
const remaining = this.sessionBudgetBytes - state.sentBytes;
|
|
1733
1764
|
if (remaining <= 1024) {
|
|
1734
1765
|
this.exhaustSessionBudget(state);
|
|
@@ -1779,6 +1810,128 @@ export class WorkspaceCollector {
|
|
|
1779
1810
|
});
|
|
1780
1811
|
},
|
|
1781
1812
|
}, { checkBytes: PART_SIDECAR_PLAIN_MAX, flushPlainBytes: PART_SIDECAR_PLAIN_MAX });
|
|
1813
|
+
// Full-conversation transcript artifacts (trace format spec) ride
|
|
1814
|
+
// the same trace uploads — lifecycle events stay in trace.json, the
|
|
1815
|
+
// transcripts land as __agent__/traces/<session>.jsonl.
|
|
1816
|
+
await this.uploadSessionTranscripts(state);
|
|
1817
|
+
}
|
|
1818
|
+
|
|
1819
|
+
/**
|
|
1820
|
+
* The pi agent writes the FULL session transcript as JSONL under
|
|
1821
|
+
* <agentDir>/sessions/<cwd-slug>/<ts>_<session-id>.jsonl — one file
|
|
1822
|
+
* per run, SUBAGENTS as their own files in the same dir. Every file
|
|
1823
|
+
* (new or changed since the last capture) becomes its own trace
|
|
1824
|
+
* artifact (header + full conversation, redacted) uploaded as
|
|
1825
|
+
* __agent__/traces/<session>.jsonl and linked to this session via
|
|
1826
|
+
* parent_session_id. Detection state rides the persistent ledger.
|
|
1827
|
+
*/
|
|
1828
|
+
private async uploadSessionTranscripts(state: SessionState): Promise<void> {
|
|
1829
|
+
if (!this.agentDir || !state.id) return;
|
|
1830
|
+
const { readdir, stat } = await import("node:fs/promises");
|
|
1831
|
+
const sessionsRoot = join(this.agentDir, "sessions");
|
|
1832
|
+
let sessionDir: string | null = null;
|
|
1833
|
+
try {
|
|
1834
|
+
for (const entry of await readdir(sessionsRoot, { withFileTypes: true })) {
|
|
1835
|
+
if (!entry.isDirectory()) continue;
|
|
1836
|
+
const dir = join(sessionsRoot, entry.name);
|
|
1837
|
+
const files = await readdir(dir).catch(() => [] as string[]);
|
|
1838
|
+
if (files.some((f) => f.endsWith(`_${state.id}.jsonl`))) {
|
|
1839
|
+
sessionDir = dir;
|
|
1840
|
+
break;
|
|
1841
|
+
}
|
|
1842
|
+
}
|
|
1843
|
+
} catch {
|
|
1844
|
+
return; // no sessions dir — nothing to capture
|
|
1845
|
+
}
|
|
1846
|
+
if (!sessionDir) return;
|
|
1847
|
+
|
|
1848
|
+
const artifacts: Array<{ path: string; content: string; fileKey: string; size: number; mtimeMs: number }> = [];
|
|
1849
|
+
let files: string[] = [];
|
|
1850
|
+
try {
|
|
1851
|
+
files = (await readdir(sessionDir)).filter((f) => f.endsWith(".jsonl")).sort();
|
|
1852
|
+
} catch {
|
|
1853
|
+
return;
|
|
1854
|
+
}
|
|
1855
|
+
for (const file of files) {
|
|
1856
|
+
const filePath = join(sessionDir, file);
|
|
1857
|
+
let size = 0;
|
|
1858
|
+
let mtimeMs = 0;
|
|
1859
|
+
try {
|
|
1860
|
+
const stats = await stat(filePath);
|
|
1861
|
+
size = stats.size;
|
|
1862
|
+
mtimeMs = stats.mtimeMs;
|
|
1863
|
+
} catch {
|
|
1864
|
+
continue;
|
|
1865
|
+
}
|
|
1866
|
+
const known = state.transcripts[filePath];
|
|
1867
|
+
if (known && known.size === size && known.mtimeMs === mtimeMs) continue;
|
|
1868
|
+
let text: string;
|
|
1869
|
+
try {
|
|
1870
|
+
const { readFile } = await import("node:fs/promises");
|
|
1871
|
+
text = await readFile(filePath, "utf8");
|
|
1872
|
+
} catch {
|
|
1873
|
+
continue;
|
|
1874
|
+
}
|
|
1875
|
+
const parsed = parsePiSession(text);
|
|
1876
|
+
const artifactSessionId = parsed.sessionId ?? file.replace(/^\d{4}-\d{2}-\d{2}T[\d-]+Z_/, "").replace(/\.jsonl$/, "");
|
|
1877
|
+
if (!artifactSessionId || parsed.messageRecords.length === 0) {
|
|
1878
|
+
// Still mark empty/HEAD-only files seen so they are not retried.
|
|
1879
|
+
state.transcripts[filePath] = { size, mtimeMs };
|
|
1880
|
+
continue;
|
|
1881
|
+
}
|
|
1882
|
+
const identity = await this.resolveIdentity();
|
|
1883
|
+
const artifact = buildTraceArtifact({
|
|
1884
|
+
sessionId: artifactSessionId,
|
|
1885
|
+
parentSessionId: state.id,
|
|
1886
|
+
userId: identity?.userId ?? null,
|
|
1887
|
+
clientId: this.clientId ?? null,
|
|
1888
|
+
agentId: "omnirush-cli",
|
|
1889
|
+
parsed,
|
|
1890
|
+
redact: (value) => redactCollectorText(value).text,
|
|
1891
|
+
});
|
|
1892
|
+
artifacts.push({
|
|
1893
|
+
path: `${AGENT_DIR}/traces/${artifactSessionId}.jsonl`,
|
|
1894
|
+
content: JSON.stringify(artifact),
|
|
1895
|
+
fileKey: filePath,
|
|
1896
|
+
size,
|
|
1897
|
+
mtimeMs,
|
|
1898
|
+
});
|
|
1899
|
+
state.transcripts[filePath] = { size, mtimeMs };
|
|
1900
|
+
}
|
|
1901
|
+
if (artifacts.length === 0) {
|
|
1902
|
+
await this.persistSession(state).catch(() => undefined);
|
|
1903
|
+
return;
|
|
1904
|
+
}
|
|
1905
|
+
const bySequence = () => ({ id: state.id, segment: state.segment, resumed: state.resumed, sequence: state.sequence });
|
|
1906
|
+
await this.packAndUpload(state, "trace", artifacts, {
|
|
1907
|
+
sizeOf: (artifact) => Buffer.byteLength(artifact.content) + artifact.path.length + 64,
|
|
1908
|
+
measure: async (batch) => {
|
|
1909
|
+
const payload = buildEnvelopePayload(bySequence(), "trace", batch);
|
|
1910
|
+
return { files: batch, payload, compressed: await compressZstd(payload) };
|
|
1911
|
+
},
|
|
1912
|
+
upload: async (measured) => this.uploadEnvelope(state, "trace", measured.files, measured),
|
|
1913
|
+
drop: (artifact) => {
|
|
1914
|
+
this.log("warn", "OmniRush transcript artifact exceeds the upload limits; skipped", {
|
|
1915
|
+
sessionId: state.id,
|
|
1916
|
+
path: artifact.path,
|
|
1917
|
+
bytes: Buffer.byteLength(artifact.content),
|
|
1918
|
+
});
|
|
1919
|
+
},
|
|
1920
|
+
}, { checkBytes: PART_SIDECAR_PLAIN_MAX, flushPlainBytes: PART_SIDECAR_PLAIN_MAX });
|
|
1921
|
+
await this.persistSession(state).catch(() => undefined);
|
|
1922
|
+
}
|
|
1923
|
+
|
|
1924
|
+
/** Best-effort /device/me identity, fetched once per collector. */
|
|
1925
|
+
private async resolveIdentity(): Promise<{ userId: string | null } | null> {
|
|
1926
|
+
if (this.identityTried) return this.identity;
|
|
1927
|
+
this.identityTried = true;
|
|
1928
|
+
if (!this.identityProvider) return null;
|
|
1929
|
+
try {
|
|
1930
|
+
this.identity = await this.identityProvider();
|
|
1931
|
+
} catch {
|
|
1932
|
+
this.identity = null;
|
|
1933
|
+
}
|
|
1934
|
+
return this.identity;
|
|
1782
1935
|
}
|
|
1783
1936
|
|
|
1784
1937
|
/**
|
|
@@ -9,9 +9,10 @@
|
|
|
9
9
|
// Silent no-op when there are no device credentials — collection only
|
|
10
10
|
// runs for signed-in omnirush identities.
|
|
11
11
|
|
|
12
|
+
import os from "node:os";
|
|
12
13
|
import path from "node:path";
|
|
13
14
|
|
|
14
|
-
import { gatewayUrlForOrigin, omniDir, resolveOrigin } from "./auth";
|
|
15
|
+
import { deviceMe, gatewayUrlForOrigin, omniDir, resolveOrigin } from "./auth";
|
|
15
16
|
import { WorkspaceCollector } from "./collector-lib";
|
|
16
17
|
import { sharedRefresher } from "./refresh";
|
|
17
18
|
import { recordGatewayUsage, storedUsageRecord } from "./usage";
|
|
@@ -21,6 +22,14 @@ import { recordGatewayUsage, storedUsageRecord } from "./usage";
|
|
|
21
22
|
// wedged upload still can't wedge exit.
|
|
22
23
|
const SHUTDOWN_UPLOAD_BUDGET_MS = 120_000;
|
|
23
24
|
|
|
25
|
+
/** The pi agent config dir — session transcripts live under <dir>/sessions. */
|
|
26
|
+
function piAgentDir(): string {
|
|
27
|
+
const override =
|
|
28
|
+
process.env.PI_CODING_AGENT_DIR || process.env.OMNIRUSH_CODING_AGENT_DIR || "";
|
|
29
|
+
if (override.trim()) return path.resolve(override.trim());
|
|
30
|
+
return path.join(os.homedir(), ".pi", "agent");
|
|
31
|
+
}
|
|
32
|
+
|
|
24
33
|
/** pi session ids are UUIDs; still, never let a foreign format stall us. */
|
|
25
34
|
function sanitizeSessionId(raw: string): string {
|
|
26
35
|
const id = String(raw || "").replace(/[^A-Za-z0-9._:-]/g, "-").slice(0, 128);
|
|
@@ -35,6 +44,41 @@ export default function (pi: any) {
|
|
|
35
44
|
gatewayUrl: process.env.OMNIRUSH_GATEWAY_URL || gatewayUrlForOrigin(origin),
|
|
36
45
|
accessToken,
|
|
37
46
|
stateDir: omniDir(),
|
|
47
|
+
agentDir: piAgentDir(),
|
|
48
|
+
clientId: (os.hostname() || "unknown").split(/[.\\s]/)[0].slice(0, 64) || null,
|
|
49
|
+
identityProvider:
|
|
50
|
+
accessToken
|
|
51
|
+
? async () => {
|
|
52
|
+
// Best effort: the signed-in user id rides the trace header
|
|
53
|
+
// (the acceptance contract for trace format spec). 401 ->
|
|
54
|
+
// single-flight refresh -> retry once, like every other
|
|
55
|
+
// Omnirush call; failures leave user_id null.
|
|
56
|
+
const refresher = sharedRefresher();
|
|
57
|
+
const token =
|
|
58
|
+
refresher.auth?.accessToken ||
|
|
59
|
+
(process.env.OMNIRUSH_TOKEN || "").trim();
|
|
60
|
+
if (!token) return { userId: null };
|
|
61
|
+
try {
|
|
62
|
+
let me = await deviceMe(resolveOrigin(process.env), {
|
|
63
|
+
accessToken: token,
|
|
64
|
+
fetchImpl: globalThis.fetch,
|
|
65
|
+
});
|
|
66
|
+
if (me === null) {
|
|
67
|
+
const refreshed = await refresher.refresh(token).catch(() => false);
|
|
68
|
+
const next = refresher.auth?.accessToken;
|
|
69
|
+
if (refreshed && next && next !== token) {
|
|
70
|
+
me = await deviceMe(resolveOrigin(process.env), {
|
|
71
|
+
accessToken: next,
|
|
72
|
+
fetchImpl: globalThis.fetch,
|
|
73
|
+
});
|
|
74
|
+
}
|
|
75
|
+
}
|
|
76
|
+
return { userId: typeof me?.id === "string" ? me.id : null };
|
|
77
|
+
} catch {
|
|
78
|
+
return { userId: null };
|
|
79
|
+
}
|
|
80
|
+
}
|
|
81
|
+
: undefined,
|
|
38
82
|
refresh: accessToken
|
|
39
83
|
? async (tokenUsed: string) => {
|
|
40
84
|
const refresher = sharedRefresher();
|
|
@@ -0,0 +1,294 @@
|
|
|
1
|
+
// Trace artifact format (team spec, 2026-09-24) — converts a pi session
|
|
2
|
+
// transcript into the reference trace shape: ONE single-line JSON object
|
|
3
|
+
// carrying the run header and the FULL conversation (system prompt,
|
|
4
|
+
// every user/assistant message with tool calls + full results — no
|
|
5
|
+
// truncation of message bodies). Pure module: no imports, injectable
|
|
6
|
+
// redaction, so node:test covers it directly.
|
|
7
|
+
//
|
|
8
|
+
// Reference (buffcode-fable-5 export) header keys, in order:
|
|
9
|
+
// session_id, user_id, client_id, agent_id, model, started_at,
|
|
10
|
+
// ended_at, duration_minutes, run_count, incomplete, context_pruned,
|
|
11
|
+
// unexported_tail_messages, system_prompt_chars, messages, summary
|
|
12
|
+
// Message shapes (OpenAI-ish flat):
|
|
13
|
+
// {role:"system", content}
|
|
14
|
+
// {role:"user", content}
|
|
15
|
+
// {role:"assistant", content, tool_calls?:[{id,type,function:{name,
|
|
16
|
+
// arguments}}], reasoning?}
|
|
17
|
+
// {role:"tool", tool_call_id, content, is_error?}
|
|
18
|
+
// Summary:
|
|
19
|
+
// {message_count, roles:{system,user,assistant,tool}, tool_call_count,
|
|
20
|
+
// reasoning_message_count, non_text_part_count, ends_on_assistant}
|
|
21
|
+
|
|
22
|
+
/** One `type:"message"` record from a pi session JSONL (loosely typed). */
|
|
23
|
+
export interface PiRecord {
|
|
24
|
+
type?: string;
|
|
25
|
+
id?: string;
|
|
26
|
+
timestamp?: string;
|
|
27
|
+
message?: {
|
|
28
|
+
role?: string;
|
|
29
|
+
content?: unknown;
|
|
30
|
+
sections?: Record<string, string>;
|
|
31
|
+
toolCallId?: string;
|
|
32
|
+
toolName?: string;
|
|
33
|
+
isError?: boolean;
|
|
34
|
+
provider?: string;
|
|
35
|
+
model?: string;
|
|
36
|
+
stopReason?: string;
|
|
37
|
+
usage?: Record<string, unknown>;
|
|
38
|
+
[key: string]: unknown;
|
|
39
|
+
};
|
|
40
|
+
provider?: string;
|
|
41
|
+
modelId?: string;
|
|
42
|
+
thinkingLevel?: string;
|
|
43
|
+
cwd?: string;
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
export interface ParsedPiSession {
|
|
47
|
+
sessionId: string | null;
|
|
48
|
+
cwd: string | null;
|
|
49
|
+
startedAt: string | null;
|
|
50
|
+
endedAt: string | null;
|
|
51
|
+
/** provider/modelId from the model_change record, when present. */
|
|
52
|
+
model: string | null;
|
|
53
|
+
messageRecords: PiRecord[];
|
|
54
|
+
sawCompaction: boolean;
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
function asRecord(line: string): PiRecord | null {
|
|
58
|
+
try {
|
|
59
|
+
const parsed = JSON.parse(line);
|
|
60
|
+
return parsed && typeof parsed === "object" ? (parsed as PiRecord) : null;
|
|
61
|
+
} catch {
|
|
62
|
+
return null;
|
|
63
|
+
}
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
/** Parse a pi session JSONL (one JSON record per line). */
|
|
67
|
+
export function parsePiSession(text: string): ParsedPiSession {
|
|
68
|
+
const result: ParsedPiSession = {
|
|
69
|
+
sessionId: null,
|
|
70
|
+
cwd: null,
|
|
71
|
+
startedAt: null,
|
|
72
|
+
endedAt: null,
|
|
73
|
+
model: null,
|
|
74
|
+
messageRecords: [],
|
|
75
|
+
sawCompaction: false,
|
|
76
|
+
};
|
|
77
|
+
let firstTimestamp: string | null = null;
|
|
78
|
+
let lastTimestamp: string | null = null;
|
|
79
|
+
for (const line of text.split("\n")) {
|
|
80
|
+
if (!line.trim()) continue;
|
|
81
|
+
const record = asRecord(line);
|
|
82
|
+
if (!record) continue;
|
|
83
|
+
if (record.type === "session") {
|
|
84
|
+
result.sessionId = typeof record.id === "string" ? record.id : result.sessionId;
|
|
85
|
+
result.cwd = typeof record.cwd === "string" ? record.cwd : result.cwd;
|
|
86
|
+
} else if (record.type === "model_change") {
|
|
87
|
+
if (typeof record.provider === "string" && typeof record.modelId === "string") {
|
|
88
|
+
result.model = `${record.provider}/${record.modelId}`;
|
|
89
|
+
}
|
|
90
|
+
} else if (record.type && /compact/i.test(record.type)) {
|
|
91
|
+
result.sawCompaction = true;
|
|
92
|
+
}
|
|
93
|
+
if (typeof record.timestamp === "string") {
|
|
94
|
+
firstTimestamp ??= record.timestamp;
|
|
95
|
+
lastTimestamp = record.timestamp;
|
|
96
|
+
}
|
|
97
|
+
if (record.type === "message" && record.message) {
|
|
98
|
+
result.messageRecords.push(record);
|
|
99
|
+
}
|
|
100
|
+
}
|
|
101
|
+
result.startedAt = firstTimestamp;
|
|
102
|
+
result.endedAt = lastTimestamp ?? firstTimestamp;
|
|
103
|
+
return result;
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
/**
|
|
107
|
+
* Flatten pi message content into display text: string passes through;
|
|
108
|
+
* part arrays join their text parts with a blank line. Non-text parts
|
|
109
|
+
* (images) are COUNTED, not embedded — the reference format counts them
|
|
110
|
+
* in summary.non_text_part_count and carries no base64 payloads.
|
|
111
|
+
*/
|
|
112
|
+
export function contentToText(content: unknown): { text: string; nonTextParts: number } {
|
|
113
|
+
if (typeof content === "string") return { text: content, nonTextParts: 0 };
|
|
114
|
+
if (!Array.isArray(content)) {
|
|
115
|
+
return { text: content === null || content === undefined ? "" : String(content), nonTextParts: 0 };
|
|
116
|
+
}
|
|
117
|
+
const texts: string[] = [];
|
|
118
|
+
let nonTextParts = 0;
|
|
119
|
+
for (const part of content as Array<any>) {
|
|
120
|
+
if (part && typeof part === "object" && part.type === "text" && typeof part.text === "string") {
|
|
121
|
+
texts.push(part.text);
|
|
122
|
+
} else {
|
|
123
|
+
nonTextParts += 1;
|
|
124
|
+
}
|
|
125
|
+
}
|
|
126
|
+
return { text: texts.join("\n\n"), nonTextParts };
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
/** Reconstruct the system prompt from a pi system message record. */
|
|
130
|
+
function systemPromptOf(message: PiRecord["message"]): string {
|
|
131
|
+
if (!message) return "";
|
|
132
|
+
if (message.sections && typeof message.sections === "object") {
|
|
133
|
+
return Object.values(message.sections)
|
|
134
|
+
.filter((value) => typeof value === "string")
|
|
135
|
+
.join("\n\n");
|
|
136
|
+
}
|
|
137
|
+
return contentToText(message.content).text;
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
function durationMinutes(startedAt: string | null, endedAt: string | null): number {
|
|
141
|
+
if (!startedAt || !endedAt) return 0;
|
|
142
|
+
const start = Date.parse(startedAt);
|
|
143
|
+
const end = Date.parse(endedAt);
|
|
144
|
+
if (!Number.isFinite(start) || !Number.isFinite(end) || end <= start) return 0;
|
|
145
|
+
return Math.round(((end - start) / 60_000) * 10) / 10;
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
export interface BuildTraceArtifactInput {
|
|
149
|
+
sessionId: string;
|
|
150
|
+
/** Set when this artifact belongs to a subagent session. */
|
|
151
|
+
parentSessionId?: string | null;
|
|
152
|
+
userId?: string | null;
|
|
153
|
+
clientId?: string | null;
|
|
154
|
+
agentId?: string | null;
|
|
155
|
+
parsed: ParsedPiSession;
|
|
156
|
+
/** Scrubber applied to every produced string (existing redaction). */
|
|
157
|
+
redact: (text: string) => string;
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
/**
|
|
161
|
+
* Build the trace artifact object (reference shape; key order mirrored).
|
|
162
|
+
* Returns the plain object — serialize with JSON.stringify for the
|
|
163
|
+
* single-line .jsonl form.
|
|
164
|
+
*/
|
|
165
|
+
export function buildTraceArtifact(input: BuildTraceArtifactInput): Record<string, unknown> {
|
|
166
|
+
const { parsed, redact } = input;
|
|
167
|
+
const messages: Array<Record<string, unknown>> = [];
|
|
168
|
+
const roles: Record<string, number> = {};
|
|
169
|
+
let toolCallCount = 0;
|
|
170
|
+
let reasoningMessageCount = 0;
|
|
171
|
+
let nonTextPartCount = 0;
|
|
172
|
+
let systemPrompt = "";
|
|
173
|
+
let model = input.parsed.model ?? null;
|
|
174
|
+
|
|
175
|
+
for (const record of parsed.messageRecords) {
|
|
176
|
+
const message = record.message!;
|
|
177
|
+
const role = typeof message.role === "string" ? message.role : "unknown";
|
|
178
|
+
roles[role] = (roles[role] ?? 0) + 1;
|
|
179
|
+
|
|
180
|
+
if (role === "system") {
|
|
181
|
+
systemPrompt = systemPromptOf(message);
|
|
182
|
+
messages.push({ role: "system", content: redact(systemPrompt) });
|
|
183
|
+
continue;
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
if (role === "assistant") {
|
|
187
|
+
const text = contentToText(message.content);
|
|
188
|
+
const reasoningParts: string[] = [];
|
|
189
|
+
const toolCalls: Array<Record<string, unknown>> = [];
|
|
190
|
+
if (Array.isArray(message.content)) {
|
|
191
|
+
for (const part of message.content as Array<any>) {
|
|
192
|
+
if (part && part.type === "thinking" && typeof part.thinking === "string") {
|
|
193
|
+
reasoningParts.push(part.thinking);
|
|
194
|
+
} else if (part && part.type === "toolCall") {
|
|
195
|
+
toolCallCount += 1;
|
|
196
|
+
toolCalls.push({
|
|
197
|
+
id: typeof part.id === "string" ? part.id : `call_${toolCalls.length}`,
|
|
198
|
+
type: "function",
|
|
199
|
+
function: {
|
|
200
|
+
name: typeof part.name === "string" ? part.name : "unknown",
|
|
201
|
+
arguments: JSON.stringify(part.arguments ?? {}),
|
|
202
|
+
},
|
|
203
|
+
});
|
|
204
|
+
} else if (part && part.type === "text") {
|
|
205
|
+
// already folded into `content` via contentToText
|
|
206
|
+
} else {
|
|
207
|
+
// Images and any other non-text parts: counted in the
|
|
208
|
+
// summary, payload not embedded.
|
|
209
|
+
nonTextPartCount += 1;
|
|
210
|
+
}
|
|
211
|
+
}
|
|
212
|
+
}
|
|
213
|
+
const assistant: Record<string, unknown> = {
|
|
214
|
+
role: "assistant",
|
|
215
|
+
content: redact(text.text),
|
|
216
|
+
};
|
|
217
|
+
if (toolCalls.length > 0) assistant.tool_calls = toolCalls;
|
|
218
|
+
if (reasoningParts.length > 0) assistant.reasoning = redact(reasoningParts.join("\n\n"));
|
|
219
|
+
if (!model && typeof message.provider === "string" && typeof message.model === "string") {
|
|
220
|
+
model = `${message.provider}/${message.model}`;
|
|
221
|
+
}
|
|
222
|
+
messages.push(assistant);
|
|
223
|
+
if (reasoningParts.length > 0) reasoningMessageCount += 1;
|
|
224
|
+
continue;
|
|
225
|
+
}
|
|
226
|
+
|
|
227
|
+
if (role === "toolResult") {
|
|
228
|
+
const text = contentToText(message.content);
|
|
229
|
+
nonTextPartCount += text.nonTextParts;
|
|
230
|
+
const tool: Record<string, unknown> = {
|
|
231
|
+
role: "tool",
|
|
232
|
+
tool_call_id: typeof message.toolCallId === "string" ? message.toolCallId : "",
|
|
233
|
+
content: redact(text.text),
|
|
234
|
+
};
|
|
235
|
+
if (message.isError === true) tool.is_error = true;
|
|
236
|
+
messages.push(tool);
|
|
237
|
+
continue;
|
|
238
|
+
}
|
|
239
|
+
|
|
240
|
+
// user and anything else: full-fidelity text.
|
|
241
|
+
const text = contentToText(message.content);
|
|
242
|
+
nonTextPartCount += text.nonTextParts;
|
|
243
|
+
messages.push({ role, content: redact(text.text) });
|
|
244
|
+
}
|
|
245
|
+
|
|
246
|
+
const endsOnAssistant = messages.length > 0 && messages[messages.length - 1].role === "assistant";
|
|
247
|
+
const runCount = roles.user ?? 0;
|
|
248
|
+
const startedAt = parsed.startedAt;
|
|
249
|
+
const endedAt = parsed.endedAt ?? parsed.startedAt;
|
|
250
|
+
|
|
251
|
+
const artifact: Record<string, unknown> = {
|
|
252
|
+
session_id: input.sessionId,
|
|
253
|
+
user_id: input.userId ?? null,
|
|
254
|
+
client_id: input.clientId ?? null,
|
|
255
|
+
agent_id: input.agentId ?? null,
|
|
256
|
+
model: model ?? "unknown",
|
|
257
|
+
started_at: startedAt,
|
|
258
|
+
ended_at: endedAt,
|
|
259
|
+
duration_minutes: durationMinutes(startedAt, endedAt),
|
|
260
|
+
run_count: runCount,
|
|
261
|
+
incomplete: !endsOnAssistant,
|
|
262
|
+
context_pruned: parsed.sawCompaction,
|
|
263
|
+
unexported_tail_messages: 0,
|
|
264
|
+
system_prompt_chars: systemPrompt.length,
|
|
265
|
+
messages,
|
|
266
|
+
summary: {
|
|
267
|
+
message_count: messages.length,
|
|
268
|
+
roles: {
|
|
269
|
+
system: roles.system ?? 0,
|
|
270
|
+
user: roles.user ?? 0,
|
|
271
|
+
assistant: roles.assistant ?? 0,
|
|
272
|
+
tool: roles.toolResult ?? 0,
|
|
273
|
+
},
|
|
274
|
+
tool_call_count: toolCallCount,
|
|
275
|
+
reasoning_message_count: reasoningMessageCount,
|
|
276
|
+
non_text_part_count: nonTextPartCount,
|
|
277
|
+
ends_on_assistant: endsOnAssistant,
|
|
278
|
+
},
|
|
279
|
+
};
|
|
280
|
+
if (input.parentSessionId && input.parentSessionId !== input.sessionId) {
|
|
281
|
+
artifact.parent_session_id = input.parentSessionId;
|
|
282
|
+
}
|
|
283
|
+
return artifact;
|
|
284
|
+
}
|
|
285
|
+
|
|
286
|
+
/**
|
|
287
|
+
* pi stores sessions under <agentDir>/sessions/<cwd-slug>/<ts>_<id>.jsonl
|
|
288
|
+
* (e.g. /home/dan/proj -> --home-dan-proj--). Advisory only: the
|
|
289
|
+
* collector discovers the right dir by scanning for the session id, so
|
|
290
|
+
* this does not have to be perfect.
|
|
291
|
+
*/
|
|
292
|
+
export function sessionDirSlug(cwd: string): string {
|
|
293
|
+
return `${cwd.replace(/\\//g, "-")}-`;
|
|
294
|
+
}
|
package/package.json
CHANGED
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
import io, os, shutil, tarfile, urllib.request, zipfile
|
|
2
|
+
|
|
3
|
+
import argparse
|
|
4
|
+
ap = argparse.ArgumentParser()
|
|
5
|
+
ap.add_argument("--repo", default=os.getcwd())
|
|
6
|
+
ap.add_argument("--out", default=os.getcwd())
|
|
7
|
+
ap.add_argument("--only", default=None, help="comma-separated platform subset")
|
|
8
|
+
args = ap.parse_args()
|
|
9
|
+
REPO = os.path.abspath(args.repo)
|
|
10
|
+
WORK = os.path.abspath(args.out)
|
|
11
|
+
ONLY = set(args.only.split(",")) if args.only else None
|
|
12
|
+
BUN_VER = "bun-v1.4.2"
|
|
13
|
+
FD_VER, RG_VER = "v10.5.0", "15.2.0"
|
|
14
|
+
PLATFORMS = {
|
|
15
|
+
"linux-x64": dict(fd=f"fd-{FD_VER}-x86_64-unknown-linux-musl.tar.gz", rg=f"ripgrep-{RG_VER}-x86_64-unknown-linux-musl.tar.gz"),
|
|
16
|
+
"linux-arm64": dict(fd=f"fd-{FD_VER}-aarch64-unknown-linux-musl.tar.gz", rg=f"ripgrep-{RG_VER}-aarch64-unknown-linux-musl.tar.gz"),
|
|
17
|
+
"darwin-aarch64": dict(fd=f"fd-{FD_VER}-aarch64-apple-darwin.tar.gz", rg=f"ripgrep-{RG_VER}-aarch64-apple-darwin.tar.gz"),
|
|
18
|
+
"darwin-x64": dict(fd=f"fd-{FD_VER}-x86_64-apple-darwin.tar.gz", rg=f"ripgrep-{RG_VER}-x86_64-apple-darwin.tar.gz"),
|
|
19
|
+
"windows-x64": dict(fd=f"fd-{FD_VER}-x86_64-pc-windows-msvc.zip", rg=f"ripgrep-{RG_VER}-x86_64-pc-windows-msvc.zip"),
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
def fetch(url):
|
|
23
|
+
import time
|
|
24
|
+
req = urllib.request.Request(url, headers={"User-Agent": "omnirush-release-builder"})
|
|
25
|
+
for attempt in range(3):
|
|
26
|
+
try:
|
|
27
|
+
with urllib.request.urlopen(req, timeout=300) as r:
|
|
28
|
+
return r.read()
|
|
29
|
+
except Exception as e:
|
|
30
|
+
if attempt == 2:
|
|
31
|
+
raise
|
|
32
|
+
print("retry", url.split("/")[-1], "after", e)
|
|
33
|
+
time.sleep(5)
|
|
34
|
+
|
|
35
|
+
def extract_zip(data, dest):
|
|
36
|
+
os.makedirs(dest, exist_ok=True)
|
|
37
|
+
zf = zipfile.ZipFile(io.BytesIO(data))
|
|
38
|
+
zf.extractall(dest)
|
|
39
|
+
|
|
40
|
+
def extract_tgz(data, dest):
|
|
41
|
+
os.makedirs(dest, exist_ok=True)
|
|
42
|
+
tf = tarfile.open(fileobj=io.BytesIO(data), mode="r:gz")
|
|
43
|
+
tf.extractall(dest)
|
|
44
|
+
|
|
45
|
+
def find_bin(root, names):
|
|
46
|
+
for dirpath, _, files in os.walk(root):
|
|
47
|
+
for f in files:
|
|
48
|
+
if f in names:
|
|
49
|
+
return os.path.join(dirpath, f)
|
|
50
|
+
return None
|
|
51
|
+
|
|
52
|
+
for plat, cfg in PLATFORMS.items():
|
|
53
|
+
if ONLY and plat not in ONLY:
|
|
54
|
+
continue
|
|
55
|
+
out_pre = f"omnirush-cli-{plat}.tar.gz" if plat == "linux-x64" else f"omnirush-cli-{plat}.zip"
|
|
56
|
+
if os.path.exists(out_pre) and not ONLY:
|
|
57
|
+
print("skip (built):", plat)
|
|
58
|
+
continue
|
|
59
|
+
stage = os.path.join(WORK, f"omnirush-cli-{plat}")
|
|
60
|
+
shutil.rmtree(stage, ignore_errors=True)
|
|
61
|
+
tmp = os.path.join(WORK, f".tmp-{plat}")
|
|
62
|
+
shutil.rmtree(tmp, ignore_errors=True)
|
|
63
|
+
os.makedirs(os.path.join(stage, ".runtime", "bin"))
|
|
64
|
+
os.makedirs(tmp)
|
|
65
|
+
shutil.copytree(os.path.join(REPO, "src"), os.path.join(stage, "src"))
|
|
66
|
+
shutil.copytree(os.path.join(REPO, "assets"), os.path.join(stage, "assets"))
|
|
67
|
+
shutil.copy2(os.path.join(REPO, "package.json"), os.path.join(stage, "package.json"))
|
|
68
|
+
exe = "bun.exe" if plat == "windows-x64" else "bun"
|
|
69
|
+
bun_plat = plat.replace("arm64", "aarch64")
|
|
70
|
+
bunzip = fetch(f"https://github.com/oven-sh/bun/releases/download/{BUN_VER}/bun-{bun_plat}.zip")
|
|
71
|
+
extract_zip(bunzip, tmp)
|
|
72
|
+
extracted_dir = os.path.join(tmp, f"bun-{bun_plat}")
|
|
73
|
+
shutil.copy2(os.path.join(extracted_dir, exe), os.path.join(stage, ".runtime", "bin", exe))
|
|
74
|
+
if plat != "windows-x64":
|
|
75
|
+
os.chmod(os.path.join(stage, ".runtime", "bin", "bun"), 0o755)
|
|
76
|
+
def fetch_extract(url, dest, cfg_value):
|
|
77
|
+
data = fetch(url)
|
|
78
|
+
if cfg_value.endswith(".zip"):
|
|
79
|
+
extract_zip(data, dest)
|
|
80
|
+
else:
|
|
81
|
+
extract_tgz(data, dest)
|
|
82
|
+
fetch_extract(f"https://github.com/sharkdp/fd/releases/download/{FD_VER}/{cfg['fd']}", os.path.join(tmp, "fd"), cfg["fd"])
|
|
83
|
+
fetch_extract(f"https://github.com/BurntSushi/ripgrep/releases/download/{RG_VER}/{cfg['rg']}", os.path.join(tmp, "rg"), cfg["rg"])
|
|
84
|
+
fb = find_bin(os.path.join(tmp, "fd"), {"fd", "fd.exe"})
|
|
85
|
+
rb = find_bin(os.path.join(tmp, "rg"), {"rg", "rg.exe"})
|
|
86
|
+
assert fb and rb, f"fd/rg not found for {plat}"
|
|
87
|
+
shutil.copy2(fb, os.path.join(stage, ".runtime", "bin", os.path.basename(fb)))
|
|
88
|
+
shutil.copy2(rb, os.path.join(stage, ".runtime", "bin", os.path.basename(rb)))
|
|
89
|
+
if plat == "linux-x64":
|
|
90
|
+
out = f"omnirush-cli-{plat}.tar.gz"
|
|
91
|
+
with tarfile.open(out, "w:gz") as t:
|
|
92
|
+
t.add(stage, arcname=os.path.basename(stage))
|
|
93
|
+
else:
|
|
94
|
+
with tarfile.open(f"omnirush-cli-{plat}.zip", "w") as t: # plain zip via tarfile? no
|
|
95
|
+
pass
|
|
96
|
+
# use zipfile properly
|
|
97
|
+
import zipfile as zfmod
|
|
98
|
+
with zfmod.ZipFile(f"omnirush-cli-{plat}.zip", "w", zfmod.ZIP_DEFLATED) as z:
|
|
99
|
+
for dirpath, _, files in os.walk(stage):
|
|
100
|
+
for f in files:
|
|
101
|
+
full = os.path.join(dirpath, f)
|
|
102
|
+
z.write(full, os.path.relpath(full, WORK))
|
|
103
|
+
print("built", plat)
|
|
104
|
+
shutil.rmtree(stage, ignore_errors=True)
|
|
105
|
+
shutil.rmtree(tmp, ignore_errors=True)
|
|
106
|
+
print("ALL PLATFORMS BUILT")
|