wtagent 0.1.0 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,9 +1,11 @@
1
1
  import { createHash } from "node:crypto";
2
2
  import {
3
3
  cdata,
4
+ extractTrailingProse,
4
5
  parseAgentResponse,
5
6
  serializeProtocolError,
6
7
  serializeToolResult,
8
+ stripUiNoiseLines,
7
9
  } from "../protocol/xml-protocol.js";
8
10
  import { appendSystemReminder } from "../protocol/markers.js";
9
11
  import {
@@ -17,14 +19,18 @@ import {
17
19
  toolResultOutput,
18
20
  userMessage,
19
21
  } from "../session/canonical-transcript.js";
20
- import { DEFAULT_LIMITS } from "../shared/limits.js";
22
+ import {
23
+ DEFAULT_LIMITS,
24
+ PRO_MODEL_TURN_TIMEOUT_MS,
25
+ isProMode,
26
+ } from "../shared/limits.js";
21
27
  import { utf8ByteLength } from "../shared/text-budget.js";
22
28
  import {
23
29
  BrowserAdapterError,
24
30
  ProtocolError,
25
31
  ToolValidationError,
26
32
  } from "../shared/errors.js";
27
- import { isConnectionLostError } from "../browser/chatgpt-web-adapter.js";
33
+ import { isConnectionLostError } from "../browser/base-web-adapter.js";
28
34
  import { isUsageLimitNotice } from "../shared/usage-limit.js";
29
35
 
30
36
  const EMPTY_ASSISTANT_CONTINUE_MESSAGE =
@@ -37,6 +43,12 @@ const DEAD_REQUEST_CONTINUE_MESSAGE =
37
43
  + "from the existing conversation context. Do not repeat any local tool operation "
38
44
  + "whose result is already present. Reply using the required <agent_response> XML protocol.";
39
45
 
46
+ const GENERATION_FAILED_CONTINUE_MESSAGE =
47
+ "The previous reply was a provider-side generation failure (server error), not an answer. "
48
+ + "Retry the immediately preceding task from the existing conversation context. Do not "
49
+ + "repeat any local tool operation whose result is already present. Reply using the "
50
+ + "required <agent_response> XML protocol.";
51
+
40
52
  function canonicalize(value) {
41
53
  if (Array.isArray(value)) {
42
54
  return value.map(canonicalize);
@@ -321,6 +333,16 @@ export class AgentRuntime {
321
333
  }
322
334
  }
323
335
  }
336
+ // ChatGPT Pro can think considerably longer before the first token;
337
+ // every other provider/mode uses the default. An explicit
338
+ // --model-turn-timeout-ms always wins (resolveLimits marks it). The
339
+ // active mode may differ from the requested one (Pro limited, fallback),
340
+ // so prefer the mode the conversation is actually on.
341
+ const modelTurnTimeoutMs = !this.limits.modelTurnTimeoutExplicit
342
+ && (isProMode(activeMode) || isProMode(requestedMode))
343
+ ? PRO_MODEL_TURN_TIMEOUT_MS
344
+ : this.limits.modelTurnTimeoutMs;
345
+
324
346
  await this.session.update({
325
347
  phase: "running",
326
348
  conversationUrl: await this.adapter.getConversationUrl(),
@@ -352,7 +374,7 @@ export class AgentRuntime {
352
374
  initialMessage = this.buildToolResultMessage(pendingToolResult, { suffix });
353
375
  initialKind = "pending_tool_result";
354
376
  } else if (resume && instruction?.trim()) {
355
- // The live ChatGPT conversation already contains the bootstrap protocol
377
+ // The live web conversation already contains the bootstrap protocol
356
378
  // and tool catalog. A normal follow-up should be the user's message, not
357
379
  // another several-thousand-character protocol bootstrap. sendMessage()
358
380
  // still appends the short format reminder.
@@ -361,7 +383,7 @@ export class AgentRuntime {
361
383
  initialKind = "follow_up";
362
384
  } else if (resume && inPlaceRecovery) {
363
385
  // The original request/tool result is already visible in this live web
364
- // conversation. Ask ChatGPT to continue without duplicating transport
386
+ // conversation. Ask the provider to continue without duplicating transport
365
387
  // payloads, attachments, or canonical transcript entries.
366
388
  initialMessage = EMPTY_ASSISTANT_CONTINUE_MESSAGE;
367
389
  initialKind = "empty_response_recovery";
@@ -415,7 +437,7 @@ export class AgentRuntime {
415
437
  for (;;) {
416
438
  try {
417
439
  raw = await this.adapter.waitForTurnComplete({
418
- timeoutMs: this.limits.modelTurnTimeoutMs,
440
+ timeoutMs: modelTurnTimeoutMs,
419
441
  stableWindowMs: this.limits.modelStableWindowMs,
420
442
  emptyResponseWindowMs: this.limits.emptyAssistantWindowMs,
421
443
  deadRequestGraceMs: this.limits.deadRequestGraceMs,
@@ -451,9 +473,11 @@ export class AgentRuntime {
451
473
  }
452
474
 
453
475
  const deadRequest = error?.code === "DEAD_ASSISTANT_REQUEST";
476
+ const generationFailed = error?.code === "GENERATION_FAILED";
454
477
  if (
455
478
  error?.code !== "EMPTY_ASSISTANT_RESPONSE"
456
479
  && !deadRequest
480
+ && !generationFailed
457
481
  ) {
458
482
  throw error;
459
483
  }
@@ -474,11 +498,14 @@ export class AgentRuntime {
474
498
  retries: emptyAssistantRetries,
475
499
  assistantMessageId: emptyAssistantMessageId,
476
500
  deadRequest,
501
+ generationFailed,
477
502
  });
478
503
  throw new BrowserAdapterError(
479
504
  deadRequest
480
- ? `ChatGPT did not respond after ${emptyAssistantRetries} continuation attempts.`
481
- : `ChatGPT returned empty responses after ${emptyAssistantRetries} continuation attempts.`,
505
+ ? `${this.adapter.providerName} did not respond after ${emptyAssistantRetries} continuation attempts.`
506
+ : generationFailed
507
+ ? `${this.adapter.providerName} generation kept failing after ${emptyAssistantRetries} continuation attempts.`
508
+ : `${this.adapter.providerName} returned empty responses after ${emptyAssistantRetries} continuation attempts.`,
482
509
  {
483
510
  code: "EMPTY_ASSISTANT_RETRIES_EXHAUSTED",
484
511
  cause: error,
@@ -493,6 +520,7 @@ export class AgentRuntime {
493
520
  maxRetries: this.limits.maxEmptyAssistantRetries,
494
521
  assistantMessageId: emptyAssistantMessageId,
495
522
  deadRequest,
523
+ generationFailed,
496
524
  });
497
525
  // Do not resend the original request or tool result: both are already
498
526
  // present in ChatGPT's conversation. This transport-only continuation
@@ -500,12 +528,16 @@ export class AgentRuntime {
500
528
  await this.sendMessage(
501
529
  deadRequest
502
530
  ? DEAD_REQUEST_CONTINUE_MESSAGE
503
- : EMPTY_ASSISTANT_CONTINUE_MESSAGE,
531
+ : generationFailed
532
+ ? GENERATION_FAILED_CONTINUE_MESSAGE
533
+ : EMPTY_ASSISTANT_CONTINUE_MESSAGE,
504
534
  );
505
535
  await this.emit("model.message_sent", {
506
536
  kind: deadRequest
507
537
  ? "dead_request_recovery"
508
- : "empty_response_recovery",
538
+ : generationFailed
539
+ ? "generation_failed_recovery"
540
+ : "empty_response_recovery",
509
541
  retry: emptyAssistantRetries,
510
542
  });
511
543
  }
@@ -545,12 +577,47 @@ export class AgentRuntime {
545
577
  snippet: raw.slice(0, 200),
546
578
  });
547
579
  throw new BrowserAdapterError(
548
- "ChatGPT reported a usage limit. Try a different thinking level "
549
- + "(e.g. wtagent resume with --mode Pro or --mode Current), wait "
550
- + "for the limit to reset, or change plans, then resume.",
580
+ `${this.adapter.providerName} reported a usage limit. Wait for the limit `
581
+ + "to reset, try a different mode on resume, or change plans, then resume.",
551
582
  { code: "USAGE_LIMIT_REACHED" },
552
583
  );
553
584
  }
585
+
586
+ // The model answered in plain prose without any <agent_response> at
587
+ // all. That cannot be a broken tool request (tools only exist inside a
588
+ // parsed envelope), so it is safe to treat the prose as the final
589
+ // answer: end the run and show it, instead of burning retries on a
590
+ // model that deliberately finished the conversation. A reply that DOES
591
+ // contain <agent_response but fails to parse keeps the retry path —
592
+ // the model tried the protocol and we must not guess its intent.
593
+ // A bare <tool_calls>/<invoke> reply (no envelope) is likewise a tool
594
+ // REQUEST, never prose: it goes to the protocol-error retry below.
595
+ const looksLikeToolRequest = /<tool_calls[\s>]|<tool_call[\s>]|<invoke[\s>]/i
596
+ .test(raw);
597
+ const plainAnswer = !raw.includes("<agent_response")
598
+ && !looksLikeToolRequest
599
+ ? stripUiNoiseLines(raw)
600
+ : "";
601
+ if (plainAnswer) {
602
+ await this.emit("protocol.plain_answer", {
603
+ snippet: plainAnswer.slice(0, 200),
604
+ });
605
+ await this.session.update({
606
+ phase: "idle",
607
+ lastMessage: plainAnswer,
608
+ pendingToolResult: null,
609
+ });
610
+ await this.session.appendTranscriptItem(assistantMessage(plainAnswer));
611
+ await this.emit("run.completed", {
612
+ message: plainAnswer,
613
+ plainAnswer: true,
614
+ });
615
+ return {
616
+ sessionId: this.session.sessionId,
617
+ message: plainAnswer,
618
+ };
619
+ }
620
+
554
621
  protocolErrors += 1;
555
622
  await this.emit("protocol.invalid", {
556
623
  message: error.message,
@@ -565,16 +632,33 @@ export class AgentRuntime {
565
632
  continue;
566
633
  }
567
634
 
635
+ // When the run finishes, some models (GLM especially) put their real
636
+ // deliverable AFTER the envelope: a short done/true stub followed by the
637
+ // full markdown/HTML answer. That trailing prose is pure display content
638
+ // (it cannot trigger a tool), so merge it into the final message instead
639
+ // of dropping it. Non-done turns keep the protocol message untouched.
640
+ let finalMessage = parsed.message;
641
+ let usedTrailingProse = false;
642
+ if (parsed.done) {
643
+ const trailing = extractTrailingProse(raw);
644
+ if (trailing) {
645
+ finalMessage = [parsed.message.trim(), trailing]
646
+ .filter(Boolean)
647
+ .join("\n\n");
648
+ usedTrailingProse = true;
649
+ }
650
+ }
651
+
568
652
  // Record the assistant's turn in the canonical transcript. The raw XML is
569
653
  // the web rendering; the transcript keeps the plain progress message.
570
- if (parsed.message?.trim()) {
654
+ if (finalMessage?.trim()) {
571
655
  await this.session.appendTranscriptItem(
572
- assistantMessage(parsed.message),
656
+ assistantMessage(finalMessage),
573
657
  );
574
658
  }
575
659
 
576
660
  if (parsed.done) {
577
- if (!parsed.message.trim()) {
661
+ if (!finalMessage.trim()) {
578
662
  const error = new ProtocolError(
579
663
  "done=true requires a non-empty final message.",
580
664
  );
@@ -582,19 +666,48 @@ export class AgentRuntime {
582
666
  await this.sendMessage(serializeProtocolError(error));
583
667
  continue;
584
668
  }
669
+ // Some models (DeepSeek especially) write their tool request inside
670
+ // the <message> of a done=true envelope instead of a <tool_call>
671
+ // element. Completing there would swallow the tool call and end the
672
+ // run with raw XML as the "answer". Treat it as a format slip: the
673
+ // protocol-error feedback tells the model where tool calls go, and
674
+ // the normal retry limit still applies.
675
+ if (
676
+ !parsed.toolCall
677
+ && /<tool_calls[\s>]|<tool_call[\s>]|<invoke[\s>]/i.test(finalMessage)
678
+ ) {
679
+ const error = new ProtocolError(
680
+ "Tool calls must use the <tool_call> element, not the <message> text.",
681
+ );
682
+ protocolErrors += 1;
683
+ await this.emit("protocol.invalid", {
684
+ message: error.message,
685
+ count: protocolErrors,
686
+ });
687
+ if (protocolErrors >= this.limits.maxProtocolErrors) {
688
+ throw new ProtocolError(
689
+ `Protocol failed ${protocolErrors} consecutive times: ${error.message}`,
690
+ );
691
+ }
692
+ await this.sendMessage(serializeProtocolError(error));
693
+ continue;
694
+ }
585
695
  // done=true completes the run. A request may be answered directly
586
696
  // (no tool call) or after any number of tools; the runtime does not
587
697
  // second-guess whether "enough" work happened — that is the model's
588
698
  // and the user's call, not a keyword heuristic.
589
699
  await this.session.update({
590
700
  phase: "idle",
591
- lastMessage: parsed.message,
701
+ lastMessage: finalMessage,
592
702
  pendingToolResult: null,
593
703
  });
594
- await this.emit("run.completed", { message: parsed.message });
704
+ await this.emit("run.completed", {
705
+ message: finalMessage,
706
+ usedTrailingProse,
707
+ });
595
708
  return {
596
709
  sessionId: this.session.sessionId,
597
- message: parsed.message,
710
+ message: finalMessage,
598
711
  };
599
712
  }
600
713
 
@@ -155,7 +155,7 @@ export class AgentSession {
155
155
  this.stateFileName = stateFileName;
156
156
  }
157
157
 
158
- static async create({ sessionsDir, tasksDir, task, projectRoot, mode }) {
158
+ static async create({ sessionsDir, tasksDir, task, projectRoot, mode, provider = "chatgpt" }) {
159
159
  const root = await ensureSessionsRoot(sessionsDir ?? tasksDir);
160
160
  const sessionId = `session_${new Date().toISOString().replaceAll(/[-:.TZ]/g, "").slice(0, 14)}_${randomUUID().slice(0, 8)}`;
161
161
  const directory = path.join(root, sessionId);
@@ -167,6 +167,7 @@ export class AgentSession {
167
167
  sessionId,
168
168
  task,
169
169
  projectRoot: path.resolve(projectRoot),
170
+ provider,
170
171
  mode,
171
172
  phase: "idle",
172
173
  turn: 0,
@@ -197,6 +198,7 @@ export class AgentSession {
197
198
  await session.appendEvent("session.created", {
198
199
  task,
199
200
  projectRoot: state.projectRoot,
201
+ provider,
200
202
  mode,
201
203
  });
202
204
  return session;
@@ -237,6 +239,7 @@ export class AgentSession {
237
239
  state.phase ??= ["completed", "paused"].includes(state.status)
238
240
  ? "idle"
239
241
  : (state.status ?? "idle");
242
+ state.provider ??= "chatgpt";
240
243
  state.runCount ??= 0;
241
244
  state.lastAssistantMessageId ??= null;
242
245
  state.activeMode ??= null;
@@ -1,7 +1,7 @@
1
1
  // Canonical conversation transcript.
2
2
  //
3
3
  // The source of truth for a task's dialogue is stored in this Codex-compatible
4
- // shape from the very first exchange. ChatGPT Web only ever sees a rendered
4
+ // shape from the very first exchange. The web provider only ever sees a rendered
5
5
  // projection (XML for tool results, a marked prompt for instructions); the
6
6
  // structured record below is what we persist and later export to Codex or
7
7
  // Claude Code sessions.
@@ -39,7 +39,7 @@ export function developerMessage(text) {
39
39
 
40
40
  // A user message. `attachments` records any @file uploads that accompanied the
41
41
  // message on the web transport (name + local path), so exporters can note them
42
- // even though the uploaded bytes live only in the ChatGPT conversation.
42
+ // even though the uploaded bytes live only in the web conversation.
43
43
  export function userMessage(text, { attachments = [] } = {}) {
44
44
  const item = {
45
45
  type: "message",
@@ -23,7 +23,7 @@ function firstTimestamp(items, fallback) {
23
23
  return items.find((item) => item.timestamp)?.timestamp ?? fallback;
24
24
  }
25
25
 
26
- // The uploaded bytes of an @file attachment live only in the ChatGPT
26
+ // The uploaded bytes of an @file attachment live only in the web provider
27
27
  // conversation, not in the portable transcript. So exports append a short,
28
28
  // self-describing note naming the attached files, keeping the exported session
29
29
  // honest about what the user actually provided.
@@ -70,7 +70,7 @@ export function toCodexRollout(transcript, { now } = {}) {
70
70
  for (const entry of items) {
71
71
  const item = entry.item;
72
72
  if (isDeveloper(item)) {
73
- // This is the XML/tool transport scaffold used only on ChatGPT Web.
73
+ // This is the XML/tool transport scaffold used only by the web provider.
74
74
  continue;
75
75
  }
76
76
 
@@ -0,0 +1,24 @@
1
+ export const NETWORK_TIMEOUT_MS = 2_000;
2
+
3
+ export async function fetchJson(url, {
4
+ timeoutMs = NETWORK_TIMEOUT_MS,
5
+ fetchImpl = fetch,
6
+ headers = { "User-Agent": "wtagent", Accept: "application/json" },
7
+ } = {}) {
8
+ const controller = new AbortController();
9
+ const timer = setTimeout(() => controller.abort(), timeoutMs);
10
+ try {
11
+ const response = await fetchImpl(url, {
12
+ signal: controller.signal,
13
+ headers,
14
+ });
15
+ if (!response.ok) {
16
+ return null;
17
+ }
18
+ return await response.json();
19
+ } catch {
20
+ return null;
21
+ } finally {
22
+ clearTimeout(timer);
23
+ }
24
+ }
@@ -1,7 +1,7 @@
1
1
  export const DEFAULT_LIMITS = Object.freeze({
2
2
  maxProtocolErrors: 3,
3
3
  maxEmptyAssistantRetries: 3,
4
- modelTurnTimeoutMs: 20 * 60_000,
4
+ modelTurnTimeoutMs: 10 * 60_000,
5
5
  modelStableWindowMs: 1_500,
6
6
  emptyAssistantWindowMs: 10_000,
7
7
  // How long a sent message may sit with neither a reply node nor a stop
@@ -19,6 +19,15 @@ export const DEFAULT_LIMITS = Object.freeze({
19
19
  maxSearchResults: 200,
20
20
  });
21
21
 
22
+ // ChatGPT Pro plans can legitimately think longer than the default before the
23
+ // first token (see the runtime's mode-aware timeout selection).
24
+ export const PRO_MODEL_TURN_TIMEOUT_MS = 16 * 60_000;
25
+
26
+ // The CLI's --model-turn-timeout-ms always wins over the mode-aware default.
27
+ export function isProMode(mode) {
28
+ return String(mode ?? "") === "Pro";
29
+ }
30
+
22
31
  export function resolveLimits({ modelTurnTimeoutMs } = {}) {
23
32
  if (modelTurnTimeoutMs == null || modelTurnTimeoutMs === "") {
24
33
  return DEFAULT_LIMITS;
@@ -34,5 +43,6 @@ export function resolveLimits({ modelTurnTimeoutMs } = {}) {
34
43
  return Object.freeze({
35
44
  ...DEFAULT_LIMITS,
36
45
  modelTurnTimeoutMs: parsed,
46
+ modelTurnTimeoutExplicit: true,
37
47
  });
38
48
  }
@@ -0,0 +1,23 @@
1
+ import { readFileSync } from "node:fs";
2
+ import path from "node:path";
3
+ import { fileURLToPath } from "node:url";
4
+
5
+ const PACKAGE_JSON_PATH = path.resolve(
6
+ path.dirname(fileURLToPath(import.meta.url)),
7
+ "../../package.json",
8
+ );
9
+
10
+ let cached;
11
+
12
+ export function getPackageInfo() {
13
+ cached ??= JSON.parse(readFileSync(PACKAGE_JSON_PATH, "utf8"));
14
+ return cached;
15
+ }
16
+
17
+ export function getPackageName() {
18
+ return getPackageInfo().name;
19
+ }
20
+
21
+ export function getPackageVersion() {
22
+ return getPackageInfo().version;
23
+ }
@@ -1,8 +1,8 @@
1
- // ChatGPT renders plan/usage limit notices as ordinary assistant messages,
1
+ // Web providers render plan/usage limit notices as ordinary assistant messages,
2
2
  // localized per UI language. Matching one means "stop, not retry". Text
3
3
  // patterns are the primary signal (a limit notice always says something
4
4
  // recognizable); callers may additionally confirm via DOM features (e.g. the
5
- // retry/upgrade button ChatGPT shows on the notice) to guard against protocol
5
+ // retry/upgrade button shown on the notice) to guard against protocol
6
6
  // replies that merely mention "limit" in their content.
7
7
  const USAGE_LIMIT_PATTERNS = [
8
8
  /reached (?:your )?(?:current )?(?:usage |plan )?limit/i,