wtagent 0.1.0 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -54,8 +54,9 @@ function summarizeArgs(name, args = {}) {
54
54
  }
55
55
 
56
56
  export class Renderer {
57
- constructor({ stream = process.stdout } = {}) {
57
+ constructor({ stream = process.stdout, providerLabel = "model" } = {}) {
58
58
  this.stream = stream;
59
+ this.providerLabel = providerLabel;
59
60
  this.isTTY = Boolean(stream.isTTY);
60
61
  this.spinner = null;
61
62
  this.timer = null;
@@ -263,10 +264,10 @@ export class Renderer {
263
264
  break;
264
265
  case "browser.auth_required":
265
266
  this.stopSpinner();
266
- this.println(`${YELLOW}Log in to ChatGPT in the opened Chrome window…${RESET}`);
267
+ this.println(`${YELLOW}Log in to ${this.providerLabel} in the opened Chrome window…${RESET}`);
267
268
  break;
268
269
  case "browser.authenticated":
269
- this.println(`${GREEN}ChatGPT login detected.${RESET}`);
270
+ this.println(`${GREEN}${this.providerLabel} login detected.${RESET}`);
270
271
  break;
271
272
  case "conversation.mode_selected": {
272
273
  const { requested, status, selectedLabel } = payload;
@@ -304,16 +305,17 @@ export class Renderer {
304
305
  this.stopSpinner();
305
306
  this.note(
306
307
  payload.deadRequest
307
- ? `no reply from ChatGPT; asking it to continue (${payload.retry}/${payload.maxRetries})`
308
- : `empty ChatGPT response; asking it to continue (${payload.retry}/${payload.maxRetries})`,
308
+ ? `no reply from ${this.providerLabel}; asking it to continue (${payload.retry}/${payload.maxRetries})`
309
+ : payload.generationFailed
310
+ ? `${this.providerLabel} generation failed (server error); asking it to retry (${payload.retry}/${payload.maxRetries})`
311
+ : `empty ${this.providerLabel} response; asking it to continue (${payload.retry}/${payload.maxRetries})`,
309
312
  );
310
313
  break;
311
314
  case "model.limit_reached":
312
315
  this.stopSpinner();
313
316
  this.note(
314
- "ChatGPT usage limit reached. Try a different thinking level on resume "
315
- + "(wtagent resume <session-id> --mode Pro or --mode Current), wait "
316
- + "for the limit to reset, or change plans.",
317
+ `${this.providerLabel} usage limit reached. Wait for the limit to reset, `
318
+ + "try a different mode on resume, or change plans.",
317
319
  );
318
320
  break;
319
321
  case "model.progress":
@@ -325,6 +327,10 @@ export class Renderer {
325
327
  case "protocol.invalid":
326
328
  this.note(`format retry${payload.count ? ` (${payload.count})` : ""}: ${truncate(payload.message, 100)}`);
327
329
  break;
330
+ case "protocol.plain_answer":
331
+ this.stopSpinner();
332
+ this.note("The model answered without the XML protocol; showing the reply as the final answer.");
333
+ break;
328
334
  case "tool.proposed":
329
335
  this.stopSpinner();
330
336
  this.#toolCall(payload.name, payload.args);
@@ -0,0 +1,145 @@
1
+ import { spawn } from "node:child_process";
2
+ import { resolveLaunchPlan } from "../platform/command-launcher.js";
3
+ import { fetchJson, NETWORK_TIMEOUT_MS } from "../shared/fetch-json.js";
4
+ import { getPackageName, getPackageVersion } from "../shared/package-info.js";
5
+
6
+ export const UPDATE_CHECK_TIMEOUT_MS = NETWORK_TIMEOUT_MS;
7
+ export const UPDATE_COMMAND_TIMEOUT_MS = 10_000;
8
+ export const INSTALL_TIMEOUT_MS = 120_000;
9
+ export const MANUAL_INSTALL_COMMAND = "npm install -g wtagent@latest";
10
+
11
+ const VERSION_RE = /^v?(\d+)\.(\d+)\.(\d+)(?:-([0-9A-Za-z.-]+))?(?:\+[0-9A-Za-z.-]+)?$/;
12
+
13
+ export function npmRegistryLatestUrl(name = getPackageName()) {
14
+ return `https://registry.npmjs.org/${encodeURIComponent(name)}/latest`;
15
+ }
16
+
17
+ export function parseVersion(value) {
18
+ const match = String(value ?? "").trim().match(VERSION_RE);
19
+ if (!match) {
20
+ return null;
21
+ }
22
+ return {
23
+ core: [Number(match[1]), Number(match[2]), Number(match[3])],
24
+ pre: match[4] ?? null,
25
+ };
26
+ }
27
+
28
+ export function compareVersions(left, right) {
29
+ const parsedLeft = parseVersion(left);
30
+ const parsedRight = parseVersion(right);
31
+ if (!parsedLeft || !parsedRight) {
32
+ return null;
33
+ }
34
+ for (let index = 0; index < 3; index += 1) {
35
+ if (parsedLeft.core[index] !== parsedRight.core[index]) {
36
+ return parsedLeft.core[index] < parsedRight.core[index] ? -1 : 1;
37
+ }
38
+ }
39
+ if (!parsedLeft.pre && !parsedRight.pre) {
40
+ return 0;
41
+ }
42
+ if (!parsedLeft.pre) {
43
+ return 1;
44
+ }
45
+ if (!parsedRight.pre) {
46
+ return -1;
47
+ }
48
+ if (parsedLeft.pre === parsedRight.pre) {
49
+ return 0;
50
+ }
51
+ return parsedLeft.pre < parsedRight.pre ? -1 : 1;
52
+ }
53
+
54
+ export function isRemoteNewer(remote, local = getPackageVersion()) {
55
+ const comparison = compareVersions(local, remote);
56
+ return comparison != null && comparison < 0;
57
+ }
58
+
59
+ export async function fetchLatestVersion({
60
+ fetchImpl,
61
+ timeoutMs = UPDATE_CHECK_TIMEOUT_MS,
62
+ packageName = getPackageName(),
63
+ } = {}) {
64
+ const document = await fetchJson(npmRegistryLatestUrl(packageName), {
65
+ fetchImpl,
66
+ timeoutMs,
67
+ });
68
+ return typeof document?.version === "string" && document.version.trim()
69
+ ? document.version.trim()
70
+ : null;
71
+ }
72
+
73
+ export async function installLatest({
74
+ spawnImpl = spawn,
75
+ planCommandImpl = (program, argv) => resolveLaunchPlan({ program, argv }),
76
+ timeoutMs = INSTALL_TIMEOUT_MS,
77
+ stdio = "inherit",
78
+ } = {}) {
79
+ const plan = planCommandImpl("npm", ["install", "-g", "wtagent@latest"]);
80
+ return await new Promise((resolve) => {
81
+ let settled = false;
82
+ let timer;
83
+ const finish = (result) => {
84
+ if (settled) {
85
+ return;
86
+ }
87
+ settled = true;
88
+ clearTimeout(timer);
89
+ resolve(result);
90
+ };
91
+
92
+ let child;
93
+ try {
94
+ child = spawnImpl(plan.command, plan.args, {
95
+ stdio,
96
+ shell: plan.shell,
97
+ windowsVerbatimArguments: plan.windowsVerbatimArguments,
98
+ });
99
+ } catch (error) {
100
+ finish({ ok: false, error });
101
+ return;
102
+ }
103
+
104
+ timer = setTimeout(() => {
105
+ child.kill();
106
+ finish({ ok: false, error: new Error("npm install timed out.") });
107
+ }, timeoutMs);
108
+
109
+ child.once("error", (error) => {
110
+ finish({ ok: false, error });
111
+ });
112
+ child.once("exit", (code) => {
113
+ finish({ ok: code === 0, code });
114
+ });
115
+ });
116
+ }
117
+
118
+ export async function runSelfUpdate({
119
+ fetchLatest = fetchLatestVersion,
120
+ install = installLatest,
121
+ write = console.log,
122
+ writeError = console.error,
123
+ currentVersion = getPackageVersion(),
124
+ } = {}) {
125
+ const latest = await fetchLatest({ timeoutMs: UPDATE_COMMAND_TIMEOUT_MS });
126
+ if (!latest) {
127
+ writeError("Could not check npm for the latest WTAgent version.");
128
+ writeError(`Install it manually: ${MANUAL_INSTALL_COMMAND}`);
129
+ return { status: "error" };
130
+ }
131
+ if (!isRemoteNewer(latest, currentVersion)) {
132
+ write(`Already up to date (${currentVersion}).`);
133
+ return { status: "current", currentVersion, latest };
134
+ }
135
+
136
+ write(`Updating WTAgent ${currentVersion} → ${latest} ...`);
137
+ const result = await install();
138
+ if (!result.ok) {
139
+ writeError("Update failed.");
140
+ writeError(`Install it manually: ${MANUAL_INSTALL_COMMAND}`);
141
+ return { status: "error", currentVersion, latest };
142
+ }
143
+ write(`Updated to ${latest}. Run wtagent again to use the new version.`);
144
+ return { status: "updated", currentVersion, latest };
145
+ }
@@ -0,0 +1,161 @@
1
+ import { fetchJson, NETWORK_TIMEOUT_MS } from "../shared/fetch-json.js";
2
+ import { getPackageVersion } from "../shared/package-info.js";
3
+ import { NoticeStore, getNoticeStorePath } from "./notice-store.js";
4
+ import {
5
+ MANUAL_INSTALL_COMMAND,
6
+ fetchLatestVersion,
7
+ installLatest,
8
+ isRemoteNewer,
9
+ } from "./self-update.js";
10
+
11
+ export const NOTICES_URL =
12
+ "https://raw.githubusercontent.com/XiXian42/wtagent/main/notices.json";
13
+ export const NOTICE_TIMEOUT_MS = NETWORK_TIMEOUT_MS;
14
+ const MAX_NOTICE_TEXT = 2_000;
15
+ const MAX_NOTICE_ID = 128;
16
+
17
+ const CYAN = "\x1b[36m";
18
+ const RESET = "\x1b[0m";
19
+
20
+ export function sanitizeNoticeText(value) {
21
+ return String(value ?? "")
22
+ .replace(/\r\n|\r/g, "\n")
23
+ .replace(/\u001b\[[0-9;?]*[ -/]*[@-~]/g, "")
24
+ .replace(/\u001b\][^\u0007\u001b]*(?:\u0007|\u001b\\)?/g, "")
25
+ .replace(/[\u0000-\u0008\u000B\u000C\u000E-\u001F\u007F-\u009F]/g, "")
26
+ .trim()
27
+ .slice(0, MAX_NOTICE_TEXT);
28
+ }
29
+
30
+ function normalizeTimes(value) {
31
+ if (value == null) {
32
+ return 1;
33
+ }
34
+ const times = typeof value === "number" ? value : Number(value);
35
+ if (!Number.isFinite(times)) {
36
+ return 1;
37
+ }
38
+ return Math.max(0, Math.floor(times));
39
+ }
40
+
41
+ export function parseNoticeDocument(document) {
42
+ const raw = Array.isArray(document?.notices) ? document.notices : [];
43
+ const notices = [];
44
+ const seen = new Set();
45
+ for (const entry of raw) {
46
+ if (!entry || typeof entry !== "object") {
47
+ continue;
48
+ }
49
+ const id = typeof entry.id === "string" ? entry.id.trim() : "";
50
+ const text = sanitizeNoticeText(entry.text);
51
+ if (!id || id.length > MAX_NOTICE_ID || !text || seen.has(id)) {
52
+ continue;
53
+ }
54
+ seen.add(id);
55
+ notices.push({
56
+ id,
57
+ text,
58
+ times: normalizeTimes(entry.times),
59
+ });
60
+ }
61
+ return notices;
62
+ }
63
+
64
+ export function selectNotice(notices, store) {
65
+ for (const notice of notices) {
66
+ if (notice.times <= 0) {
67
+ continue;
68
+ }
69
+ if (store.shownCount(notice.id) >= notice.times) {
70
+ continue;
71
+ }
72
+ return notice;
73
+ }
74
+ return null;
75
+ }
76
+
77
+ export function formatNotice(text) {
78
+ return `\n${CYAN}Notice${RESET}\n${text}\n`;
79
+ }
80
+
81
+ export async function fetchNoticeDocument({
82
+ fetchImpl,
83
+ timeoutMs = NOTICE_TIMEOUT_MS,
84
+ url = NOTICES_URL,
85
+ } = {}) {
86
+ const document = await fetchJson(url, { fetchImpl, timeoutMs });
87
+ return document && typeof document === "object" ? document : null;
88
+ }
89
+
90
+ export async function runStartupChecks({
91
+ appDataDir,
92
+ interactive = true,
93
+ fetchLatest = fetchLatestVersion,
94
+ fetchNotices = fetchNoticeDocument,
95
+ install = installLatest,
96
+ promptUpdate,
97
+ write = (text) => {
98
+ console.log(text);
99
+ },
100
+ writeError = (text) => {
101
+ console.error(text);
102
+ },
103
+ now,
104
+ } = {}) {
105
+ if (!interactive) {
106
+ return "continue";
107
+ }
108
+
109
+ const store = new NoticeStore({
110
+ filePath: getNoticeStorePath(appDataDir),
111
+ now,
112
+ });
113
+ await store.ensureLoaded();
114
+
115
+ const needUpdateCheck = !store.wasUpdatePromptedToday();
116
+ const needNotice = !store.wasNoticeShownToday();
117
+ const latestPromise = needUpdateCheck
118
+ ? fetchLatest({ timeoutMs: NOTICE_TIMEOUT_MS })
119
+ : Promise.resolve(null);
120
+ const noticesPromise = needNotice
121
+ ? fetchNotices({ timeoutMs: NOTICE_TIMEOUT_MS })
122
+ : Promise.resolve(null);
123
+
124
+ const currentVersion = getPackageVersion();
125
+ const latest = await latestPromise;
126
+ if (needUpdateCheck && latest) {
127
+ if (isRemoteNewer(latest, currentVersion)) {
128
+ write(`A newer WTAgent is available: ${currentVersion} → ${latest}`);
129
+ const choice = await promptUpdate({ currentVersion, latest });
130
+ if (choice == null) {
131
+ return "aborted";
132
+ }
133
+ await store.markUpdatePrompted();
134
+ if (choice) {
135
+ write(`Updating WTAgent ${currentVersion} → ${latest} ...`);
136
+ const result = await install();
137
+ if (result.ok) {
138
+ write(`Updated to ${latest}. Restart wtagent to use the new version.`);
139
+ return "updated";
140
+ }
141
+ writeError("Update failed.");
142
+ writeError(`Install it manually: ${MANUAL_INSTALL_COMMAND}`);
143
+ }
144
+ } else {
145
+ await store.markUpdatePrompted();
146
+ }
147
+ }
148
+
149
+ if (!needNotice) {
150
+ return "continue";
151
+ }
152
+
153
+ const notices = parseNoticeDocument(await noticesPromise);
154
+ const selected = selectNotice(notices, store);
155
+ if (!selected) {
156
+ return "continue";
157
+ }
158
+ write(formatNotice(selected.text).trimEnd());
159
+ await store.recordNoticeShown(selected.id);
160
+ return "continue";
161
+ }
@@ -63,6 +63,13 @@ export function getChromeProfileDir(appDataDir = getAppDataDir()) {
63
63
  return path.join(appDataDir, "chrome-profile");
64
64
  }
65
65
 
66
+ // Resolves a named Chrome profile directory under the app data dir. The
67
+ // provider→basename mapping lives in the provider registry; this helper only
68
+ // joins the path so paths.js stays free of provider dependencies.
69
+ export function getProfileDir(appDataDir, basename) {
70
+ return path.join(appDataDir, basename);
71
+ }
72
+
66
73
  export function getTasksDir(appDataDir = getAppDataDir()) {
67
74
  return path.join(appDataDir, "tasks");
68
75
  }
@@ -1,5 +1,5 @@
1
1
  // Delimiters that mark WTAgent-specific control content inside the plain-text
2
- // messages exchanged with ChatGPT Web. ChatGPT Web has no real system/tool
2
+ // messages exchanged with a web AI. A normal web chat has no real system/tool
3
3
  // channel, so the protocol instructions, tool catalog, and per-turn reminders
4
4
  // all travel as ordinary chat text. These markers let the session exporters
5
5
  // deterministically strip that scaffolding when converting a transcript into a
@@ -12,12 +12,13 @@
12
12
  export const SYSTEM_PROMPT_TAG = "agent_protocol";
13
13
  export const SYSTEM_REMINDER_TAG = "system_reminder";
14
14
  export const DEFAULT_SYSTEM_REMINDER = [
15
- "This is a reminder of the user's requested response format for the WTAgent integration, not a ChatGPT system message.",
15
+ "This is a reminder of the user's requested response format for the WTAgent integration, not a provider-native system message.",
16
16
  "You do not need native tool access: write the XML request as text and the user's local Runtime will process it after your reply.",
17
17
  "Your next response must use the XML application protocol.",
18
18
  "It must contain exactly one <agent_response>...</agent_response> envelope.",
19
19
  "Inside the optional `xml` code fence, the first content must be <agent_response>",
20
20
  "and the last content must be </agent_response>; do not put text outside the envelope.",
21
+ "Write large code or file payloads as CDATA (<![CDATA[...]]>) inside <content>, <new_text>, or <message> — never raw code directly inside the envelope — and never include the sequence ]]> inside CDATA content.",
21
22
  ].join(" ");
22
23
 
23
24
  export function wrapSystemPrompt(text) {
@@ -29,14 +29,14 @@ const DONE_SEMANTICS = `Current-run completion semantics:
29
29
  function buildBootstrapScaffold({ projectRoot, tools }) {
30
30
  const toolDocs = tools.map(formatTool).join("\n\n");
31
31
 
32
- return `The user is running WTAgent, a local application that uses this ChatGPT conversation for reasoning. The following is the user's requested application-level response format and collaboration contract; it is not a claim that ChatGPT has native filesystem or function-call tools.
32
+ return `The user is running WTAgent, a local application that uses this web AI conversation for reasoning. The following is the user's requested application-level response format and collaboration contract; it is not a claim that this web chat has native filesystem or function-call tools.
33
33
 
34
- You do not need direct filesystem access or visible ChatGPT tool buttons. Return local operation requests as XML text. After your reply is complete, the user's local Node.js Runtime will parse the XML, validate the arguments, apply local policy, and may execute the requested operation. Its result will arrive in the next user message as <tool_result>. XML by itself never guarantees execution.
34
+ You do not need direct filesystem access or provider-native tool buttons. Return local operation requests as XML text. After your reply is complete, the user's local Node.js Runtime will parse the XML, validate the arguments, apply local policy, and may execute the requested operation. Its result will arrive in the next user message as <tool_result>. XML by itself never guarantees execution.
35
35
 
36
36
  You are not limited to coding tasks. You can answer questions, write text, brainstorm, analyze, summarize, and — when the task requires it — request that the user's Runtime read, create, or modify files or run commands.
37
37
 
38
38
  ## Filesystem boundary
39
- The project filesystem described below is a logical, virtual filesystem namespace exposed by the local Runtime. It is not mounted in ChatGPT's own environment and cannot be inspected directly from this webpage.
39
+ The project filesystem described below is a logical, virtual filesystem namespace exposed by the local Runtime. It is not mounted in the web provider's own environment and cannot be inspected directly from this webpage.
40
40
 
41
41
  Do not inspect /workspace, /mnt/data, or any ambient, cloud, or sandbox filesystem. Those locations are unrelated to the user's project. Request all project reads, listings, writes, edits, and commands only through the XML operations declared below.
42
42
 
@@ -98,11 +98,15 @@ The project directory may contain content unrelated to the current task: depende
98
98
  }
99
99
 
100
100
  // Returns the pieces needed by both transports:
101
- // web - the exact text to send to ChatGPT Web (scaffold is wrapped in
101
+ // web - the exact text to send to the web AI (scaffold is wrapped in
102
102
  // <agent_protocol> markers, the user task follows outside them)
103
103
  // developer - the transport scaffold, exposed for diagnostics/tests only
104
104
  // user - the user task, for the canonical user message
105
- export function buildBootstrapPrompt({ task, projectRoot, tools }) {
105
+ export function buildBootstrapPrompt({
106
+ task,
107
+ projectRoot,
108
+ tools,
109
+ }) {
106
110
  const developer = buildBootstrapScaffold({ projectRoot, tools });
107
111
  const web = `${wrapSystemPrompt(developer)}\n\n## User task\n${task}`;
108
112
  return { web, developer, user: task };
@@ -63,26 +63,187 @@ function escapeBareAmpersands(text) {
63
63
  return out;
64
64
  }
65
65
 
66
+ function closeDanglingToolCall(envelope) {
67
+ const open = (envelope.match(/<tool_call(?:\s|>)/gi) ?? []).length;
68
+ const close = (envelope.match(/<\/tool_call>/gi) ?? []).length;
69
+ if (open !== close + 1) {
70
+ return envelope;
71
+ }
72
+ return envelope.replace(
73
+ /(<\/args>\s*)(<\/agent_response>\s*)$/i,
74
+ "$1</tool_call>\n$2",
75
+ );
76
+ }
77
+
78
+ function wrapBareToolCall(text) {
79
+ const start = text.search(/<tool_call(?:\s|>)/i);
80
+ const endTag = "</tool_call>";
81
+ const end = start < 0 ? -1 : text.indexOf(endTag, start);
82
+ if (start >= 0 && end >= start) {
83
+ const toolCall = text.slice(start, end + endTag.length);
84
+ if (/<args[\s>/]/i.test(toolCall)) {
85
+ return [
86
+ "<agent_response>",
87
+ " <done>false</done>",
88
+ " <message></message>",
89
+ ` ${toolCall}`,
90
+ "</agent_response>",
91
+ ].join("\n");
92
+ }
93
+ }
94
+ return wrapClaudeStyleInvoke(text);
95
+ }
96
+
97
+ // DeepSeek (and some others) occasionally emit Claude-style tool XML:
98
+ // <tool_calls><invoke name="fs.read"><parameter name="path">README.md</parameter></invoke></tool_calls>
99
+ // and sometimes annotate parameters with attributes, e.g.
100
+ // <parameter name="program" string="true">npm</parameter>
101
+ // <parameter name="argv" string="false">["test"]</parameter>
102
+ // Map a single complete invoke onto our envelope so the turn can proceed
103
+ // instead of hanging or burning a format retry.
104
+ function wrapClaudeStyleInvoke(text) {
105
+ const invoke = /<invoke\s+name="([A-Za-z0-9_.-]+)"(?:\s[^>]*)?>([\s\S]*?)<\/invoke>/i.exec(text);
106
+ if (!invoke) {
107
+ return null;
108
+ }
109
+ const name = invoke[1];
110
+ const body = invoke[2];
111
+ const args = [];
112
+ const paramRe = /<parameter\s+name="([A-Za-z0-9_.-]+)"(?:\s[^>]*)?>([\s\S]*?)<\/parameter>/gi;
113
+ for (const match of body.matchAll(paramRe)) {
114
+ const key = match[1];
115
+ const value = String(match[2] ?? "").trim();
116
+ if (!/^[A-Za-z_][A-Za-z0-9_.-]*$/.test(key)) {
117
+ return null;
118
+ }
119
+ args.push(`<${key}>${value}</${key}>`);
120
+ }
121
+ if (args.length === 0) {
122
+ return null;
123
+ }
124
+ return [
125
+ "<agent_response>",
126
+ " <done>false</done>",
127
+ " <message></message>",
128
+ ` <tool_call name="${name}">`,
129
+ ` <args>${args.join("")}</args>`,
130
+ " </tool_call>",
131
+ "</agent_response>",
132
+ ].join("\n");
133
+ }
134
+
66
135
  function extractEnvelope(text) {
67
136
  const cleaned = stripSingleCodeFence(text);
68
137
  const start = cleaned.indexOf("<agent_response");
69
138
  const endTag = "</agent_response>";
70
- const end = cleaned.lastIndexOf(endTag);
139
+ const end = start < 0 ? -1 : cleaned.indexOf(endTag, start);
71
140
 
72
- if (start < 0 || end < 0 || end < start) {
141
+ if (start < 0 || end < 0) {
142
+ const wrapped = wrapBareToolCall(cleaned);
143
+ if (wrapped) {
144
+ return wrapped;
145
+ }
73
146
  throw new ProtocolError(
74
147
  "Response must contain one complete <agent_response> envelope.",
75
148
  { details: { raw: cleaned } },
76
149
  );
77
150
  }
78
151
 
79
- // ChatGPT Web may prepend a preamble (e.g. "Sure, here is the response:")
80
- // or append trailing text / render rich cards around the XML. We only care
81
- // about the envelope itself, so surrounding text is stripped instead of
82
- // being treated as a protocol violation.
152
+ // Take the first complete envelope only. Web UIs (especially Kimi) often
153
+ // render the same reply twice — a code-fence copy plus the visible markdown
154
+ // — so first-open + last-close would glue two envelopes together and fail
155
+ // with "Extra text at the end". Trailing chatter after that first envelope
156
+ // is ignored the same way a preamble before it is.
83
157
  return cleaned.slice(start, end + endTag.length);
84
158
  }
85
159
 
160
+ // Web-UIs render provider chrome into the assistant text: thinking-block
161
+ // headers, code-fence language banners, and code-block action button labels.
162
+ // These tokens are UI noise, never model content, but only when they appear as
163
+ // standalone leading lines — a report may legitimately contain the word "运行"
164
+ // inside its prose, so we only strip complete noise lines at the start.
165
+ const UI_NOISE_TOKENS = new Set([
166
+ // 中文
167
+ "思考过程",
168
+ "正在思考",
169
+ "思考已完成",
170
+ "思考完成",
171
+ "跳过",
172
+ "复制",
173
+ "复制代码",
174
+ "下载",
175
+ "运行",
176
+ // English
177
+ "thinking",
178
+ "thinking process",
179
+ "reasoning",
180
+ "skip",
181
+ "copy",
182
+ "copy code",
183
+ "download",
184
+ "run",
185
+ // 日本語
186
+ "考え中",
187
+ "検討中",
188
+ "スキップ",
189
+ "コピー",
190
+ "コードをコピー",
191
+ "ダウンロード",
192
+ "実行",
193
+ // 한국어
194
+ "생각 중",
195
+ "복사",
196
+ "코드 복사",
197
+ "다운로드",
198
+ "실행",
199
+ "건너뛰기",
200
+ // Language banner on rendered code blocks (locale-independent).
201
+ "xml",
202
+ ]);
203
+
204
+ export function stripUiNoiseLines(text) {
205
+ const lines = String(text ?? "").split(/\r?\n/).map((line) => line.trim());
206
+ while (
207
+ lines.length
208
+ && (
209
+ lines[0] === ""
210
+ || UI_NOISE_TOKENS.has(lines[0])
211
+ || UI_NOISE_TOKENS.has(lines[0].toLowerCase())
212
+ )
213
+ ) {
214
+ lines.shift();
215
+ }
216
+ return lines.join("\n").trim();
217
+ }
218
+
219
+ // Returns the substantive prose that follows the first complete
220
+ // <agent_response> envelope, or null when there is none.
221
+ //
222
+ // Some models (GLM especially) put their REAL deliverable after the envelope:
223
+ // the XML carries only a short done/true stub and the full answer is rendered
224
+ // as ordinary markdown/HTML text right after it. That text is pure display
225
+ // content — it can never trigger a tool call — so the runtime may surface it
226
+ // as the final answer instead of dropping it.
227
+ //
228
+ // Defensive rule: if the trailing text contains another <agent_response (some
229
+ // UIs render the reply twice, a code-fence copy plus the visible copy), we
230
+ // cannot cleanly separate real content from the duplicated XML, so return null
231
+ // and let the caller keep the envelope's own message.
232
+ export function extractTrailingProse(rawText) {
233
+ const text = String(rawText ?? "");
234
+ const endTag = "</agent_response>";
235
+ const end = text.indexOf(endTag);
236
+ if (end < 0) {
237
+ return null;
238
+ }
239
+ const trailing = text.slice(end + endTag.length);
240
+ if (trailing.includes("<agent_response")) {
241
+ return null;
242
+ }
243
+ const withoutFenceCloser = trailing.replace(/\n*```\s*$/, "");
244
+ return stripUiNoiseLines(withoutFenceCloser) || null;
245
+ }
246
+
86
247
  function normalizeXmlValue(value) {
87
248
  if (value == null) {
88
249
  return "";
@@ -206,10 +367,13 @@ export function parseAgentResponse(rawText) {
206
367
  throw new ProtocolError("DTD and XML entities are not allowed.");
207
368
  }
208
369
 
209
- // Repair the single most common corruption first: bare ampersands the model
210
- // wrote outside CDATA. This is done before validation so an otherwise
211
- // well-formed envelope with a stray "&" parses instead of triggering a retry.
212
- const envelope = escapeBareAmpersands(rawEnvelope);
370
+ // Repair the most common, meaning-preserving corruptions before validation:
371
+ // 1. bare ampersands the model wrote outside CDATA
372
+ // 2. a missing </tool_call> immediately before </agent_response>
373
+ // GLM in particular often streams a complete <tool_call>…</args> and then
374
+ // closes the envelope without the matching </tool_call>. Inserting that one
375
+ // tag is safe: we never invent arguments, only finish an already-complete call.
376
+ const envelope = closeDanglingToolCall(escapeBareAmpersands(rawEnvelope));
213
377
 
214
378
  const validation = XMLValidator.validate(envelope);
215
379
  if (validation !== true) {
@@ -401,9 +565,27 @@ export function serializeToolResult(result, { maxBytes = Infinity } = {}) {
401
565
  }
402
566
 
403
567
  export function serializeProtocolError(error) {
568
+ // Structural XML failures need more than the raw parser message. The two
569
+ // recurring causes: (a) raw code placed directly inside the envelope — bare
570
+ // < or & outside CDATA is illegal XML (the parser reports things like
571
+ // "Invalid space after '<'"); (b) a reply cut off before </agent_response>,
572
+ // so nothing in it was executed. Tell the model exactly how to fix both.
573
+ const message = String(error?.message ?? "");
574
+ const structural = /complete <agent_response> envelope|closing tag|invalid xml|space after '<'/i.test(message);
575
+ const guidance = structural
576
+ ? cdata(
577
+ "Wrap ALL code and file content in CDATA (<![CDATA[...]]>) inside "
578
+ + "<content>, <new_text>, or <message>. Never put raw code directly "
579
+ + "inside the envelope: XML forbids a bare < or & in text. Split large "
580
+ + "changes into SMALL fs.edit calls (each new_text at most ~30 lines) "
581
+ + "and close the envelope with </agent_response> immediately after "
582
+ + "the last tool_call. Never leave the envelope open.",
583
+ )
584
+ : "";
404
585
  return [
405
586
  "<protocol_error>",
406
- cdata(error.message),
587
+ cdata(message),
588
+ guidance,
407
589
  "</protocol_error>",
408
590
  ].join("");
409
591
  }