wtagent 0.3.0 → 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. package/README.md +163 -9
  2. package/package.json +9 -8
  3. package/src/artifacts/artifact-store.js +154 -0
  4. package/src/audio/music-artifact.js +66 -0
  5. package/src/audio/native-music-receiver.js +91 -0
  6. package/src/browser/base-web-adapter.js +2418 -201
  7. package/src/browser/cdp-browser.js +223 -25
  8. package/src/browser/cdp-state.js +8 -5
  9. package/src/browser/chatgpt-dom.js +179 -0
  10. package/src/browser/chatgpt-web-adapter.js +831 -69
  11. package/src/browser/claude-web-adapter.js +13 -0
  12. package/src/browser/fake-web-model-adapter.js +152 -3
  13. package/src/browser/gemini-web-adapter.js +71 -20
  14. package/src/browser/glm-web-adapter.js +2 -2
  15. package/src/browser/grok-web-adapter.js +35 -3
  16. package/src/browser/rendered-text.js +11 -0
  17. package/src/cli/i18n.js +50 -20
  18. package/src/cli/main.js +124 -40
  19. package/src/cli/prompt-input.js +68 -0
  20. package/src/image/browser-image-driver.js +238 -0
  21. package/src/image/browser-image-session-pool.js +96 -0
  22. package/src/image/image-generation-service.js +101 -0
  23. package/src/image/image-provider-driver.js +22 -0
  24. package/src/image/native-image-receiver.js +66 -0
  25. package/src/image/provider-registry.js +23 -0
  26. package/src/image/providers/chatgpt-image-driver.js +211 -0
  27. package/src/image/providers/gemini-image-driver.js +100 -0
  28. package/src/image/providers/grok-image-driver.js +193 -0
  29. package/src/platform/command-launcher.js +1 -1
  30. package/src/platform/paths.js +19 -0
  31. package/src/platform/windows-diagnostics.js +9 -16
  32. package/src/policy/path-guard.js +5 -1
  33. package/src/policy/policy-engine.js +10 -5
  34. package/src/protocol/envelope.js +74 -0
  35. package/src/protocol/markers.js +16 -0
  36. package/src/protocol/prompt-builder.js +12 -9
  37. package/src/protocol/xml-protocol.js +36 -18
  38. package/src/runtime/agent-runtime.js +1383 -238
  39. package/src/session/agent-session.js +1047 -121
  40. package/src/tools/default-tools.js +1 -1
  41. package/src/tools/image-tools.js +92 -0
  42. package/src/tools/registry.js +13 -1
  43. package/docs/technical-design.md +0 -866
@@ -0,0 +1,74 @@
1
+ // Locate transport tags without mistaking quoted attributes, comments, or
2
+ // CDATA payloads for markup. This is a boundary scanner, not an XML validator.
3
+ // Malformed markup is still passed to the strict protocol parser.
4
+ export function* xmlTags(value) {
5
+ const text = String(value ?? "");
6
+ let index = 0;
7
+ while ((index = text.indexOf("<", index)) >= 0) {
8
+ const start = index;
9
+ const special = text.startsWith("<![CDATA[", index)
10
+ ? [9, "]]>"]
11
+ : text.startsWith("<!--", index)
12
+ ? [4, "-->"]
13
+ : text.startsWith("<?", index) ? [2, "?>"] : null;
14
+ if (special) {
15
+ const end = text.indexOf(special[1], index + special[0]);
16
+ if (end < 0) return;
17
+ index = end + special[1].length;
18
+ continue;
19
+ }
20
+ const match = /^<(\/)?([A-Za-z_][\w:.-]*)(?=[\s/>])/.exec(text.slice(index));
21
+ if (!match) {
22
+ index += 1;
23
+ continue;
24
+ }
25
+ index += match[0].length;
26
+ let quote = null;
27
+ for (; index < text.length; index += 1) {
28
+ const character = text[index];
29
+ if (quote) {
30
+ if (character === quote) quote = null;
31
+ } else if (character === '"' || character === "'") {
32
+ quote = character;
33
+ } else if (character === "<" || character === ">") {
34
+ break;
35
+ }
36
+ }
37
+ // Keep an unfinished opening tag visible to the boundary finder; otherwise
38
+ // a tool inside a malformed outer root could be mistaken for a bare call.
39
+ if (index >= text.length || text[index] === "<") {
40
+ yield { name: match[2], closing: Boolean(match[1]), selfClosing: false, start, end: null };
41
+ if (index >= text.length) return;
42
+ continue;
43
+ }
44
+ index += 1;
45
+ yield {
46
+ name: match[2],
47
+ closing: Boolean(match[1]),
48
+ selfClosing: /\/\s*>$/.test(text.slice(start, index)),
49
+ start,
50
+ end: index,
51
+ };
52
+ }
53
+ }
54
+
55
+ export function findXmlElement(text, name = "agent_response") {
56
+ let start = null;
57
+ let depth = 0;
58
+ for (const tag of xmlTags(text)) {
59
+ if (tag.name !== name) continue;
60
+ if (start == null) {
61
+ if (tag.closing) continue;
62
+ start = tag.start;
63
+ }
64
+ if (tag.end == null) return { start, end: null };
65
+ if (tag.closing) depth -= 1;
66
+ else if (!tag.selfClosing) depth += 1;
67
+ if (depth === 0) return { start, end: tag.end };
68
+ }
69
+ return start == null ? null : { start, end: null };
70
+ }
71
+
72
+ export function hasCompleteAgentEnvelope(text) {
73
+ return findXmlElement(text)?.end != null;
74
+ }
@@ -21,6 +21,22 @@ export const DEFAULT_SYSTEM_REMINDER = [
21
21
  "Write large code or file payloads as CDATA (<![CDATA[...]]>) inside <content>, <new_text>, or <message> — never raw code directly inside the envelope — and never include the sequence ]]> inside CDATA content.",
22
22
  ].join(" ");
23
23
 
24
+ // A native image is an assistant response, never a local XML tool call.
25
+ export const NATIVE_IMAGE_CONTRACT = [
26
+ "When the task needs generated artwork, generate and output the image directly using your native image-generation capability.",
27
+ "Image responses are exempt from the XML format: output the image itself, without an XML wrapper or drawing code.",
28
+ "The Runtime automatically downloads images from your completed reply and returns verified local paths; then continue the user's task using those paths.",
29
+ "There is no local image-generation tool. This replaces any image-tool instructions from earlier in this conversation.",
30
+ "If generation returns only text, that text will be returned as the provider result so you can explain the limitation or ask the user; do not claim an image was saved.",
31
+ ].join(" ");
32
+
33
+ export const NATIVE_IMAGE_SYSTEM_REMINDER = [
34
+ NATIVE_IMAGE_CONTRACT,
35
+ "For text answers and local operations, use exactly one complete <agent_response> XML envelope in one xml code fence.",
36
+ "Use done=true for a deliverable answer or a question that requires user input; use done=false with one local tool request when needed.",
37
+ "Use CDATA for code or long payloads. Local tools execute only after the Runtime returns their results.",
38
+ ].join(" ");
39
+
24
40
  export function wrapSystemPrompt(text) {
25
41
  return `<${SYSTEM_PROMPT_TAG}>\n${String(text ?? "")}\n</${SYSTEM_PROMPT_TAG}>`;
26
42
  }
@@ -1,4 +1,4 @@
1
- import { wrapSystemPrompt } from "./markers.js";
1
+ import { NATIVE_IMAGE_CONTRACT, wrapSystemPrompt } from "./markers.js";
2
2
 
3
3
  function formatTool(tool) {
4
4
  // Flat, so multi-line descriptions (arrays) keep their line breaks instead
@@ -19,19 +19,19 @@ const DONE_SEMANTICS = `Current-run completion semantics:
19
19
  - Set done=true only when the current user request has a complete, deliverable answer, or when you have a specific question that must be answered by the user before useful work can continue.
20
20
  - For informational or conversational tasks (e.g. answering a question, summarizing text, brainstorming), you can reply with done=true and your answer directly — no tool call is required.
21
21
  - For tasks that require reading, creating, or modifying files on the user's machine, use the local tools and verify the result before done=true.
22
- - The Runtime validates completion against successful local tool evidence. Never claim that you created, changed, read, tested, or verified local state unless the corresponding tool results were returned in this run.
22
+ - Base claims about creating, changing, reading, testing, or verifying local state on tool results returned in this run. The Runtime validates the protocol and tracks tool execution; you are responsible for assessing whether the user's request is satisfied.
23
23
  - Use done=false only when you are about to call a local tool and need its result before you can continue.
24
24
  - Tool count, elapsed turns, or lack of an immediately obvious next action never proves completion.`;
25
25
 
26
26
  // The protocol + tool catalog are WTAgent-specific transport scaffolding.
27
27
  // They are wrapped for the web message and never persisted into the portable
28
28
  // Codex rollout.
29
- function buildBootstrapScaffold({ projectRoot, tools }) {
29
+ function buildBootstrapScaffold({ projectRoot, tools, nativeImages }) {
30
30
  const toolDocs = tools.map(formatTool).join("\n\n");
31
31
 
32
32
  return `The user is running WTAgent, a local application that uses this web AI conversation for reasoning. The following is the user's requested application-level response format and collaboration contract; it is not a claim that this web chat has native filesystem or function-call tools.
33
33
 
34
- You do not need direct filesystem access or provider-native tool buttons. Return local operation requests as XML text. After your reply is complete, the user's local Node.js Runtime will parse the XML, validate the arguments, apply local policy, and may execute the requested operation. Its result will arrive in the next user message as <tool_result>. XML by itself never guarantees execution.
34
+ You do not need direct filesystem access for local operations. Return local operation requests as XML text. After your reply is complete, the user's local Node.js Runtime will parse the XML, validate the arguments, apply local policy, and may execute the requested operation. Its result will arrive in the next user message as <tool_result>. XML by itself never guarantees execution.
35
35
 
36
36
  You are not limited to coding tasks. You can answer questions, write text, brainstorm, analyze, summarize, and — when the task requires it — request that the user's Runtime read, create, or modify files or run commands.
37
37
 
@@ -40,8 +40,8 @@ The project filesystem described below is a logical, virtual filesystem namespac
40
40
 
41
41
  Do not inspect /workspace, /mnt/data, or any ambient, cloud, or sandbox filesystem. Those locations are unrelated to the user's project. Request all project reads, listings, writes, edits, and commands only through the XML operations declared below.
42
42
 
43
- ## Output protocol
44
- Every reply must contain exactly one complete XML root node inside a single \`xml\` code fence, with no text outside the fence. The code fence guarantees that JavaScript backticks and other source characters are not swallowed by the web Markdown renderer:
43
+ ${nativeImages ? `## Native images\n${NATIVE_IMAGE_CONTRACT}\n\n` : ""}## Output protocol
44
+ ${nativeImages ? "Every text reply" : "Every reply"} must contain exactly one complete XML root node inside a single \`xml\` code fence, with no text outside the fence. The code fence guarantees that JavaScript backticks and other source characters are not swallowed by the web Markdown renderer:
45
45
 
46
46
  \`\`\`xml
47
47
  <agent_response>
@@ -106,18 +106,19 @@ export function buildBootstrapPrompt({
106
106
  task,
107
107
  projectRoot,
108
108
  tools,
109
+ nativeImages = false,
109
110
  }) {
110
- const developer = buildBootstrapScaffold({ projectRoot, tools });
111
+ const developer = buildBootstrapScaffold({ projectRoot, tools, nativeImages });
111
112
  const web = `${wrapSystemPrompt(developer)}\n\n## User task\n${task}`;
112
113
  return { web, developer, user: task };
113
114
  }
114
115
 
115
- function buildResumeScaffold({ tools, followUpRule, state, nextInstruction }) {
116
+ function buildResumeScaffold({ tools, followUpRule, state, nextInstruction, nativeImages }) {
116
117
  const toolDocs = tools.map(formatTool).join("\n\n");
117
118
 
118
119
  return `Continue the same WTAgent session using the user's requested XML application protocol. You do not need native tool access: write a <tool_call> request as text, and the user's local Runtime will validate it, may execute it, and will return <tool_result> in the next user message.
119
120
 
120
- Still place your single <agent_response> XML inside one \`xml\` code fence with no text outside it and call at most one tool per turn. This preserves JavaScript backticks inside file contents.
121
+ ${nativeImages ? `${NATIVE_IMAGE_CONTRACT}\n\nFor text replies: ` : ""}Still place your single <agent_response> XML inside one \`xml\` code fence with no text outside it and call at most one tool per turn. This preserves JavaScript backticks inside file contents.
121
122
 
122
123
  ${DONE_SEMANTICS}
123
124
 
@@ -138,6 +139,7 @@ export function buildResumePrompt({
138
139
  instruction,
139
140
  state,
140
141
  tools,
142
+ nativeImages = false,
141
143
  }) {
142
144
  const nextInstruction = instruction?.trim()
143
145
  || "Continue the interrupted run based on the current project state and return a deliverable result.";
@@ -150,6 +152,7 @@ export function buildResumePrompt({
150
152
  followUpRule,
151
153
  state,
152
154
  nextInstruction,
155
+ nativeImages,
153
156
  });
154
157
  return {
155
158
  web: wrapSystemPrompt(developer),
@@ -1,5 +1,6 @@
1
1
  import { XMLParser, XMLValidator } from "fast-xml-parser";
2
2
  import { ProtocolError } from "../shared/errors.js";
3
+ import { findXmlElement, xmlTags } from "./envelope.js";
3
4
  import {
4
5
  truncateUtf8HeadTail,
5
6
  utf8ByteLength,
@@ -64,8 +65,9 @@ function escapeBareAmpersands(text) {
64
65
  }
65
66
 
66
67
  function closeDanglingToolCall(envelope) {
67
- const open = (envelope.match(/<tool_call(?:\s|>)/gi) ?? []).length;
68
- const close = (envelope.match(/<\/tool_call>/gi) ?? []).length;
68
+ const tags = [...xmlTags(envelope)].filter((tag) => tag.name === "tool_call");
69
+ const open = tags.filter((tag) => !tag.closing && !tag.selfClosing).length;
70
+ const close = tags.filter((tag) => tag.closing).length;
69
71
  if (open !== close + 1) {
70
72
  return envelope;
71
73
  }
@@ -76,11 +78,9 @@ function closeDanglingToolCall(envelope) {
76
78
  }
77
79
 
78
80
  function wrapBareToolCall(text) {
79
- const start = text.search(/<tool_call(?:\s|>)/i);
80
- const endTag = "</tool_call>";
81
- const end = start < 0 ? -1 : text.indexOf(endTag, start);
82
- if (start >= 0 && end >= start) {
83
- const toolCall = text.slice(start, end + endTag.length);
81
+ const bounds = findXmlElement(text, "tool_call");
82
+ if (bounds?.end != null) {
83
+ const toolCall = text.slice(bounds.start, bounds.end);
84
84
  if (/<args[\s>/]/i.test(toolCall)) {
85
85
  return [
86
86
  "<agent_response>",
@@ -134,12 +134,11 @@ function wrapClaudeStyleInvoke(text) {
134
134
 
135
135
  function extractEnvelope(text) {
136
136
  const cleaned = stripSingleCodeFence(text);
137
- const start = cleaned.indexOf("<agent_response");
138
- const endTag = "</agent_response>";
139
- const end = start < 0 ? -1 : cleaned.indexOf(endTag, start);
137
+ const bounds = findXmlElement(cleaned);
140
138
 
141
- if (start < 0 || end < 0) {
142
- const wrapped = wrapBareToolCall(cleaned);
139
+ if (bounds?.end == null) {
140
+ // Never salvage an operation out of an unfinished outer response.
141
+ const wrapped = bounds == null ? wrapBareToolCall(cleaned) : null;
143
142
  if (wrapped) {
144
143
  return wrapped;
145
144
  }
@@ -154,7 +153,7 @@ function extractEnvelope(text) {
154
153
  // — so first-open + last-close would glue two envelopes together and fail
155
154
  // with "Extra text at the end". Trailing chatter after that first envelope
156
155
  // is ignored the same way a preamble before it is.
157
- return cleaned.slice(start, end + endTag.length);
156
+ return cleaned.slice(bounds.start, bounds.end);
158
157
  }
159
158
 
160
159
  // Web-UIs render provider chrome into the assistant text: thinking-block
@@ -231,12 +230,11 @@ export function stripUiNoiseLines(text) {
231
230
  // and let the caller keep the envelope's own message.
232
231
  export function extractTrailingProse(rawText) {
233
232
  const text = String(rawText ?? "");
234
- const endTag = "</agent_response>";
235
- const end = text.indexOf(endTag);
236
- if (end < 0) {
233
+ const bounds = findXmlElement(text);
234
+ if (bounds?.end == null) {
237
235
  return null;
238
236
  }
239
- const trailing = text.slice(end + endTag.length);
237
+ const trailing = text.slice(bounds.end);
240
238
  if (trailing.includes("<agent_response")) {
241
239
  return null;
242
240
  }
@@ -332,9 +330,21 @@ function looseTagText(xml, tag) {
332
330
  // guessing the arguments of a side-effecting operation from broken XML is
333
331
  // unsafe, so those still fall through to a format retry.
334
332
  function recoverMessageOnlyResponse(envelope) {
335
- if (/<tool_call[\s>]/i.test(envelope)) {
333
+ const tags = [...xmlTags(envelope)];
334
+ if (
335
+ tags.some((tag) => /^(tool_call|tool_calls|invoke|action|parameter)$/.test(tag.name))
336
+ || tags.filter((tag) => tag.name === "agent_response" && !tag.closing).length !== 1
337
+ ) {
336
338
  return null;
337
339
  }
340
+ // Tolerate malformed display markup only inside <message>. Unknown sibling
341
+ // fields must not disappear when a damaged operation is mistaken for prose.
342
+ const root = tags.find((tag) => tag.name === "agent_response" && !tag.closing);
343
+ const closing = tags.findLast((tag) => tag.name === "agent_response" && tag.closing);
344
+ const siblings = envelope.slice(root.end, closing.start)
345
+ .replace(/<done(?:\s[^>]*)?>[\s\S]*?<\/done>/g, "")
346
+ .replace(/<message(?:\s[^>]*)?>[\s\S]*?<\/message>/g, "");
347
+ if (siblings.trim()) return null;
338
348
  const doneText = looseTagText(envelope, "done");
339
349
  const message = looseTagText(envelope, "message");
340
350
  if (doneText == null || message == null) {
@@ -366,6 +376,11 @@ export function parseAgentResponse(rawText) {
366
376
  if (/<!DOCTYPE|<!ENTITY/i.test(structuralXml)) {
367
377
  throw new ProtocolError("DTD and XML entities are not allowed.");
368
378
  }
379
+ if ([...xmlTags(rawEnvelope)].filter((tag) => (
380
+ tag.name === "agent_response" && !tag.closing
381
+ )).length !== 1) {
382
+ throw new ProtocolError("Nested <agent_response> envelopes are not allowed.");
383
+ }
369
384
 
370
385
  // Repair the most common, meaning-preserving corruptions before validation:
371
386
  // 1. bare ampersands the model wrote outside CDATA
@@ -402,6 +417,9 @@ export function parseAgentResponse(rawText) {
402
417
  if (!response || typeof response !== "object" || Array.isArray(response)) {
403
418
  throw new ProtocolError("Missing <agent_response> root.");
404
419
  }
420
+ if (["action", "invoke", "tool_calls"].some((name) => response[name] != null)) {
421
+ throw new ProtocolError("Operations must use one <tool_call> element with <args>.");
422
+ }
405
423
 
406
424
  const done = parseDone(response.done);
407
425
  const message = scalarText(response.message);