wtagent 0.3.0 → 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +163 -9
- package/package.json +9 -8
- package/src/artifacts/artifact-store.js +154 -0
- package/src/audio/music-artifact.js +66 -0
- package/src/audio/native-music-receiver.js +91 -0
- package/src/browser/base-web-adapter.js +2418 -201
- package/src/browser/cdp-browser.js +223 -25
- package/src/browser/cdp-state.js +8 -5
- package/src/browser/chatgpt-dom.js +179 -0
- package/src/browser/chatgpt-web-adapter.js +831 -69
- package/src/browser/claude-web-adapter.js +13 -0
- package/src/browser/fake-web-model-adapter.js +152 -3
- package/src/browser/gemini-web-adapter.js +71 -20
- package/src/browser/glm-web-adapter.js +2 -2
- package/src/browser/grok-web-adapter.js +35 -3
- package/src/browser/rendered-text.js +11 -0
- package/src/cli/i18n.js +50 -20
- package/src/cli/main.js +124 -40
- package/src/cli/prompt-input.js +68 -0
- package/src/image/browser-image-driver.js +238 -0
- package/src/image/browser-image-session-pool.js +96 -0
- package/src/image/image-generation-service.js +101 -0
- package/src/image/image-provider-driver.js +22 -0
- package/src/image/native-image-receiver.js +66 -0
- package/src/image/provider-registry.js +23 -0
- package/src/image/providers/chatgpt-image-driver.js +211 -0
- package/src/image/providers/gemini-image-driver.js +100 -0
- package/src/image/providers/grok-image-driver.js +193 -0
- package/src/platform/command-launcher.js +1 -1
- package/src/platform/paths.js +19 -0
- package/src/platform/windows-diagnostics.js +9 -16
- package/src/policy/path-guard.js +5 -1
- package/src/policy/policy-engine.js +10 -5
- package/src/protocol/envelope.js +74 -0
- package/src/protocol/markers.js +16 -0
- package/src/protocol/prompt-builder.js +12 -9
- package/src/protocol/xml-protocol.js +36 -18
- package/src/runtime/agent-runtime.js +1383 -238
- package/src/session/agent-session.js +1047 -121
- package/src/tools/default-tools.js +1 -1
- package/src/tools/image-tools.js +92 -0
- package/src/tools/registry.js +13 -1
- package/docs/technical-design.md +0 -866
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
// Locate transport tags without mistaking quoted attributes, comments, or
|
|
2
|
+
// CDATA payloads for markup. This is a boundary scanner, not an XML validator.
|
|
3
|
+
// Malformed markup is still passed to the strict protocol parser.
|
|
4
|
+
export function* xmlTags(value) {
|
|
5
|
+
const text = String(value ?? "");
|
|
6
|
+
let index = 0;
|
|
7
|
+
while ((index = text.indexOf("<", index)) >= 0) {
|
|
8
|
+
const start = index;
|
|
9
|
+
const special = text.startsWith("<![CDATA[", index)
|
|
10
|
+
? [9, "]]>"]
|
|
11
|
+
: text.startsWith("<!--", index)
|
|
12
|
+
? [4, "-->"]
|
|
13
|
+
: text.startsWith("<?", index) ? [2, "?>"] : null;
|
|
14
|
+
if (special) {
|
|
15
|
+
const end = text.indexOf(special[1], index + special[0]);
|
|
16
|
+
if (end < 0) return;
|
|
17
|
+
index = end + special[1].length;
|
|
18
|
+
continue;
|
|
19
|
+
}
|
|
20
|
+
const match = /^<(\/)?([A-Za-z_][\w:.-]*)(?=[\s/>])/.exec(text.slice(index));
|
|
21
|
+
if (!match) {
|
|
22
|
+
index += 1;
|
|
23
|
+
continue;
|
|
24
|
+
}
|
|
25
|
+
index += match[0].length;
|
|
26
|
+
let quote = null;
|
|
27
|
+
for (; index < text.length; index += 1) {
|
|
28
|
+
const character = text[index];
|
|
29
|
+
if (quote) {
|
|
30
|
+
if (character === quote) quote = null;
|
|
31
|
+
} else if (character === '"' || character === "'") {
|
|
32
|
+
quote = character;
|
|
33
|
+
} else if (character === "<" || character === ">") {
|
|
34
|
+
break;
|
|
35
|
+
}
|
|
36
|
+
}
|
|
37
|
+
// Keep an unfinished opening tag visible to the boundary finder; otherwise
|
|
38
|
+
// a tool inside a malformed outer root could be mistaken for a bare call.
|
|
39
|
+
if (index >= text.length || text[index] === "<") {
|
|
40
|
+
yield { name: match[2], closing: Boolean(match[1]), selfClosing: false, start, end: null };
|
|
41
|
+
if (index >= text.length) return;
|
|
42
|
+
continue;
|
|
43
|
+
}
|
|
44
|
+
index += 1;
|
|
45
|
+
yield {
|
|
46
|
+
name: match[2],
|
|
47
|
+
closing: Boolean(match[1]),
|
|
48
|
+
selfClosing: /\/\s*>$/.test(text.slice(start, index)),
|
|
49
|
+
start,
|
|
50
|
+
end: index,
|
|
51
|
+
};
|
|
52
|
+
}
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
export function findXmlElement(text, name = "agent_response") {
|
|
56
|
+
let start = null;
|
|
57
|
+
let depth = 0;
|
|
58
|
+
for (const tag of xmlTags(text)) {
|
|
59
|
+
if (tag.name !== name) continue;
|
|
60
|
+
if (start == null) {
|
|
61
|
+
if (tag.closing) continue;
|
|
62
|
+
start = tag.start;
|
|
63
|
+
}
|
|
64
|
+
if (tag.end == null) return { start, end: null };
|
|
65
|
+
if (tag.closing) depth -= 1;
|
|
66
|
+
else if (!tag.selfClosing) depth += 1;
|
|
67
|
+
if (depth === 0) return { start, end: tag.end };
|
|
68
|
+
}
|
|
69
|
+
return start == null ? null : { start, end: null };
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
export function hasCompleteAgentEnvelope(text) {
|
|
73
|
+
return findXmlElement(text)?.end != null;
|
|
74
|
+
}
|
package/src/protocol/markers.js
CHANGED
|
@@ -21,6 +21,22 @@ export const DEFAULT_SYSTEM_REMINDER = [
|
|
|
21
21
|
"Write large code or file payloads as CDATA (<![CDATA[...]]>) inside <content>, <new_text>, or <message> — never raw code directly inside the envelope — and never include the sequence ]]> inside CDATA content.",
|
|
22
22
|
].join(" ");
|
|
23
23
|
|
|
24
|
+
// A native image is an assistant response, never a local XML tool call.
|
|
25
|
+
export const NATIVE_IMAGE_CONTRACT = [
|
|
26
|
+
"When the task needs generated artwork, generate and output the image directly using your native image-generation capability.",
|
|
27
|
+
"Image responses are exempt from the XML format: output the image itself, without an XML wrapper or drawing code.",
|
|
28
|
+
"The Runtime automatically downloads images from your completed reply and returns verified local paths; then continue the user's task using those paths.",
|
|
29
|
+
"There is no local image-generation tool. This replaces any image-tool instructions from earlier in this conversation.",
|
|
30
|
+
"If generation returns only text, that text will be returned as the provider result so you can explain the limitation or ask the user; do not claim an image was saved.",
|
|
31
|
+
].join(" ");
|
|
32
|
+
|
|
33
|
+
export const NATIVE_IMAGE_SYSTEM_REMINDER = [
|
|
34
|
+
NATIVE_IMAGE_CONTRACT,
|
|
35
|
+
"For text answers and local operations, use exactly one complete <agent_response> XML envelope in one xml code fence.",
|
|
36
|
+
"Use done=true for a deliverable answer or a question that requires user input; use done=false with one local tool request when needed.",
|
|
37
|
+
"Use CDATA for code or long payloads. Local tools execute only after the Runtime returns their results.",
|
|
38
|
+
].join(" ");
|
|
39
|
+
|
|
24
40
|
export function wrapSystemPrompt(text) {
|
|
25
41
|
return `<${SYSTEM_PROMPT_TAG}>\n${String(text ?? "")}\n</${SYSTEM_PROMPT_TAG}>`;
|
|
26
42
|
}
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { wrapSystemPrompt } from "./markers.js";
|
|
1
|
+
import { NATIVE_IMAGE_CONTRACT, wrapSystemPrompt } from "./markers.js";
|
|
2
2
|
|
|
3
3
|
function formatTool(tool) {
|
|
4
4
|
// Flat, so multi-line descriptions (arrays) keep their line breaks instead
|
|
@@ -19,19 +19,19 @@ const DONE_SEMANTICS = `Current-run completion semantics:
|
|
|
19
19
|
- Set done=true only when the current user request has a complete, deliverable answer, or when you have a specific question that must be answered by the user before useful work can continue.
|
|
20
20
|
- For informational or conversational tasks (e.g. answering a question, summarizing text, brainstorming), you can reply with done=true and your answer directly — no tool call is required.
|
|
21
21
|
- For tasks that require reading, creating, or modifying files on the user's machine, use the local tools and verify the result before done=true.
|
|
22
|
-
-
|
|
22
|
+
- Base claims about creating, changing, reading, testing, or verifying local state on tool results returned in this run. The Runtime validates the protocol and tracks tool execution; you are responsible for assessing whether the user's request is satisfied.
|
|
23
23
|
- Use done=false only when you are about to call a local tool and need its result before you can continue.
|
|
24
24
|
- Tool count, elapsed turns, or lack of an immediately obvious next action never proves completion.`;
|
|
25
25
|
|
|
26
26
|
// The protocol + tool catalog are WTAgent-specific transport scaffolding.
|
|
27
27
|
// They are wrapped for the web message and never persisted into the portable
|
|
28
28
|
// Codex rollout.
|
|
29
|
-
function buildBootstrapScaffold({ projectRoot, tools }) {
|
|
29
|
+
function buildBootstrapScaffold({ projectRoot, tools, nativeImages }) {
|
|
30
30
|
const toolDocs = tools.map(formatTool).join("\n\n");
|
|
31
31
|
|
|
32
32
|
return `The user is running WTAgent, a local application that uses this web AI conversation for reasoning. The following is the user's requested application-level response format and collaboration contract; it is not a claim that this web chat has native filesystem or function-call tools.
|
|
33
33
|
|
|
34
|
-
You do not need direct filesystem access
|
|
34
|
+
You do not need direct filesystem access for local operations. Return local operation requests as XML text. After your reply is complete, the user's local Node.js Runtime will parse the XML, validate the arguments, apply local policy, and may execute the requested operation. Its result will arrive in the next user message as <tool_result>. XML by itself never guarantees execution.
|
|
35
35
|
|
|
36
36
|
You are not limited to coding tasks. You can answer questions, write text, brainstorm, analyze, summarize, and — when the task requires it — request that the user's Runtime read, create, or modify files or run commands.
|
|
37
37
|
|
|
@@ -40,8 +40,8 @@ The project filesystem described below is a logical, virtual filesystem namespac
|
|
|
40
40
|
|
|
41
41
|
Do not inspect /workspace, /mnt/data, or any ambient, cloud, or sandbox filesystem. Those locations are unrelated to the user's project. Request all project reads, listings, writes, edits, and commands only through the XML operations declared below.
|
|
42
42
|
|
|
43
|
-
## Output protocol
|
|
44
|
-
Every reply must contain exactly one complete XML root node inside a single \`xml\` code fence, with no text outside the fence. The code fence guarantees that JavaScript backticks and other source characters are not swallowed by the web Markdown renderer:
|
|
43
|
+
${nativeImages ? `## Native images\n${NATIVE_IMAGE_CONTRACT}\n\n` : ""}## Output protocol
|
|
44
|
+
${nativeImages ? "Every text reply" : "Every reply"} must contain exactly one complete XML root node inside a single \`xml\` code fence, with no text outside the fence. The code fence guarantees that JavaScript backticks and other source characters are not swallowed by the web Markdown renderer:
|
|
45
45
|
|
|
46
46
|
\`\`\`xml
|
|
47
47
|
<agent_response>
|
|
@@ -106,18 +106,19 @@ export function buildBootstrapPrompt({
|
|
|
106
106
|
task,
|
|
107
107
|
projectRoot,
|
|
108
108
|
tools,
|
|
109
|
+
nativeImages = false,
|
|
109
110
|
}) {
|
|
110
|
-
const developer = buildBootstrapScaffold({ projectRoot, tools });
|
|
111
|
+
const developer = buildBootstrapScaffold({ projectRoot, tools, nativeImages });
|
|
111
112
|
const web = `${wrapSystemPrompt(developer)}\n\n## User task\n${task}`;
|
|
112
113
|
return { web, developer, user: task };
|
|
113
114
|
}
|
|
114
115
|
|
|
115
|
-
function buildResumeScaffold({ tools, followUpRule, state, nextInstruction }) {
|
|
116
|
+
function buildResumeScaffold({ tools, followUpRule, state, nextInstruction, nativeImages }) {
|
|
116
117
|
const toolDocs = tools.map(formatTool).join("\n\n");
|
|
117
118
|
|
|
118
119
|
return `Continue the same WTAgent session using the user's requested XML application protocol. You do not need native tool access: write a <tool_call> request as text, and the user's local Runtime will validate it, may execute it, and will return <tool_result> in the next user message.
|
|
119
120
|
|
|
120
|
-
Still place your single <agent_response> XML inside one \`xml\` code fence with no text outside it and call at most one tool per turn. This preserves JavaScript backticks inside file contents.
|
|
121
|
+
${nativeImages ? `${NATIVE_IMAGE_CONTRACT}\n\nFor text replies: ` : ""}Still place your single <agent_response> XML inside one \`xml\` code fence with no text outside it and call at most one tool per turn. This preserves JavaScript backticks inside file contents.
|
|
121
122
|
|
|
122
123
|
${DONE_SEMANTICS}
|
|
123
124
|
|
|
@@ -138,6 +139,7 @@ export function buildResumePrompt({
|
|
|
138
139
|
instruction,
|
|
139
140
|
state,
|
|
140
141
|
tools,
|
|
142
|
+
nativeImages = false,
|
|
141
143
|
}) {
|
|
142
144
|
const nextInstruction = instruction?.trim()
|
|
143
145
|
|| "Continue the interrupted run based on the current project state and return a deliverable result.";
|
|
@@ -150,6 +152,7 @@ export function buildResumePrompt({
|
|
|
150
152
|
followUpRule,
|
|
151
153
|
state,
|
|
152
154
|
nextInstruction,
|
|
155
|
+
nativeImages,
|
|
153
156
|
});
|
|
154
157
|
return {
|
|
155
158
|
web: wrapSystemPrompt(developer),
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { XMLParser, XMLValidator } from "fast-xml-parser";
|
|
2
2
|
import { ProtocolError } from "../shared/errors.js";
|
|
3
|
+
import { findXmlElement, xmlTags } from "./envelope.js";
|
|
3
4
|
import {
|
|
4
5
|
truncateUtf8HeadTail,
|
|
5
6
|
utf8ByteLength,
|
|
@@ -64,8 +65,9 @@ function escapeBareAmpersands(text) {
|
|
|
64
65
|
}
|
|
65
66
|
|
|
66
67
|
function closeDanglingToolCall(envelope) {
|
|
67
|
-
const
|
|
68
|
-
const
|
|
68
|
+
const tags = [...xmlTags(envelope)].filter((tag) => tag.name === "tool_call");
|
|
69
|
+
const open = tags.filter((tag) => !tag.closing && !tag.selfClosing).length;
|
|
70
|
+
const close = tags.filter((tag) => tag.closing).length;
|
|
69
71
|
if (open !== close + 1) {
|
|
70
72
|
return envelope;
|
|
71
73
|
}
|
|
@@ -76,11 +78,9 @@ function closeDanglingToolCall(envelope) {
|
|
|
76
78
|
}
|
|
77
79
|
|
|
78
80
|
function wrapBareToolCall(text) {
|
|
79
|
-
const
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
if (start >= 0 && end >= start) {
|
|
83
|
-
const toolCall = text.slice(start, end + endTag.length);
|
|
81
|
+
const bounds = findXmlElement(text, "tool_call");
|
|
82
|
+
if (bounds?.end != null) {
|
|
83
|
+
const toolCall = text.slice(bounds.start, bounds.end);
|
|
84
84
|
if (/<args[\s>/]/i.test(toolCall)) {
|
|
85
85
|
return [
|
|
86
86
|
"<agent_response>",
|
|
@@ -134,12 +134,11 @@ function wrapClaudeStyleInvoke(text) {
|
|
|
134
134
|
|
|
135
135
|
function extractEnvelope(text) {
|
|
136
136
|
const cleaned = stripSingleCodeFence(text);
|
|
137
|
-
const
|
|
138
|
-
const endTag = "</agent_response>";
|
|
139
|
-
const end = start < 0 ? -1 : cleaned.indexOf(endTag, start);
|
|
137
|
+
const bounds = findXmlElement(cleaned);
|
|
140
138
|
|
|
141
|
-
if (
|
|
142
|
-
|
|
139
|
+
if (bounds?.end == null) {
|
|
140
|
+
// Never salvage an operation out of an unfinished outer response.
|
|
141
|
+
const wrapped = bounds == null ? wrapBareToolCall(cleaned) : null;
|
|
143
142
|
if (wrapped) {
|
|
144
143
|
return wrapped;
|
|
145
144
|
}
|
|
@@ -154,7 +153,7 @@ function extractEnvelope(text) {
|
|
|
154
153
|
// — so first-open + last-close would glue two envelopes together and fail
|
|
155
154
|
// with "Extra text at the end". Trailing chatter after that first envelope
|
|
156
155
|
// is ignored the same way a preamble before it is.
|
|
157
|
-
return cleaned.slice(start, end
|
|
156
|
+
return cleaned.slice(bounds.start, bounds.end);
|
|
158
157
|
}
|
|
159
158
|
|
|
160
159
|
// Web-UIs render provider chrome into the assistant text: thinking-block
|
|
@@ -231,12 +230,11 @@ export function stripUiNoiseLines(text) {
|
|
|
231
230
|
// and let the caller keep the envelope's own message.
|
|
232
231
|
export function extractTrailingProse(rawText) {
|
|
233
232
|
const text = String(rawText ?? "");
|
|
234
|
-
const
|
|
235
|
-
|
|
236
|
-
if (end < 0) {
|
|
233
|
+
const bounds = findXmlElement(text);
|
|
234
|
+
if (bounds?.end == null) {
|
|
237
235
|
return null;
|
|
238
236
|
}
|
|
239
|
-
const trailing = text.slice(end
|
|
237
|
+
const trailing = text.slice(bounds.end);
|
|
240
238
|
if (trailing.includes("<agent_response")) {
|
|
241
239
|
return null;
|
|
242
240
|
}
|
|
@@ -332,9 +330,21 @@ function looseTagText(xml, tag) {
|
|
|
332
330
|
// guessing the arguments of a side-effecting operation from broken XML is
|
|
333
331
|
// unsafe, so those still fall through to a format retry.
|
|
334
332
|
function recoverMessageOnlyResponse(envelope) {
|
|
335
|
-
|
|
333
|
+
const tags = [...xmlTags(envelope)];
|
|
334
|
+
if (
|
|
335
|
+
tags.some((tag) => /^(tool_call|tool_calls|invoke|action|parameter)$/.test(tag.name))
|
|
336
|
+
|| tags.filter((tag) => tag.name === "agent_response" && !tag.closing).length !== 1
|
|
337
|
+
) {
|
|
336
338
|
return null;
|
|
337
339
|
}
|
|
340
|
+
// Tolerate malformed display markup only inside <message>. Unknown sibling
|
|
341
|
+
// fields must not disappear when a damaged operation is mistaken for prose.
|
|
342
|
+
const root = tags.find((tag) => tag.name === "agent_response" && !tag.closing);
|
|
343
|
+
const closing = tags.findLast((tag) => tag.name === "agent_response" && tag.closing);
|
|
344
|
+
const siblings = envelope.slice(root.end, closing.start)
|
|
345
|
+
.replace(/<done(?:\s[^>]*)?>[\s\S]*?<\/done>/g, "")
|
|
346
|
+
.replace(/<message(?:\s[^>]*)?>[\s\S]*?<\/message>/g, "");
|
|
347
|
+
if (siblings.trim()) return null;
|
|
338
348
|
const doneText = looseTagText(envelope, "done");
|
|
339
349
|
const message = looseTagText(envelope, "message");
|
|
340
350
|
if (doneText == null || message == null) {
|
|
@@ -366,6 +376,11 @@ export function parseAgentResponse(rawText) {
|
|
|
366
376
|
if (/<!DOCTYPE|<!ENTITY/i.test(structuralXml)) {
|
|
367
377
|
throw new ProtocolError("DTD and XML entities are not allowed.");
|
|
368
378
|
}
|
|
379
|
+
if ([...xmlTags(rawEnvelope)].filter((tag) => (
|
|
380
|
+
tag.name === "agent_response" && !tag.closing
|
|
381
|
+
)).length !== 1) {
|
|
382
|
+
throw new ProtocolError("Nested <agent_response> envelopes are not allowed.");
|
|
383
|
+
}
|
|
369
384
|
|
|
370
385
|
// Repair the most common, meaning-preserving corruptions before validation:
|
|
371
386
|
// 1. bare ampersands the model wrote outside CDATA
|
|
@@ -402,6 +417,9 @@ export function parseAgentResponse(rawText) {
|
|
|
402
417
|
if (!response || typeof response !== "object" || Array.isArray(response)) {
|
|
403
418
|
throw new ProtocolError("Missing <agent_response> root.");
|
|
404
419
|
}
|
|
420
|
+
if (["action", "invoke", "tool_calls"].some((name) => response[name] != null)) {
|
|
421
|
+
throw new ProtocolError("Operations must use one <tool_call> element with <args>.");
|
|
422
|
+
}
|
|
405
423
|
|
|
406
424
|
const done = parseDone(response.done);
|
|
407
425
|
const message = scalarText(response.message);
|