wtagent 0.1.0 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,246 @@
1
+ import { BaseWebAdapter, firstVisible } from "./base-web-adapter.js";
2
+ import { isUsageLimitNotice } from "../shared/usage-limit.js";
3
+
4
+ export { isConnectionLostError } from "./base-web-adapter.js";
5
+
6
+ const GLM_URL = "https://chat.z.ai/";
7
+
8
+ // Preferred models, newest first: try GLM-5.3 when present, else GLM-5.2. The
9
+ // site sometimes exposes 5.3 and sometimes only 5.2, so selection walks this
10
+ // list and clicks the first one present in the menu.
11
+ const PREFERRED_MODELS = ["GLM-5.3", "GLM-5.2"];
12
+
13
+ // GLM / Z.ai (chat.z.ai) adapter.
14
+ //
15
+ // chat.z.ai is an Open WebUI (Svelte) frontend, verified against the live app:
16
+ // - composer: `textarea#chat-input` (placeholder "有什么我能帮您的?" / "有什么我能帮您的?")
17
+ // - messages: `#messages-container [id^="message-<uuid>"]`, each id carrying a
18
+ // STABLE per-message UUID; the user row has class `user-message`, the
19
+ // assistant row instead contains `#response-content-container`. So identity
20
+ // uses the base's default id-based ladder (like Kimi, unlike DeepSeek).
21
+ // - assistant answer markdown is `.markdown-prose` / `.prose`; a "思考过程"
22
+ // (deep-thinking) block may precede it and is excluded when reading the reply
23
+ // - a live conversation URL is /c/<uuid>
24
+ // - model switcher: `button.modelSelectorButton`; the registry defaultMode
25
+ // "latest" picks the newest available model (GLM-5.3, else GLM-5.2)
26
+ // - Cloudflare guards the site; the base throwIfBlockedPage surfaces the
27
+ // window so the user can pass the check (wtagent's own CDP launch is not
28
+ // fingerprinted the way headless automation is)
29
+ export class GLMWebAdapter extends BaseWebAdapter {
30
+ constructor(options = {}) {
31
+ super({
32
+ ...options,
33
+ baseUrl: options.baseUrl ?? GLM_URL,
34
+ providerName: "GLM",
35
+ });
36
+ }
37
+
38
+ conversationUrlPattern() {
39
+ return /^\/c\//;
40
+ }
41
+
42
+ composerLocators() {
43
+ return [
44
+ this.page.locator("#chat-input"),
45
+ this.page.locator('textarea[placeholder*="帮您"]'),
46
+ this.page.locator("main textarea"),
47
+ ];
48
+ }
49
+
50
+ sendButtonLocators() {
51
+ return [
52
+ this.page.locator("#send-message-button"),
53
+ this.page.getByRole("button", { name: /send|发送/i }),
54
+ ];
55
+ }
56
+
57
+ stopButtonLocators() {
58
+ return [
59
+ this.page.locator("#stop-response-button"),
60
+ this.page.getByRole("button", { name: /stop|停止|中断/i }),
61
+ ];
62
+ }
63
+
64
+ newConversationControls() {
65
+ return [
66
+ this.page.locator("#sidebar-new-chat-button"),
67
+ this.page.getByRole("button", { name: /新聊天|新建聊天|新对话|new chat/i }),
68
+ ];
69
+ }
70
+
71
+ loginControlLocators() {
72
+ return [
73
+ this.page.getByRole("button", { name: /^登录$|^log in$|^sign in$/i }),
74
+ ];
75
+ }
76
+
77
+ // Logged-out visitors are redirected to /auth — a locale-independent signal.
78
+ authUrlPattern() {
79
+ return /^\/auth/;
80
+ }
81
+
82
+ authTextPattern() {
83
+ return /手机号登录|发送验证码|欢迎回来|登录即表示同意|sign in|log in|send code|welcome back/i;
84
+ }
85
+
86
+ assistantMessages() {
87
+ // Assistant message rows: a stable-id message that is NOT the user row.
88
+ return this.page.locator(
89
+ '#messages-container [id^="message-"]:not([id$="-start"]):not(.user-message)',
90
+ );
91
+ }
92
+
93
+ userMessages() {
94
+ return this.page.locator(
95
+ '#messages-container [id^="message-"]:not([id$="-start"]).user-message',
96
+ );
97
+ }
98
+
99
+ conversationMessages() {
100
+ return this.page.locator(
101
+ '#messages-container [id^="message-"]:not([id$="-start"])',
102
+ );
103
+ }
104
+
105
+ // Each message row carries a stable UUID in its element id (message-<uuid>),
106
+ // so the base's default id-based identity ladder works directly.
107
+ async messageIdentity(message) {
108
+ const raw = await message.getAttribute("id").catch(() => null);
109
+ const id = raw ? raw.replace(/^message-/, "") : null;
110
+ return { id, turn: null };
111
+ }
112
+
113
+ async assistantText(message) {
114
+ // GLM (with 深度思考 on) often renders the whole reply — including the
115
+ // protocol XML — inside a `思考过程` thinking-chain block, and streams a
116
+ // "正在思考 / 跳过" placeholder first. Rather than try to isolate a separate
117
+ // answer node (there often isn't one), return the full message text: the
118
+ // protocol parser tolerates the leading 思考过程 text and extracts the
119
+ // <agent_response> envelope.
120
+ return await message.innerText().catch(() => "");
121
+ }
122
+
123
+ // Structural completion signal — the same idea as Kimi's action bar, and
124
+ // fully locale-independent: GLM renders the response action bar
125
+ // (.copy-response-button / .regenerate-response-button) under an assistant
126
+ // message only after it has FULLY finished streaming. While generating (or
127
+ // during the 深度思考 thinking phase), those Svelte slots stay empty, so no
128
+ // buttons exist yet. This also covers the blank row GLM mounts a few seconds
129
+ // before the first token: no buttons = still generating.
130
+ hasReliableCompletionSignal() {
131
+ return true;
132
+ }
133
+
134
+ async isAssistantGenerating(message) {
135
+ const buttons = message.locator(
136
+ ".copy-response-button, .regenerate-response-button",
137
+ );
138
+ return await buttons.count().catch(() => 0) === 0;
139
+ }
140
+
141
+ async findUsageLimitMarker(message) {
142
+ const text = await message.innerText().catch(() => "");
143
+ if (!isUsageLimitNotice(text)) {
144
+ return null;
145
+ }
146
+ const control = await firstVisible([
147
+ message.getByRole("button", { name: /重试|重新生成|retry|升级|upgrade/i }),
148
+ message.locator('[class*="error" i]'),
149
+ ]);
150
+ return control ? text.trim().slice(0, 120) : null;
151
+ }
152
+
153
+ // GLM keeps 深度思考 on and shows no persistent stop button, so — like DeepSeek
154
+ // and Kimi — the silent phase before the first token must not be misread as a
155
+ // dead request.
156
+ deadRequestGraceMultiplier() {
157
+ return 5;
158
+ }
159
+
160
+ sentUserWaitAttempts() {
161
+ return 120;
162
+ }
163
+
164
+ // Selects the newest available model. The registry's defaultMode "latest" maps
165
+ // to PREFERRED_MODELS (GLM-5.3, else GLM-5.2). Best-effort and non-throwing.
166
+ //
167
+ // The switcher is `button.modelSelectorButton`; opening it lists options whose
168
+ // visible text is the exact model name. After clicking, the switcher label
169
+ // becomes the selected model name.
170
+ async selectMode(mode) {
171
+ this.requirePage();
172
+ if (mode !== "latest") {
173
+ return { status: "skipped", requested: mode, attempts: 0 };
174
+ }
175
+
176
+ const switcher = this.page.locator("button.modelSelectorButton").first();
177
+ await switcher.waitFor({ state: "visible", timeout: 10_000 }).catch(() => null);
178
+ if (await switcher.count().catch(() => 0) === 0) {
179
+ await this.writeDiagnostics("glm-model-switcher-not-found");
180
+ return {
181
+ status: "switcher_not_found",
182
+ requested: mode,
183
+ attempts: 0,
184
+ reason: "Model switcher was not found.",
185
+ };
186
+ }
187
+
188
+ const current = (await switcher.innerText().catch(() => "")).trim();
189
+ // Already on the most-preferred model that exists? If the current label is
190
+ // the first preferred model, nothing to do.
191
+ if (current.startsWith(PREFERRED_MODELS[0])) {
192
+ return {
193
+ status: "already",
194
+ requested: mode,
195
+ selectedLabel: PREFERRED_MODELS[0],
196
+ attempts: 0,
197
+ reason: `Already using ${PREFERRED_MODELS[0]}.`,
198
+ };
199
+ }
200
+
201
+ for (const model of PREFERRED_MODELS) {
202
+ await switcher.click({ timeout: 5_000 }).catch(() => null);
203
+ await this.page.waitForTimeout(600);
204
+ const option = this.page.getByText(model, { exact: true }).first();
205
+ if (await option.count().catch(() => 0) === 0) {
206
+ // Not in the menu; close and try the next preferred model.
207
+ await this.page.keyboard.press("Escape").catch(() => null);
208
+ continue;
209
+ }
210
+ await option.click({ timeout: 5_000 }).catch(() => null);
211
+ await this.page.waitForTimeout(600);
212
+ const after = (await switcher.innerText().catch(() => "")).trim();
213
+ if (after.startsWith(model)) {
214
+ // The model menu stays open after a selection; a click in the page
215
+ // center dismisses it so it does not cover the composer.
216
+ await this.#dismissModelMenu();
217
+ return {
218
+ status: current.startsWith(model) ? "already" : "select",
219
+ requested: mode,
220
+ selectedLabel: model,
221
+ attempts: 1,
222
+ reason: `Selected ${model}.`,
223
+ };
224
+ }
225
+ }
226
+
227
+ await this.#dismissModelMenu();
228
+ await this.writeDiagnostics("glm-mode-latest-unresolved");
229
+ return {
230
+ status: "unresolved",
231
+ requested: mode,
232
+ attempts: 1,
233
+ reason: `Could not select any of: ${PREFERRED_MODELS.join(", ")}.`,
234
+ };
235
+ }
236
+
237
+ async #dismissModelMenu() {
238
+ const viewport = this.page.viewportSize?.() ?? { width: 1280, height: 800 };
239
+ await this.page.mouse.click(
240
+ Math.floor(viewport.width / 2),
241
+ Math.floor(viewport.height / 2),
242
+ ).catch(() => null);
243
+ await this.page.keyboard.press("Escape").catch(() => null);
244
+ await this.page.waitForTimeout(200);
245
+ }
246
+ }
@@ -0,0 +1,285 @@
1
+ import { BaseWebAdapter, firstVisible, hasCompleteAgentEnvelope } from "./base-web-adapter.js";
2
+ import { isUsageLimitNotice } from "../shared/usage-limit.js";
3
+
4
+ export { isConnectionLostError } from "./base-web-adapter.js";
5
+
6
+ const KIMI_URL = "https://www.kimi.com/";
7
+
8
+ // Kimi (www.kimi.com, Moonshot) adapter.
9
+ //
10
+ // Kimi's chat DOM is clean and semantic, verified against the live app:
11
+ // - composer: a Lexical contenteditable `.chat-input-editor` (like ChatGPT),
12
+ // so the base fill/keyboard path works; the send control is
13
+ // `.send-button-container` (disabled when empty), and Enter also sends
14
+ // - messages: `.chat-content-item.chat-content-item-user` and
15
+ // `.chat-content-item.chat-content-item-assistant`
16
+ // - every message carries a STABLE per-message UUID in `data-archer-id`, so
17
+ // the base's default id-based identity ladder works directly (no volatile
18
+ // virtual-list key like DeepSeek)
19
+ // - assistant replies may include a `.thinking-container` reasoning block
20
+ // before the answer; assistantText reads the answer markdown outside it
21
+ // - a live conversation URL is /chat/<uuid>
22
+ // - a model switcher (`.current-model`) opens a Naive-UI popover
23
+ // (`.models-container .model-item`); WTAgent defaults it to K3 — see
24
+ // selectMode.
25
+ export class KimiWebAdapter extends BaseWebAdapter {
26
+ constructor(options = {}) {
27
+ super({
28
+ ...options,
29
+ baseUrl: options.baseUrl ?? KIMI_URL,
30
+ providerName: "Kimi",
31
+ });
32
+ }
33
+
34
+ conversationUrlPattern() {
35
+ return /^\/chat\//;
36
+ }
37
+
38
+ composerLocators() {
39
+ return [
40
+ this.page.locator(".chat-input-editor"),
41
+ this.page.locator('div[contenteditable="true"][data-lexical-editor="true"]'),
42
+ this.page.locator('main div[contenteditable="true"]'),
43
+ this.page.locator('textarea'),
44
+ ];
45
+ }
46
+
47
+ sendButtonLocators() {
48
+ // The send control is a div, not a <button>; it carries `disabled` in its
49
+ // class when the composer is empty. firstVisible + isEnabled won't detect
50
+ // the class-based disable, so the base's Enter fallback is the reliable
51
+ // path — these are best-effort only.
52
+ return [
53
+ this.page.locator(".send-button-container:not(.disabled)"),
54
+ this.page.getByRole("button", { name: /send|发送/i }),
55
+ ];
56
+ }
57
+
58
+ // Kimi shows no persistent labeled stop button; generation liveness is not
59
+ // proven by a stop control, so the wider dead-request grace applies (below).
60
+ stopButtonLocators() {
61
+ return [
62
+ this.page.getByRole("button", { name: /stop|停止|中断/i }),
63
+ ];
64
+ }
65
+
66
+ newConversationControls() {
67
+ return [
68
+ this.page.getByRole("button", { name: /新建会话|新对话|new chat/i }),
69
+ this.page.getByRole("link", { name: /新建会话|新对话|new chat/i }),
70
+ ];
71
+ }
72
+
73
+ loginControlLocators() {
74
+ return [
75
+ this.page.getByRole("button", { name: /^(登录|log in|sign in)$/i }),
76
+ ];
77
+ }
78
+
79
+ authTextPattern() {
80
+ return /手机号快捷登录|手机号登录|发送验证码|登录以同步|log in to sync|sign in/i;
81
+ }
82
+
83
+ assistantMessages() {
84
+ return this.page.locator(".chat-content-item-assistant");
85
+ }
86
+
87
+ userMessages() {
88
+ return this.page.locator(".chat-content-item-user");
89
+ }
90
+
91
+ conversationMessages() {
92
+ return this.page.locator(".chat-content-item");
93
+ }
94
+
95
+ // Kimi tags every message with a stable UUID in data-archer-id — a genuine
96
+ // per-message id, so the base id-based identity ladder works unchanged.
97
+ async messageIdentity(message) {
98
+ const id = await message.getAttribute("data-archer-id").catch(() => null);
99
+ return { id, turn: null };
100
+ }
101
+
102
+ async assistantText(message) {
103
+ // Prefer the visible answer markdown. Native-tool cards and thinking
104
+ // blocks share `.markdown` too, so exclude those first; if the only text
105
+ // left is a mid-tool failure such as "文件阅读失败", keep waiting for the
106
+ // later protocol envelope instead of treating that card as the reply.
107
+ // `.markdown` is nested inside `.markdown-container`. Selecting both
108
+ // duplicates the same reply (Kimi then looks like it emitted two envelopes).
109
+ const answer = message.locator(
110
+ ".markdown:not(.thinking-container .markdown):not(.toolcall-container .markdown)",
111
+ );
112
+ if (await answer.count().catch(() => 0)) {
113
+ const chunks = [];
114
+ const seen = new Set();
115
+ const count = await answer.count();
116
+ for (let index = 0; index < count; index += 1) {
117
+ const text = (await answer.nth(index).innerText().catch(() => "")).trim();
118
+ if (!text || seen.has(text)) {
119
+ continue;
120
+ }
121
+ seen.add(text);
122
+ chunks.push(text);
123
+ }
124
+ const joined = chunks.join("\n");
125
+ if (joined.includes("<agent_response") || !this.#isNativeToolPlaceholder(joined)) {
126
+ return joined;
127
+ }
128
+ }
129
+ return await message.innerText().catch(() => "");
130
+ }
131
+
132
+ #isNativeToolPlaceholder(text) {
133
+ const trimmed = String(text ?? "").trim();
134
+ if (!trimmed || trimmed.includes("<agent_response")) {
135
+ return false;
136
+ }
137
+ return /文件阅读失败|文件读取失败|阅读失败|read file failed|tool (?:call )?failed/i.test(trimmed);
138
+ }
139
+
140
+ hasReliableCompletionSignal() {
141
+ return true;
142
+ }
143
+
144
+ async isAssistantGenerating(message) {
145
+ if (await this.#isNativeToolRunning(message)) {
146
+ return true;
147
+ }
148
+ // Kimi renders the action bar (.segment-assistant-actions — copy/share/…
149
+ // icons) under a reply only once it has FULLY finished streaming. Its
150
+ // absence is the reliable completion signal: a structural DOM check that
151
+ // does not depend on localized status text ("思考中" vs "思考已完成"), so
152
+ // a mid-stream pause can never read as a finished reply.
153
+ return await message
154
+ .locator(".segment-assistant-actions")
155
+ .count()
156
+ .catch(() => 0) === 0;
157
+ }
158
+
159
+ async extraStableWindowMs(message, text) {
160
+ if (hasCompleteAgentEnvelope(text)) {
161
+ return 0;
162
+ }
163
+ if (await this.#hasNativeToolCard(message) || this.#isNativeToolPlaceholder(text)) {
164
+ return 8_000;
165
+ }
166
+ return 0;
167
+ }
168
+
169
+ async #isNativeToolRunning(message) {
170
+ // Completed cards show completion markers: 已完成/失败, or a result count
171
+ // ("9 个结果" / "N 条结果" / "N results") for a finished web search. Only a
172
+ // card free of ALL markers can be in flight.
173
+ const running = message.locator(
174
+ ".toolcall-container:not(.thinking-container):not(:has-text('已完成')):not(:has-text('失败'))"
175
+ + ":not(:has-text('个结果')):not(:has-text('条结果')):not(:has-text('results')), "
176
+ + ".toolcall-title-container:not(.thinking-container .toolcall-title-container)",
177
+ );
178
+ if (await running.count().catch(() => 0) === 0) {
179
+ return false;
180
+ }
181
+ const title = await running.first().innerText().catch(() => "");
182
+ if (/已完成|失败|failed|个结果|条结果|results/i.test(title)) {
183
+ return false;
184
+ }
185
+ return /阅读|读取|搜索|执行|运行中|running|reading|searching/i.test(title);
186
+ }
187
+
188
+ async #hasNativeToolCard(message) {
189
+ const cards = message.locator(
190
+ ".toolcall-container:not(.thinking-container), "
191
+ + ".toolcall-title-container:not(.thinking-container .toolcall-title-container)",
192
+ );
193
+ return await cards.count().catch(() => 0) > 0;
194
+ }
195
+
196
+ async findUsageLimitMarker(message) {
197
+ const text = await message.innerText().catch(() => "");
198
+ if (!isUsageLimitNotice(text)) {
199
+ return null;
200
+ }
201
+ const control = await firstVisible([
202
+ message.getByRole("button", { name: /重试|重新生成|retry|升级|upgrade/i }),
203
+ message.locator('[class*="error" i]'),
204
+ ]);
205
+ return control ? text.trim().slice(0, 120) : null;
206
+ }
207
+
208
+ // Kimi does not expose a persistent stop button, so — like DeepSeek — the
209
+ // silent phase before the first token must not be misread as a dead request.
210
+ deadRequestGraceMultiplier() {
211
+ return 5;
212
+ }
213
+
214
+ // Kimi has a model switcher (快速 / K3 / K3 集群). The registry's defaultMode
215
+ // "k3" asks to switch to K3 ("擅长对话与 Agent 任务,全能旗舰") on every fresh
216
+ // conversation, silently. Any other value keeps the current model.
217
+ //
218
+ // The switcher is `.current-model`; it opens a Naive-UI popover whose rows are
219
+ // `.models-container .model-item`, the selected one carrying `checked`. After
220
+ // selecting, the switcher label starts with the model name (e.g. "K3 进阶").
221
+ // Best-effort and non-throwing, mirroring runModeSelection's contract.
222
+ async selectMode(mode) {
223
+ this.requirePage();
224
+ if (mode !== "k3") {
225
+ return { status: "skipped", requested: mode, attempts: 0 };
226
+ }
227
+
228
+ const switcher = this.page.locator(".current-model").first();
229
+ // The switcher can mount a beat after the composer on a fresh conversation;
230
+ // wait briefly before concluding it is absent.
231
+ await switcher.waitFor({ state: "visible", timeout: 10_000 }).catch(() => null);
232
+ if (await switcher.count().catch(() => 0) === 0) {
233
+ await this.writeDiagnostics("kimi-model-switcher-not-found");
234
+ return {
235
+ status: "switcher_not_found",
236
+ requested: mode,
237
+ attempts: 0,
238
+ reason: "Model switcher was not found.",
239
+ };
240
+ }
241
+
242
+ // Already on K3? The switcher label starts with "K3" (but not "K3 集群").
243
+ const label = (await switcher.innerText().catch(() => "")).trim();
244
+ if (/^K3(?!\s*集群)/.test(label)) {
245
+ return {
246
+ status: "already",
247
+ requested: mode,
248
+ selectedLabel: "K3",
249
+ attempts: 0,
250
+ reason: "Already using K3.",
251
+ };
252
+ }
253
+
254
+ await switcher.click({ timeout: 5_000 }).catch(() => null);
255
+ await this.page.locator(".models-container .model-item")
256
+ .first().waitFor({ state: "visible", timeout: 5_000 }).catch(() => null);
257
+
258
+ // Click the row whose title line is exactly "K3" (not "快速" / "K3 集群").
259
+ const k3 = this.page.locator(".models-container .model-item").filter({
260
+ hasText: /^K3(?!\s*集群)/,
261
+ }).first();
262
+ const clicked = await k3.count().catch(() => 0) > 0
263
+ && await k3.click({ timeout: 5_000 }).then(() => true).catch(() => false);
264
+ await this.page.waitForTimeout(500);
265
+
266
+ const after = (await switcher.innerText().catch(() => "")).trim();
267
+ if (clicked && /^K3(?!\s*集群)/.test(after)) {
268
+ return {
269
+ status: "select",
270
+ requested: mode,
271
+ selectedLabel: "K3",
272
+ attempts: 1,
273
+ reason: "Selected K3.",
274
+ };
275
+ }
276
+ await this.page.keyboard.press("Escape").catch(() => null);
277
+ await this.writeDiagnostics("kimi-mode-k3-unresolved");
278
+ return {
279
+ status: "unresolved",
280
+ requested: mode,
281
+ attempts: 1,
282
+ reason: "Could not confirm K3 was selected.",
283
+ };
284
+ }
285
+ }