wtagent 0.2.2 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,145 @@
1
+ import { BaseWebAdapter, firstVisible } from "./base-web-adapter.js";
2
+ import { isUsageLimitNotice } from "../shared/usage-limit.js";
3
+
4
+ export { isConnectionLostError } from "./base-web-adapter.js";
5
+
6
+ const GROK_URL = "https://grok.com/";
7
+
8
+ export class GrokWebAdapter extends BaseWebAdapter {
9
+ constructor(options = {}) {
10
+ super({
11
+ ...options,
12
+ baseUrl: options.baseUrl ?? GROK_URL,
13
+ providerName: "Grok",
14
+ });
15
+ }
16
+
17
+ conversationUrlPattern() {
18
+ return /^\/c\//;
19
+ }
20
+
21
+ composerLocators() {
22
+ return [
23
+ this.page.locator("main textarea").first(),
24
+ this.page.locator("textarea[aria-label]").first(),
25
+ this.page.locator('[role="textbox"]').first(),
26
+ ];
27
+ }
28
+
29
+ sendButtonLocators() {
30
+ return [
31
+ this.page.locator('button[aria-label*="send" i]'),
32
+ this.page.locator('button[aria-label*="发送"]'),
33
+ this.page.locator('button[type="submit"]'),
34
+ ];
35
+ }
36
+
37
+ stopButtonLocators() {
38
+ return [
39
+ this.page.locator('button[aria-label*="停止模型响应"]'),
40
+ this.page.locator('button[aria-label*="stop" i]'),
41
+ ];
42
+ }
43
+
44
+ newConversationControls() {
45
+ return [
46
+ this.page.locator('[data-testid="new-chat"]'),
47
+ this.page.locator('a[href="/"]'),
48
+ ];
49
+ }
50
+
51
+ loginControlLocators() {
52
+ return [
53
+ this.page.getByRole("button", {
54
+ name: /^(log in|sign in|登录|登入|ログイン|로그인)$/i,
55
+ }),
56
+ this.page.getByRole("link", {
57
+ name: /^(log in|sign in|登录|登入|ログイン|로그인)$/i,
58
+ }),
59
+ ];
60
+ }
61
+
62
+ authUrlPattern() {
63
+ return /^\/(?:sign-in|signin|login|auth)(?:\/|$)/i;
64
+ }
65
+
66
+ authTextPattern() {
67
+ return /sign in to grok|log in to grok|continue with x|continue with google|登录 Grok|登入 Grok/i;
68
+ }
69
+
70
+ assistantMessages() {
71
+ return this.page.locator('[data-testid="assistant-message"]');
72
+ }
73
+
74
+ userMessages() {
75
+ return this.page.locator(
76
+ '[data-testid="user-message"], [role="article"][aria-label="You"]',
77
+ );
78
+ }
79
+
80
+ conversationMessages() {
81
+ return this.page.locator(
82
+ '[data-testid="user-message"], [data-testid="assistant-message"], '
83
+ + '[role="article"][aria-label="You"]',
84
+ );
85
+ }
86
+
87
+ async messageIdentity(message) {
88
+ const testId = await message.getAttribute("data-testid").catch(() => null);
89
+ const aria = await message.getAttribute("aria-label").catch(() => null);
90
+ const scope = testId === "user-message" || aria === "You"
91
+ ? this.userMessages()
92
+ : this.assistantMessages();
93
+ return { id: null, turn: await scope.count().catch(() => 0) };
94
+ }
95
+
96
+ isNewAssistantIdentity({ turn, text }) {
97
+ if (turn != null && turn > (this.assistantCountBeforeSend ?? 0)) {
98
+ return true;
99
+ }
100
+ const previous = String(this.lastAssistantTextBeforeSend ?? "");
101
+ return Boolean(previous && String(text ?? "") !== previous);
102
+ }
103
+
104
+ async assistantText(message) {
105
+ const response = message.locator(".response-content-markdown");
106
+ if (await response.count().catch(() => 0) > 0) {
107
+ return await response.last().innerText().catch(() => "");
108
+ }
109
+ return await message.innerText().catch(() => "");
110
+ }
111
+
112
+ hasReliableCompletionSignal() {
113
+ return true;
114
+ }
115
+
116
+ async isAssistantGenerating() {
117
+ return Boolean(await firstVisible(this.stopButtonLocators()));
118
+ }
119
+
120
+ async findUsageLimitMarker(message) {
121
+ const text = await message.innerText().catch(() => "");
122
+ if (!isUsageLimitNotice(text)) return null;
123
+ const control = await firstVisible([
124
+ message.getByRole("button", { name: /retry|try again|upgrade|重试|升级/i }),
125
+ this.page.getByRole("button", { name: /retry|try again|upgrade|重试|升级/i }),
126
+ ]);
127
+ return control ? text.trim().slice(0, 120) : null;
128
+ }
129
+
130
+ async findGenerationErrorMarker(message) {
131
+ const text = await message.innerText().catch(() => "");
132
+ if (!/something went wrong|failed to generate|generation failed|try again|生成失败|出错了/i.test(text)) {
133
+ return null;
134
+ }
135
+ const control = await firstVisible([
136
+ message.getByRole("button", { name: /retry|try again|重试|重新生成/i }),
137
+ this.page.getByRole("button", { name: /retry|try again|重试|重新生成/i }),
138
+ ]);
139
+ return control ? text.trim().slice(0, 120) : null;
140
+ }
141
+
142
+ deadRequestGraceMultiplier() {
143
+ return 3;
144
+ }
145
+ }
@@ -19,9 +19,8 @@ const KIMI_URL = "https://www.kimi.com/";
19
19
  // - assistant replies may include a `.thinking-container` reasoning block
20
20
  // before the answer; assistantText reads the answer markdown outside it
21
21
  // - a live conversation URL is /chat/<uuid>
22
- // - a model switcher (`.current-model`) opens a Naive-UI popover
23
- // (`.models-container .model-item`); WTAgent defaults it to K3 — see
24
- // selectMode.
22
+ // Model choice is intentionally left to the user on kimi.com; the adapter does
23
+ // not inspect or change the site's current model selection.
25
24
  export class KimiWebAdapter extends BaseWebAdapter {
26
25
  constructor(options = {}) {
27
26
  super({
@@ -211,75 +210,4 @@ export class KimiWebAdapter extends BaseWebAdapter {
211
210
  return 5;
212
211
  }
213
212
 
214
- // Kimi has a model switcher (快速 / K3 / K3 集群). The registry's defaultMode
215
- // "k3" asks to switch to K3 ("擅长对话与 Agent 任务,全能旗舰") on every fresh
216
- // conversation, silently. Any other value keeps the current model.
217
- //
218
- // The switcher is `.current-model`; it opens a Naive-UI popover whose rows are
219
- // `.models-container .model-item`, the selected one carrying `checked`. After
220
- // selecting, the switcher label starts with the model name (e.g. "K3 进阶").
221
- // Best-effort and non-throwing, mirroring runModeSelection's contract.
222
- async selectMode(mode) {
223
- this.requirePage();
224
- if (mode !== "k3") {
225
- return { status: "skipped", requested: mode, attempts: 0 };
226
- }
227
-
228
- const switcher = this.page.locator(".current-model").first();
229
- // The switcher can mount a beat after the composer on a fresh conversation;
230
- // wait briefly before concluding it is absent.
231
- await switcher.waitFor({ state: "visible", timeout: 10_000 }).catch(() => null);
232
- if (await switcher.count().catch(() => 0) === 0) {
233
- await this.writeDiagnostics("kimi-model-switcher-not-found");
234
- return {
235
- status: "switcher_not_found",
236
- requested: mode,
237
- attempts: 0,
238
- reason: "Model switcher was not found.",
239
- };
240
- }
241
-
242
- // Already on K3? The switcher label starts with "K3" (but not "K3 集群").
243
- const label = (await switcher.innerText().catch(() => "")).trim();
244
- if (/^K3(?!\s*集群)/.test(label)) {
245
- return {
246
- status: "already",
247
- requested: mode,
248
- selectedLabel: "K3",
249
- attempts: 0,
250
- reason: "Already using K3.",
251
- };
252
- }
253
-
254
- await switcher.click({ timeout: 5_000 }).catch(() => null);
255
- await this.page.locator(".models-container .model-item")
256
- .first().waitFor({ state: "visible", timeout: 5_000 }).catch(() => null);
257
-
258
- // Click the row whose title line is exactly "K3" (not "快速" / "K3 集群").
259
- const k3 = this.page.locator(".models-container .model-item").filter({
260
- hasText: /^K3(?!\s*集群)/,
261
- }).first();
262
- const clicked = await k3.count().catch(() => 0) > 0
263
- && await k3.click({ timeout: 5_000 }).then(() => true).catch(() => false);
264
- await this.page.waitForTimeout(500);
265
-
266
- const after = (await switcher.innerText().catch(() => "")).trim();
267
- if (clicked && /^K3(?!\s*集群)/.test(after)) {
268
- return {
269
- status: "select",
270
- requested: mode,
271
- selectedLabel: "K3",
272
- attempts: 1,
273
- reason: "Selected K3.",
274
- };
275
- }
276
- await this.page.keyboard.press("Escape").catch(() => null);
277
- await this.writeDiagnostics("kimi-mode-k3-unresolved");
278
- return {
279
- status: "unresolved",
280
- requested: mode,
281
- attempts: 1,
282
- reason: "Could not confirm K3 was selected.",
283
- };
284
- }
285
213
  }
@@ -2,6 +2,7 @@ import { ChatGPTWebAdapter } from "./chatgpt-web-adapter.js";
2
2
  import { ClaudeWebAdapter } from "./claude-web-adapter.js";
3
3
  import { DeepSeekWebAdapter } from "./deepseek-web-adapter.js";
4
4
  import { GeminiWebAdapter } from "./gemini-web-adapter.js";
5
+ import { GrokWebAdapter } from "./grok-web-adapter.js";
5
6
  import { KimiWebAdapter } from "./kimi-web-adapter.js";
6
7
  import { GLMWebAdapter } from "./glm-web-adapter.js";
7
8
  import { getProfileDir } from "../platform/paths.js";
@@ -18,15 +19,13 @@ import { getProfileDir } from "../platform/paths.js";
18
19
  // data dir; each provider logs in once, independently
19
20
  // status - "active" (has a working adapter) | "planned" (named
20
21
  // but not implemented yet)
21
- // promptsForMode - whether the CLI shows an interactive mode picker at
22
- // conversation start (ChatGPT has Pro/Current; most
23
- // providers do not prompt)
24
- // defaultMode - mode applied silently at conversation start when the
25
- // provider does not prompt (null = keep the site's
26
- // current setting). The adapter's selectMode()
27
- // interprets this value.
28
22
  // adapter - the adapter class, or null until implemented
29
23
  //
24
+ // WTAgent deliberately does not encode provider-specific model names or choose
25
+ // a model automatically. After login, interactive runs give the user a chance
26
+ // to choose directly on the provider website; pressing Enter keeps whatever
27
+ // the website currently has selected.
28
+ //
30
29
  // ChatGPT keeps the historical "chrome-profile" basename so existing logins,
31
30
  // the logout guard, and cli.test.js keep working unchanged.
32
31
  export const PROVIDERS = Object.freeze({
@@ -36,8 +35,6 @@ export const PROVIDERS = Object.freeze({
36
35
  baseUrl: "https://chatgpt.com/",
37
36
  profileBasename: "chrome-profile",
38
37
  status: "active",
39
- promptsForMode: true,
40
- defaultMode: null,
41
38
  adapter: ChatGPTWebAdapter,
42
39
  }),
43
40
  deepseek: Object.freeze({
@@ -46,10 +43,6 @@ export const PROVIDERS = Object.freeze({
46
43
  baseUrl: "https://chat.deepseek.com/",
47
44
  profileBasename: "deepseek-profile",
48
45
  status: "active",
49
- // DeepSeek does not prompt; every new conversation silently switches to
50
- // 专家模式 (Expert) + 深度思考 (Deep Thinking) — see DeepSeekWebAdapter.selectMode.
51
- promptsForMode: false,
52
- defaultMode: "expert-thinking",
53
46
  adapter: DeepSeekWebAdapter,
54
47
  }),
55
48
  claude: Object.freeze({
@@ -58,10 +51,6 @@ export const PROVIDERS = Object.freeze({
58
51
  baseUrl: "https://claude.ai/",
59
52
  profileBasename: "claude-profile",
60
53
  status: "active",
61
- promptsForMode: false,
62
- // Keep whatever model claude.ai selects in this browser profile. WTAgent
63
- // does not open the model menu or override the account/site default.
64
- defaultMode: null,
65
54
  adapter: ClaudeWebAdapter,
66
55
  }),
67
56
  grok: Object.freeze({
@@ -69,10 +58,8 @@ export const PROVIDERS = Object.freeze({
69
58
  label: "Grok",
70
59
  baseUrl: "https://grok.com/",
71
60
  profileBasename: "grok-profile",
72
- status: "planned",
73
- promptsForMode: false,
74
- defaultMode: null,
75
- adapter: null,
61
+ status: "active",
62
+ adapter: GrokWebAdapter,
76
63
  }),
77
64
  kimi: Object.freeze({
78
65
  id: "kimi",
@@ -80,10 +67,6 @@ export const PROVIDERS = Object.freeze({
80
67
  baseUrl: "https://www.kimi.com/",
81
68
  profileBasename: "kimi-profile",
82
69
  status: "active",
83
- // Kimi does not prompt; every new conversation silently switches to the K3
84
- // flagship model — see KimiWebAdapter.selectMode.
85
- promptsForMode: false,
86
- defaultMode: "k3",
87
70
  adapter: KimiWebAdapter,
88
71
  }),
89
72
  glm: Object.freeze({
@@ -92,10 +75,6 @@ export const PROVIDERS = Object.freeze({
92
75
  baseUrl: "https://chat.z.ai/",
93
76
  profileBasename: "glm-profile",
94
77
  status: "active",
95
- // GLM does not prompt; every new conversation silently selects the newest
96
- // available model (GLM-5.3, else GLM-5.2) — see GLMWebAdapter.selectMode.
97
- promptsForMode: false,
98
- defaultMode: "latest",
99
78
  adapter: GLMWebAdapter,
100
79
  }),
101
80
  gemini: Object.freeze({
@@ -104,9 +83,6 @@ export const PROVIDERS = Object.freeze({
104
83
  baseUrl: "https://gemini.google.com/app",
105
84
  profileBasename: "gemini-profile",
106
85
  status: "active",
107
- promptsForMode: false,
108
- // Preserve the model currently selected by Gemini in this profile.
109
- defaultMode: null,
110
86
  adapter: GeminiWebAdapter,
111
87
  }),
112
88
  });
@@ -138,24 +114,6 @@ export function getProvider(providerId) {
138
114
  return provider;
139
115
  }
140
116
 
141
- // `--mode` is ChatGPT's Pro/Current switch, but people often type
142
- // `--mode kimi` meaning the Kimi provider. If the value is a known provider
143
- // id, treat it as `--model` and clear `--mode` so ChatGPT mode parsing is not
144
- // applied. Conflicting `--model` + `--mode <provider>` is rejected.
145
- export function resolveCliProviderSelection({ model, mode } = {}) {
146
- const modeAsProvider = findProvider(mode);
147
- if (!modeAsProvider) {
148
- return { model, mode };
149
- }
150
- if (model != null && getProvider(model).id !== modeAsProvider.id) {
151
- throw new Error(
152
- `--mode ${mode} selects ${modeAsProvider.label}, but --model ${model} is already set. `
153
- + `Use --model ${modeAsProvider.id}.`,
154
- );
155
- }
156
- return { model: modeAsProvider.id, mode: undefined };
157
- }
158
-
159
117
  // Resolves the dedicated Chrome profile directory for a provider under the
160
118
  // given app data dir. ChatGPT resolves to the historical "chrome-profile".
161
119
  export function getProviderProfileDir(appDataDir, providerId) {