@xoxoai/checkmate 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (158) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +244 -0
  3. package/dist/ai/client.d.ts +49 -0
  4. package/dist/ai/client.d.ts.map +1 -0
  5. package/dist/ai/client.js +244 -0
  6. package/dist/ai/client.js.map +1 -0
  7. package/dist/ai/message-handler.d.ts +11 -0
  8. package/dist/ai/message-handler.d.ts.map +1 -0
  9. package/dist/ai/message-handler.js +33 -0
  10. package/dist/ai/message-handler.js.map +1 -0
  11. package/dist/ai/message-history.d.ts +19 -0
  12. package/dist/ai/message-history.d.ts.map +1 -0
  13. package/dist/ai/message-history.js +48 -0
  14. package/dist/ai/message-history.js.map +1 -0
  15. package/dist/ai/prompts.d.ts +4 -0
  16. package/dist/ai/prompts.d.ts.map +1 -0
  17. package/dist/ai/prompts.js +24 -0
  18. package/dist/ai/prompts.js.map +1 -0
  19. package/dist/ai/rate-limit-policy.d.ts +7 -0
  20. package/dist/ai/rate-limit-policy.d.ts.map +1 -0
  21. package/dist/ai/rate-limit-policy.js +17 -0
  22. package/dist/ai/rate-limit-policy.js.map +1 -0
  23. package/dist/ai/response-processor.d.ts +20 -0
  24. package/dist/ai/response-processor.d.ts.map +1 -0
  25. package/dist/ai/response-processor.js +62 -0
  26. package/dist/ai/response-processor.js.map +1 -0
  27. package/dist/ai/token-pricing.d.ts +13 -0
  28. package/dist/ai/token-pricing.d.ts.map +1 -0
  29. package/dist/ai/token-pricing.js +348 -0
  30. package/dist/ai/token-pricing.js.map +1 -0
  31. package/dist/ai/token-tracker.d.ts +33 -0
  32. package/dist/ai/token-tracker.d.ts.map +1 -0
  33. package/dist/ai/token-tracker.js +138 -0
  34. package/dist/ai/token-tracker.js.map +1 -0
  35. package/dist/ai/tool-response-handler.d.ts +21 -0
  36. package/dist/ai/tool-response-handler.d.ts.map +1 -0
  37. package/dist/ai/tool-response-handler.js +54 -0
  38. package/dist/ai/tool-response-handler.js.map +1 -0
  39. package/dist/config/runtime-config.d.ts +21 -0
  40. package/dist/config/runtime-config.d.ts.map +1 -0
  41. package/dist/config/runtime-config.js +98 -0
  42. package/dist/config/runtime-config.js.map +1 -0
  43. package/dist/core.d.ts +8 -0
  44. package/dist/core.d.ts.map +1 -0
  45. package/dist/core.js +4 -0
  46. package/dist/core.js.map +1 -0
  47. package/dist/index.d.ts +2 -0
  48. package/dist/index.d.ts.map +1 -0
  49. package/dist/index.js +2 -0
  50. package/dist/index.js.map +1 -0
  51. package/dist/integrations/salesforce/authenticator.d.ts +27 -0
  52. package/dist/integrations/salesforce/authenticator.d.ts.map +1 -0
  53. package/dist/integrations/salesforce/authenticator.js +27 -0
  54. package/dist/integrations/salesforce/authenticator.js.map +1 -0
  55. package/dist/integrations/salesforce/cli-handler.d.ts +14 -0
  56. package/dist/integrations/salesforce/cli-handler.d.ts.map +1 -0
  57. package/dist/integrations/salesforce/cli-handler.js +50 -0
  58. package/dist/integrations/salesforce/cli-handler.js.map +1 -0
  59. package/dist/logging/index.d.ts +2 -0
  60. package/dist/logging/index.d.ts.map +1 -0
  61. package/dist/logging/index.js +4 -0
  62. package/dist/logging/index.js.map +1 -0
  63. package/dist/logging/logger.d.ts +5 -0
  64. package/dist/logging/logger.d.ts.map +1 -0
  65. package/dist/logging/logger.js +15 -0
  66. package/dist/logging/logger.js.map +1 -0
  67. package/dist/playwright.d.ts +102 -0
  68. package/dist/playwright.d.ts.map +1 -0
  69. package/dist/playwright.js +116 -0
  70. package/dist/playwright.js.map +1 -0
  71. package/dist/runtime/extension.d.ts +274 -0
  72. package/dist/runtime/extension.d.ts.map +1 -0
  73. package/dist/runtime/extension.js +171 -0
  74. package/dist/runtime/extension.js.map +1 -0
  75. package/dist/runtime/runner.d.ts +94 -0
  76. package/dist/runtime/runner.d.ts.map +1 -0
  77. package/dist/runtime/runner.js +86 -0
  78. package/dist/runtime/runner.js.map +1 -0
  79. package/dist/runtime/step-execution.d.ts +17 -0
  80. package/dist/runtime/step-execution.d.ts.map +1 -0
  81. package/dist/runtime/step-execution.js +49 -0
  82. package/dist/runtime/step-execution.js.map +1 -0
  83. package/dist/runtime/types.d.ts +79 -0
  84. package/dist/runtime/types.d.ts.map +1 -0
  85. package/dist/runtime/types.js +2 -0
  86. package/dist/runtime/types.js.map +1 -0
  87. package/dist/salesforce.d.ts +71 -0
  88. package/dist/salesforce.d.ts.map +1 -0
  89. package/dist/salesforce.js +73 -0
  90. package/dist/salesforce.js.map +1 -0
  91. package/dist/tools/browser/screenshot-service.d.ts +10 -0
  92. package/dist/tools/browser/screenshot-service.d.ts.map +1 -0
  93. package/dist/tools/browser/screenshot-service.js +23 -0
  94. package/dist/tools/browser/screenshot-service.js.map +1 -0
  95. package/dist/tools/browser/snapshot-filter/index.d.ts +4 -0
  96. package/dist/tools/browser/snapshot-filter/index.d.ts.map +1 -0
  97. package/dist/tools/browser/snapshot-filter/index.js +4 -0
  98. package/dist/tools/browser/snapshot-filter/index.js.map +1 -0
  99. package/dist/tools/browser/snapshot-filter/semantic-scorer.d.ts +15 -0
  100. package/dist/tools/browser/snapshot-filter/semantic-scorer.d.ts.map +1 -0
  101. package/dist/tools/browser/snapshot-filter/semantic-scorer.js +99 -0
  102. package/dist/tools/browser/snapshot-filter/semantic-scorer.js.map +1 -0
  103. package/dist/tools/browser/snapshot-filter/snapshot-filter.d.ts +4 -0
  104. package/dist/tools/browser/snapshot-filter/snapshot-filter.d.ts.map +1 -0
  105. package/dist/tools/browser/snapshot-filter/snapshot-filter.js +57 -0
  106. package/dist/tools/browser/snapshot-filter/snapshot-filter.js.map +1 -0
  107. package/dist/tools/browser/snapshot-filter/tree-reconstructor.d.ts +3 -0
  108. package/dist/tools/browser/snapshot-filter/tree-reconstructor.d.ts.map +1 -0
  109. package/dist/tools/browser/snapshot-filter/tree-reconstructor.js +85 -0
  110. package/dist/tools/browser/snapshot-filter/tree-reconstructor.js.map +1 -0
  111. package/dist/tools/browser/snapshot-service.d.ts +19 -0
  112. package/dist/tools/browser/snapshot-service.d.ts.map +1 -0
  113. package/dist/tools/browser/snapshot-service.js +67 -0
  114. package/dist/tools/browser/snapshot-service.js.map +1 -0
  115. package/dist/tools/browser/tool.d.ts +34 -0
  116. package/dist/tools/browser/tool.d.ts.map +1 -0
  117. package/dist/tools/browser/tool.js +226 -0
  118. package/dist/tools/browser/tool.js.map +1 -0
  119. package/dist/tools/browser/transient-state-tracker.d.ts +16 -0
  120. package/dist/tools/browser/transient-state-tracker.d.ts.map +1 -0
  121. package/dist/tools/browser/transient-state-tracker.js +266 -0
  122. package/dist/tools/browser/transient-state-tracker.js.map +1 -0
  123. package/dist/tools/define-agent-tool.d.ts +45 -0
  124. package/dist/tools/define-agent-tool.d.ts.map +1 -0
  125. package/dist/tools/define-agent-tool.js +55 -0
  126. package/dist/tools/define-agent-tool.js.map +1 -0
  127. package/dist/tools/dispatcher.d.ts +11 -0
  128. package/dist/tools/dispatcher.d.ts.map +1 -0
  129. package/dist/tools/dispatcher.js +49 -0
  130. package/dist/tools/dispatcher.js.map +1 -0
  131. package/dist/tools/loop-detector.d.ts +27 -0
  132. package/dist/tools/loop-detector.d.ts.map +1 -0
  133. package/dist/tools/loop-detector.js +76 -0
  134. package/dist/tools/loop-detector.js.map +1 -0
  135. package/dist/tools/registry.d.ts +21 -0
  136. package/dist/tools/registry.d.ts.map +1 -0
  137. package/dist/tools/registry.js +46 -0
  138. package/dist/tools/registry.js.map +1 -0
  139. package/dist/tools/salesforce/login-tool.d.ts +7 -0
  140. package/dist/tools/salesforce/login-tool.d.ts.map +1 -0
  141. package/dist/tools/salesforce/login-tool.js +25 -0
  142. package/dist/tools/salesforce/login-tool.js.map +1 -0
  143. package/dist/tools/step/result-tool.d.ts +7 -0
  144. package/dist/tools/step/result-tool.d.ts.map +1 -0
  145. package/dist/tools/step/result-tool.js +32 -0
  146. package/dist/tools/step/result-tool.js.map +1 -0
  147. package/dist/tools/tool-contract.d.ts +3 -0
  148. package/dist/tools/tool-contract.d.ts.map +1 -0
  149. package/dist/tools/tool-contract.js +2 -0
  150. package/dist/tools/tool-contract.js.map +1 -0
  151. package/dist/tools/types.d.ts +139 -0
  152. package/dist/tools/types.d.ts.map +1 -0
  153. package/dist/tools/types.js +4 -0
  154. package/dist/tools/types.js.map +1 -0
  155. package/docs/EXTENSIONS.md +233 -0
  156. package/docs/GUIDE.md +467 -0
  157. package/docs/ROADMAP.md +47 -0
  158. package/package.json +106 -0
package/docs/GUIDE.md ADDED
@@ -0,0 +1,467 @@
1
+ # **_checkmate_** docs
2
+
3
+ Technical documentation for **_checkmate_** - AI test automation with Playwright.
4
+
5
+ ## Table of Contents
6
+
7
+ - [Core Concepts](#core-concepts)
8
+ - [Configuration Reference](#configuration-reference)
9
+ - [Writing Effective Tests](#writing-effective-tests)
10
+ - [Cost Management](#cost-management)
11
+ - [Web Extension](#web-extension)
12
+ - [Salesforce Extension](#salesforce-extension)
13
+ - [Test Reports](#test-reports)
14
+ - [Troubleshooting](#troubleshooting)
15
+ - [Architecture](#architecture)
16
+ - [Advanced Topics](#advanced-topics)
17
+
18
+ ## Core Concepts
19
+
20
+ **_checkmate_** is an AI-driven test runner. You describe a step in natural language, **_checkmate_** runs a tool loop, and the step passes or fails based on the observed result.
21
+
22
+ Main building blocks:
23
+
24
+ - **Runner**: The object that executes steps. The main programmatic entry point is `createRunner()` from `@xoxoai/checkmate/core`.
25
+ - **Step**: A plain object with `action` and `expect`. This is the main unit of execution.
26
+ - **Extensions**: Composable modules that add tools and runtime behavior. Built-ins include `web()` and `salesforce()`.
27
+ - **Fixtures**: Convenience Playwright entry points that provide an `ai` runner in tests.
28
+
29
+ Published entry points:
30
+
31
+ - `@xoxoai/checkmate/core`: Build your own runner with extensions.
32
+ - `@xoxoai/checkmate/playwright`: Use the built-in web extension with Playwright `test` and `expect`.
33
+ - `@xoxoai/checkmate/salesforce`: Use the built-in web + Salesforce extensions with the same `ai` fixture shape.
34
+
35
+ Most users start here:
36
+
37
+ ```typescript
38
+ import { test } from '@xoxoai/checkmate/playwright'
39
+
40
+ test('search flow', async ({ ai }) => {
41
+ await ai.run({
42
+ action: 'Search for playwright documentation',
43
+ expect: 'Search results are displayed',
44
+ })
45
+ })
46
+ ```
47
+
48
+ ## Configuration Reference
49
+
50
+ ### AI API Settings
51
+
52
+ | Variable | Default | Description |
53
+ | --------------------------------------- | ------------ | --------------------------------------------------------------------------------------------------------- |
54
+ | `OPENAI_API_KEY` | - | **Required** - Your OpenAI API key (or compatible provider) |
55
+ | `OPENAI_BASE_URL` | - | Optional - Override for compatible providers (Claude, Gemini, local LLMs) |
56
+ | `OPENAI_MODEL` | `gpt-5-mini` | Model: gpt-5, gemini-2.5-flash, claude-4-5-sonnet etc. |
57
+ | `OPENAI_TEMPERATURE` | `1.0` | Creativity (below 0.5 = deterministic, above 0.5 = creative) |
58
+ | `OPENAI_REASONING_EFFORT` | - | Optional - Reasoning effort for models: low, medium, high |
59
+ | `OPENAI_TIMEOUT_SECONDS` | `60` | API request timeout in seconds |
60
+ | `OPENAI_API_RATE_LIMIT_DELAY_SECONDS` | `0` | Optional fixed delay before each API call, useful when your provider is sensitive to burst traffic |
61
+ | `OPENAI_RETRY_MAX_ATTEMPTS` | `3` | Max retries with backoff (1s, 10s, 60s) for rate limits and server errors |
62
+ | `OPENAI_TOOL_CHOICE` | `required` | Tool choice: auto, required, none |
63
+ | `OPENAI_ALLOWED_TOOLS` | - | Comma-separated list of allowed tools (if not set, all tools available) |
64
+ | `OPENAI_INCLUDE_SCREENSHOT_IN_SNAPSHOT` | `false` | Include compressed screenshots in snapshot responses |
65
+ | `OPENAI_API_TOKEN_BUDGET_USD` | - | Optional - USD budget for total OpenAI API spend per test run. Only positive decimal values are enforced. |
66
+ | `OPENAI_API_TOKEN_BUDGET_COUNT` | - | Optional - Token count limit for total tokens per test run. Only positive integers are enforced. |
67
+ | `OPENAI_LOOP_MAX_REPETITIONS` | `5` | Number of repetitive tool call patterns to detect before triggering loop recovery with random temperature |
68
+ | `CHECKMATE_LOG_LEVEL` | `off` | Logging verbosity: debug, info, warn, error, off |
69
+ | `CHECKMATE_SNAPSHOT_FILTERING` | `false` | Enable semantic page snapshot filtering before requests are sent to the model |
70
+
71
+ ### Playwright Configuration
72
+
73
+ Browser settings (viewport, headless mode, video recording, timeouts, etc.) are configured in [playwright.config.ts](../playwright.config.ts) using Playwright's [standard](https://playwright.dev/docs/test-configuration) configuration mechanism.
74
+
75
+ ## Writing Effective Tests
76
+
77
+ ### Best Practices
78
+
79
+ 1. **Be Specific** - Clear expectations help the AI validate success
80
+ 2. **One Action Per Step** - Break complex flows into discrete steps
81
+ 3. **Include Context** - Mention relevant UI elements and expected behavior
82
+ 4. **Add Timing Hints** - For slow operations, mention expected wait times
83
+ 5. **Handle Popups** - Explicitly mention consent dialogs or modals
84
+
85
+ ### Basic Example
86
+
87
+ ```typescript
88
+ import { expect, test } from '@xoxoai/checkmate/playwright'
89
+
90
+ test('search for playwright documentation', async ({ page, ai }) => {
91
+ await test.step('Navigate to Google', async () => {
92
+ await ai.run({
93
+ action: `Open the browser and navigate to google.com`,
94
+ expect: `google.com is loaded and the search bar is visible`,
95
+ })
96
+ })
97
+
98
+ await test.step('Search for Playwright', async () => {
99
+ await ai.run({
100
+ action: `Type 'playwright test automation' in the search bar and press Enter`,
101
+ expect: `Search results contain the playwright.dev link`,
102
+ })
103
+ })
104
+
105
+ await expect(page.getByRole('link', { name: /playwright/i }).first()).toBeVisible()
106
+ })
107
+ ```
108
+
109
+ ### Complex Interactions
110
+
111
+ ```typescript
112
+ await test.step('Fill form and submit', async () => {
113
+ await ai.run({
114
+ action: `
115
+ Wait for the newsletter popup (takes ~30 seconds),
116
+ then close it by clicking the X button.
117
+ Scroll to the comment section and click to activate it.
118
+ Type 'Great article!' into the comment textarea.
119
+ Click the Submit button.
120
+ `,
121
+ expect: `
122
+ The comment is submitted,
123
+ and either a success message appears
124
+ or a login form is displayed if not authenticated.
125
+ `,
126
+ })
127
+ })
128
+ ```
129
+
130
+ ### Programmatic Composition
131
+
132
+ Use `@xoxoai/checkmate/core` when you want to build your own runner explicitly:
133
+
134
+ ```typescript
135
+ import { createRunner } from '@xoxoai/checkmate/core'
136
+ import { web } from '@xoxoai/checkmate/playwright'
137
+ import { jira, notion, database } from 'your-own-extension-examples'
138
+
139
+ const ai = createRunner({
140
+ extensions: [web({ page }), jira(), notion(), database()],
141
+ })
142
+ ```
143
+
144
+ ## Cost Management
145
+
146
+ **_checkmate_** includes built-in token usage monitoring:
147
+
148
+ ```json
149
+ {
150
+ "response input": "2543 @ $0.00$",
151
+ "response output": "456 @ $0.00$",
152
+ "history (estimated)": 45234,
153
+ "step input": "5123 @ $0.00$",
154
+ "step output": "892 @ $0.00$",
155
+ "test input": "25678 @ $0.01$",
156
+ "test output": "4521 @ $0.01$"
157
+ }
158
+ ```
159
+
160
+ ### Cost Optimization Features
161
+
162
+ 1. **Smart Snapshots** - Instead of full HTML, only the ARIA accessibility tree is sent to the AI
163
+ 2. **History Filtering** - Continuously filters old page snapshots (reduces token usage by up to 50%)
164
+ 3. **Snapshot Minification** - Removes unnecessary whitespace and quotes from ARIA snapshots
165
+ 4. **Snapshot Filtering** - Local semantic filtering of page snapshots using the current step description (reduces token usage by up to 90%)
166
+ 5. **Screenshots** - Normalized and compressed locally, helps vision models understand UI better
167
+ 6. **Chat Recycling** - New session per step to prevent context bloat and isolation
168
+ 7. **Token Counting** - Real-time usage tracking per step and test with budgets
169
+ 8. **Loop Detection** - Detects and mitigates repetitive tool call patterns, preventing AI runaway costs
170
+
171
+ ### Budgeting & Cost Limits
172
+
173
+ You can set one or both token budget environment variables to enforce limits during a single test run.
174
+
175
+ - `OPENAI_API_TOKEN_BUDGET_USD` — Sets a USD budget (e.g. 0.50) per test execution. The framework checks the current estimated cost (input+output tokens) and throws an error if the budget is exceeded.
176
+ - `OPENAI_API_TOKEN_BUDGET_COUNT` — Sets a token limit (e.g. 100000). The framework tracks input and output tokens across the test and throws an error when the total exceeds this limit.
177
+
178
+ Notes:
179
+
180
+ - Only positive numbers are enforced; `0` or non-positive values are effectively treated as disabled.
181
+ - If the env var is unset or invalid (non-number), it is ignored.
182
+
183
+ ### Using Snapshot Filtering for Token Optimization
184
+
185
+ When snapshot filtering is enabled, **_checkmate_** scores the page snapshot locally with a semantic embedding model and keeps the most relevant branches of the accessibility tree.
186
+
187
+ Default behavior:
188
+
189
+ - Build one query from `action + expect`
190
+ - Score snapshot keys and string leaves against that query
191
+ - If `search` is provided on the step, use those keywords instead of semantic `action + expect`
192
+ - Keep the top `10%` of scored elements by default
193
+ - If top-percent selection yields nothing, fall back to hard threshold `0.3`
194
+
195
+ **This feature significantly reduces the payload size, minimizing costs while improving AI determinism, reliability and speed.**
196
+
197
+ ```typescript
198
+ await ai.run({
199
+ action: `Click on the link that leads to playwright.dev`,
200
+ expect: `The playwright.dev homepage is displayed`,
201
+
202
+ // optional snapshot filtering override
203
+ topPercent: 20,
204
+ })
205
+ ```
206
+
207
+ ```
208
+ debug: Scored 107 elements
209
+ debug: Filtered to 21 elements from top 20%
210
+ debug: Reduced snapshot from 4283 to 326 chars (92% reduction)
211
+ ```
212
+
213
+ Feature is controlled by the `CHECKMATE_SNAPSHOT_FILTERING` environment variable (default: `false`). Set it explicitly to `true` to enable filtering. `search` is now an explicit keyword query override, and `topPercent` lets you tune how much of the scored snapshot should be kept for a specific step.
214
+
215
+ The model can still request a full snapshot with the browser snapshot tool if the filtered tree is insufficient, so steps should not fail just because the initial snapshot was compact.
216
+
217
+ For optimal results, write concrete `action` and `expect` text. Use `topPercent` as a real percentage from `1` to `100` when you need to keep more or less of the scored snapshot. Optional `search` terms still help when you want direct keyword control.
218
+
219
+ **Tips for effective step text:**
220
+
221
+ - Include relevant UI element types (button, input, link, checkbox, etc.)
222
+ - Include key text that appears on the page
223
+ - Include action-related terms (search, filter, submit, etc.)
224
+ - Keep the step focused on one user intent
225
+ - Use `topPercent` only when you need to tune how aggressively snapshot content is pruned
226
+
227
+ ### Estimated Costs
228
+
229
+ **Gemini-2.5-flash / GPT-5-mini**:
230
+
231
+ - Simple test (~5 steps): ~$0.01 - $0.05
232
+ - Complex test (~20 steps): ~$0.10 - $0.40
233
+ - Full E2E suite (~50 complex tests): ~$5.00 - $20.00
234
+
235
+ **GPT-OSS-20B via groq**:
236
+
237
+ - Simple test (~5 steps): ~$0.001 - $0.01
238
+ - Complex test (~20 steps): ~$0.01 - $0.05
239
+ - Full E2E suite (~50 complex tests): ~$1.00 - $2.00
240
+
241
+ _Costs vary based on model, screenshot size and count, and page complexity_
242
+
243
+ ## Web Extension
244
+
245
+ `@xoxoai/checkmate/playwright` is the pre-built web entry point. It composes the core runner with the built-in `web()` extension and exposes a Playwright-friendly `ai` fixture.
246
+
247
+ What it adds:
248
+
249
+ - browser tools for navigation and interaction
250
+ - initial page snapshots and optional screenshots
251
+ - `test`, `expect`, `web()`, and `createPlaywrightRunner(page)` exports
252
+
253
+ ```typescript
254
+ import { test } from '@xoxoai/checkmate/playwright'
255
+
256
+ test('search flow', async ({ ai }) => {
257
+ await ai.run({
258
+ action: 'Search for playwright documentation',
259
+ expect: 'Search results are displayed',
260
+ })
261
+ })
262
+ ```
263
+
264
+ ## Salesforce Extension
265
+
266
+ `@xoxoai/checkmate/salesforce` builds on the web extension. It adds Salesforce-specific tools and keeps the same `ai` fixture shape as the Playwright entry point.
267
+
268
+ What it adds:
269
+
270
+ - the built-in `salesforce()` extension
271
+ - `test`, `expect`, and `createSalesforceRunner(page)` exports
272
+ - the `login_to_salesforce_org` tool backed by the Salesforce CLI
273
+
274
+ Prerequisites:
275
+
276
+ ```bash
277
+ # Install Salesforce CLI
278
+ npm install -g @salesforce/cli
279
+
280
+ # Authenticate to your org and set is as default
281
+ sf org login web --alias my-checkmate-org --set-default
282
+ ```
283
+
284
+ ```typescript
285
+ import { test } from '@xoxoai/checkmate/salesforce'
286
+
287
+ test('create and configure itinerary', async ({ ai }) => {
288
+ await test.step('Login to Salesforce', async () => {
289
+ await ai.run({
290
+ action: 'Login to Salesforce org and open Test QA Application',
291
+ expect: 'Test QA homepage is displayed',
292
+ })
293
+ })
294
+ })
295
+ ```
296
+
297
+ The `login_to_salesforce_org` tool handles the authentication flow by retrieving a front-door URL from the authenticated SF CLI session and navigating the browser for you.
298
+
299
+ ## Test Reports
300
+
301
+ Multiple report formats are generated after each run:
302
+
303
+ - **HTML Report**: `test-reports/html/index.html` (interactive - no screenshots/video yet though)
304
+ - **JUnit XML**: `test-reports/junit/results.xml` (CI/CD integration)
305
+ - **Console Output**: Real-time step results and token usage
306
+
307
+ ```bash
308
+ # Open HTML report in browser
309
+ npx playwright show-report test-reports/html
310
+ ```
311
+
312
+ ## Troubleshooting
313
+
314
+ ### AI makes incorrect decisions
315
+
316
+ **Symptoms**: The AI clicks wrong elements, misinterprets the page, or fails to complete actions correctly.
317
+
318
+ **Solutions**:
319
+
320
+ - Provide more precise descriptions in `action` and more focused assertions in `expect`
321
+ - Reference specific element identifiers and roles (for example: text, label, button, list)
322
+ - Break complex workflows into single-action steps; use a step-by-step approach
323
+
324
+ ### Tests loop during step execution
325
+
326
+ **Symptoms**: The AI repeats the same actions or gets stuck in a loop, consuming tokens unnecessarily.
327
+
328
+ **Solutions**:
329
+
330
+ - Increase `OPENAI_TEMPERATURE` to encourage exploration
331
+ - Use a reasoning/thinking model (if available) to improve planning and avoid repetitive loops
332
+
333
+ ### High token costs
334
+
335
+ **Symptoms**: Tests consume more tokens than expected, leading to high API costs.
336
+
337
+ **Solutions**:
338
+
339
+ - Set a lower reasoning effort: `OPENAI_REASONING_EFFORT`
340
+ - Consider disabling `OPENAI_INCLUDE_SCREENSHOT_IN_SNAPSHOT`
341
+ - Use a cheaper model, lower-end models often perform well (e.g., `gemini-2.5-flash-lite` or `gpt-5-nano`)
342
+
343
+ ### Rate limiting errors
344
+
345
+ **Symptoms**: API calls fail with 429 errors or rate limit messages.
346
+
347
+ **Solutions**:
348
+
349
+ - The framework automatically retries with backoff (1s, 10s, 60s)
350
+ - Upgrade your API plan with your provider
351
+ - Reduce concurrent test execution
352
+ - Increase `OPENAI_TIMEOUT_SECONDS` if needed
353
+
354
+ ### Timeout errors
355
+
356
+ **Symptoms**: Tests fail with timeout errors before completing actions.
357
+
358
+ **Solutions**:
359
+
360
+ - Increase `OPENAI_TIMEOUT_SECONDS` in your `.env` file
361
+ - Mention expected wait times in your action descriptions
362
+ - Break long-running actions into smaller steps
363
+
364
+ ## Architecture
365
+
366
+ **_checkmate_** combines multiple components to enable AI-driven test automation:
367
+
368
+ ```
369
+ @xoxoai/checkmate/core
370
+ │
371
+ ├── createRunner({ extensions })
372
+ ├── runtime/
373
+ │ ├── CheckmateRunner
374
+ │ ├── StepExecution
375
+ │ └── ExtensionHost
376
+ │
377
+ ├── ai/
378
+ │ ├── AiClient
379
+ │ ├── ResponseProcessor
380
+ │ ├── MessageHistory
381
+ │ └── TokenTracker
382
+ │
383
+ ├── tools/
384
+ │ └── step/
385
+ │ └── StepResultTools
386
+ │
387
+ ├── @xoxoai/checkmate/playwright
388
+ │ └── web()
389
+ │ ├── BrowserToolRuntime
390
+ │ ├── SnapshotService
391
+ │ └── Browser tools
392
+ │
393
+ └── @xoxoai/checkmate/salesforce
394
+ └── salesforce()
395
+ ├── SalesforceTools
396
+ └── Salesforce CLI integration
397
+ ```
398
+
399
+ ### Key Components
400
+
401
+ **Test Layer**
402
+
403
+ - Playwright Test framework manages test execution, reporting, and fixtures
404
+ - Tests written in natural language via `ai.run()` fixtures
405
+
406
+ **Core Engine**
407
+
408
+ - **createRunner**: Public composition entry point for building runners from extensions
409
+ - **CheckmateRunner**: Runtime instance returned by `createRunner`
410
+ - **AiClient**: Manages model interactions, retries, and tool-calling requests
411
+ - **Response Processor**: Handles tool responses, append-only history, and retries through the step loop
412
+ - **ExtensionHost**: Registers tools, instructions, step context builders, and post-tool hooks from extensions
413
+ - **Tool Registry**: Owns Zod-defined tool declarations and explicit tool resolution
414
+
415
+ **Tools**
416
+
417
+ - **Core Tools**: Step control (pass/fail step assertions)
418
+ - **Web Extension**: Playwright-powered browser tools, snapshots, and screenshots
419
+ - **Salesforce Extension**: SF CLI login flow layered on top of the web extension
420
+
421
+ **Cost Optimization**
422
+
423
+ - Token tracking with budget enforcement
424
+ - History filtering (removes old snapshots)
425
+ - Snapshot minification and screenshot compression
426
+ - Loop detection and mitigation
427
+
428
+ **Configuration**
429
+
430
+ - Test, Reporting and Browser settings: [playwright.config.ts](../playwright.config.ts)
431
+ - API & AI settings: `.env` file
432
+
433
+ ## Advanced Topics
434
+
435
+ ### Custom Tool Integration
436
+
437
+ For custom tools, extensions, built-in extension composition, and custom runners, see the dedicated [Extensions guide](./EXTENSIONS.md).
438
+
439
+ ### Performance Optimization
440
+
441
+ For large test suites:
442
+
443
+ - Use faster models for simple tests (e.g., `gemini-3-flash-preview` or `gpt-5-mini`)
444
+ - Set token budgets to prevent runaway costs
445
+ - Disable screenshots in snapshots when visual context isn't needed
446
+ - Consider parallel test execution with Playwright's workers
447
+
448
+ ### CI/CD Integration
449
+
450
+ **_checkmate_** generates JUnit XML reports compatible with most CI/CD systems:
451
+
452
+ ```yaml
453
+ # Example GitHub Actions
454
+ - name: Run Tests
455
+ run: npm test
456
+
457
+ - name: Upload Reports
458
+ uses: actions/upload-artifact@v3
459
+ with:
460
+ name: test-reports
461
+ path: test-reports/
462
+ ```
463
+
464
+ ## See Also
465
+
466
+ - [EXTENSIONS](./EXTENSIONS.md)
467
+ - [README](../README.md)
@@ -0,0 +1,47 @@
1
+ # Roadmap
2
+
3
+ ## Current State:
4
+
5
+ - ✅ Extension-composed runtime via `createRunner({ extensions })`
6
+ - ✅ Clear top-level module boundaries: `runtime`, `ai`, `tools`, `integrations`, `config`, `logging`
7
+ - ✅ Explicit tool registration and dispatch
8
+ - ✅ Browser snapshot filtering with semantic scoring
9
+ - ✅ Token tracking, retry handling, loop detection, and screenshot support
10
+ - ✅ Salesforce login integration through the SF CLI
11
+ - ✅ Published subpath entry points for `@xoxoai/checkmate/core`, `@xoxoai/checkmate/playwright`, and `@xoxoai/checkmate/salesforce`
12
+
13
+ ## Near Term
14
+
15
+ Focus: Stability, extension points, and better contributor ergonomics.
16
+
17
+ - ✅ Custom tool registration API for external integrations
18
+ - ✅ Better public examples for programmatic runner usage
19
+ - ✅ Publishable npm package layout with dedicated `core`, `playwright`, and `salesforce` entry points
20
+ - [ ] Visual interactions (click, drag, etc.) in the Playwright extension
21
+ - [ ] Snapshot filtering tuning hooks beyond top-percent selection
22
+ - [ ] Better reporting around filtered snapshot size and selected branches
23
+
24
+ ## Mid Term
25
+
26
+ Focus: Product usability and broader workflow support.
27
+
28
+ - [ ] UI layer for recording, editing, and replaying AI-driven steps
29
+ - [ ] Flow-level execution mode for multi-step business journeys
30
+ - [ ] Richer debugging output for model/tool reasoning failures
31
+ - [ ] Better parallel execution support across large suites
32
+
33
+ ## Long Term
34
+
35
+ Focus: Production hardening and ecosystem.
36
+
37
+ - [ ] Stronger observability and explainable AI
38
+ - [ ] Test generation from specs and recorded user behavior
39
+ - [ ] Advanced reporting with AI-assisted failure summaries
40
+ - [ ] Enterprise-focused environment and secret management support
41
+
42
+ ## Ongoing Research
43
+
44
+ - 🔄 Faster local retrieval/filtering for very large page snapshots
45
+ - 🔄 Hybrid semantic plus structural ranking for element selection
46
+ - 🔄 Multi-agent execution models for planning and validation
47
+ - 🔄 Confidence signals for tool selection and assertions
package/package.json ADDED
@@ -0,0 +1,106 @@
1
+ {
2
+ "name": "@xoxoai/checkmate",
3
+ "version": "0.4.0",
4
+ "description": "AI‑driven e2e test automation framework built on Playwright Test and OpenAI API",
5
+ "homepage": "https://github.com/dawiddiwad/checkmate",
6
+ "repository": {
7
+ "type": "git",
8
+ "url": "https://github.com/dawiddiwad/checkmate.git"
9
+ },
10
+ "keywords": [
11
+ "test-automation",
12
+ "e2e-testing",
13
+ "playwright",
14
+ "openai",
15
+ "ai-testing",
16
+ "low-code",
17
+ "salesforce",
18
+ "gemini",
19
+ "claude",
20
+ "groq"
21
+ ],
22
+ "files": [
23
+ "dist",
24
+ "docs/EXTENSIONS.md",
25
+ "docs/GUIDE.md",
26
+ "docs/ROADMAP.md",
27
+ "README.md",
28
+ "LICENSE"
29
+ ],
30
+ "publishConfig": {
31
+ "access": "public"
32
+ },
33
+ "main": "dist/core.js",
34
+ "types": "dist/core.d.ts",
35
+ "type": "module",
36
+ "exports": {
37
+ ".": {
38
+ "types": "./dist/core.d.ts",
39
+ "default": "./dist/core.js"
40
+ },
41
+ "./core": {
42
+ "types": "./dist/core.d.ts",
43
+ "default": "./dist/core.js"
44
+ },
45
+ "./playwright": {
46
+ "types": "./dist/playwright.d.ts",
47
+ "default": "./dist/playwright.js"
48
+ },
49
+ "./salesforce": {
50
+ "types": "./dist/salesforce.d.ts",
51
+ "default": "./dist/salesforce.js"
52
+ }
53
+ },
54
+ "scripts": {
55
+ "checkmate:install": "npm install && npx playwright install",
56
+ "build": "npm run validation:check && npm run build:dist",
57
+ "build:dist": "tsc -p tsconfig.build.json",
58
+ "clean": "rm -rf dist",
59
+ "prepack": "npm run clean && npm run build",
60
+ "validation:check": "npm run compile:check && npm run lint:check && npm run format:fix && npm run test:unit:run",
61
+ "compile:check": "npx tsc --noEmit",
62
+ "lint:check": "eslint .",
63
+ "lint:fix": "eslint . --fix",
64
+ "format:check": "prettier . --check",
65
+ "format:fix": "prettier . --write",
66
+ "test:unit": "vitest --config src/test/vitest.config.ts",
67
+ "test:unit:ui": "vitest --ui --config src/test/vitest.config.ts",
68
+ "test:unit:coverage": "vitest --coverage --config src/test/vitest.config.ts",
69
+ "test:unit:run": "vitest run --config src/test/vitest.config.ts",
70
+ "test:web": "npm run build:dist && npx playwright test --project=web",
71
+ "test:salesforce": "npm run build:dist && npx playwright test --project=salesforce",
72
+ "test:web:example": "npm run build:dist && npx playwright test --project=web --grep=ollama",
73
+ "show:report": "npx playwright show-report test-reports/html"
74
+ },
75
+ "author": "Dawid Dobrowolski SoftQA",
76
+ "license": "MIT",
77
+ "dependencies": {
78
+ "@huggingface/transformers": "^4.0.1",
79
+ "openai": "^6.34.0",
80
+ "sharp": "^0.34.5",
81
+ "string-similarity": "^4.0.4",
82
+ "strip-ansi": "^7.2.0",
83
+ "winston": "^3.19.0",
84
+ "yaml": "^2.8.3",
85
+ "zod": "^4.3.6"
86
+ },
87
+ "peerDependencies": {
88
+ "@playwright/test": "^1.59.1"
89
+ },
90
+ "devDependencies": {
91
+ "@eslint/js": "^10.0.1",
92
+ "@playwright/test": "^1.59.1",
93
+ "@types/node": "^25.6.0",
94
+ "@types/string-similarity": "^4.0.2",
95
+ "@vitest/coverage-v8": "^4.1.4",
96
+ "@vitest/ui": "^4.1.2",
97
+ "dotenv": "^17.4.1",
98
+ "eslint": "^10.2.0",
99
+ "eslint-config-prettier": "^10.1.8",
100
+ "prettier": "^3.8.3",
101
+ "tsx": "^4.21.0",
102
+ "typescript": "^5.9.3",
103
+ "typescript-eslint": "^8.58.2",
104
+ "vitest": "^4.1.2"
105
+ }
106
+ }