prompttest 1.4.0 → 1.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,55 @@
1
+ /**
2
+ * @module adaptive-timing
3
+ * @description Device-Aware Adaptive Dynamic Waiting & Auto-Profiling.
4
+ *
5
+ * Automatically balances test timing between ultra-fast physical devices (USB 3.0)
6
+ * and resource-constrained CI emulators (software-rendered VMs in GitHub Actions).
7
+ *
8
+ * Implements exponential backoff condition-driven polling with instant short-circuiting:
9
+ * fast devices finish immediately upon event completion, while slow emulators receive
10
+ * adaptive headroom without CPU exhaustion.
11
+ */
12
+ import type { AndroidDriver } from './adb.js';
13
+ export interface DeviceProfile {
14
+ /** Whether the target device is an emulator / virtual device */
15
+ isEmulator: boolean;
16
+ /** Whether the process is executing inside a CI / CD environment (e.g. GitHub Actions) */
17
+ isCI: boolean;
18
+ /** Calculated dynamic velocity multiplier (1.0 = real device, 1.5 = local emulator, 2.5 = CI) */
19
+ velocityMultiplier: number;
20
+ /** Default timeout ceiling for UI element detection */
21
+ defaultTimeoutMs: number;
22
+ /** Extended timeout ceiling for multi-step transitions and cold loads */
23
+ extendedTimeoutMs: number;
24
+ /** Minimum stability duration for dual-sample quiescence verification */
25
+ quiescenceSettleMs: number;
26
+ /** Initial polling interval in milliseconds */
27
+ initialPollMs: number;
28
+ /** Maximum backoff polling interval in milliseconds */
29
+ maxPollIntervalMs: number;
30
+ }
31
+ /**
32
+ * Probes the connected device and host runtime to construct an adaptive timing profile.
33
+ * Cached per session so that subsequent steps reuse the profiled characteristics.
34
+ *
35
+ * @param driver - Android driver instance
36
+ * @param serial - Optional target device serial
37
+ */
38
+ export declare function profileDeviceEnvironment(driver: AndroidDriver, serial?: string): Promise<DeviceProfile>;
39
+ /**
40
+ * Resets the cached device profile (primarily used in testing).
41
+ */
42
+ export declare function resetDeviceProfile(): void;
43
+ /**
44
+ * Polls an asynchronous condition using adaptive backoff with immediate short-circuiting.
45
+ *
46
+ * Fast devices short-circuit the instant the condition returns a truthy value.
47
+ * Slow emulators receive backoff polling that preserves CPU cycles for rendering.
48
+ *
49
+ * @param checkFn - Async callback returning a truthy value when the condition is met.
50
+ * @param timeoutMs - Maximum duration to wait before giving up.
51
+ * @param initialIntervalMs - Starting interval between checks.
52
+ * @param maxIntervalMs - Cap for the backoff interval.
53
+ * @returns The resolved value of the condition, or throws a timeout Error.
54
+ */
55
+ export declare function pollWithAdaptiveBackoff<T>(checkFn: () => Promise<T | null | undefined | false>, timeoutMs: number, initialIntervalMs?: number, maxIntervalMs?: number): Promise<T>;
package/dist/lib/adb.d.ts CHANGED
@@ -406,6 +406,7 @@ export declare class AndroidDriver {
406
406
  }, localDestPath: string, serial?: string): Promise<string>;
407
407
  /**
408
408
  * Checks if the virtual software keyboard (IME) is currently active and visible on screen.
409
+ * Uses fast shell grep filtering across Android IME & Window properties with resilient fallbacks.
409
410
  * @param serial Optional target device serial.
410
411
  * @returns Promise resolving to true if virtual keyboard is displayed.
411
412
  */
@@ -0,0 +1,59 @@
1
+ /**
2
+ * @module jail-guard
3
+ * @description Zero-Overhead In-Memory App Boundary Guardian for PromptTest.
4
+ *
5
+ * Enforces strict package boundary isolation to guarantee that automation
6
+ * NEVER sends touch, text, or gesture events into third-party personal apps
7
+ * (e.g. banking apps, email, browsers) or home screen launchers.
8
+ *
9
+ * Operates directly on the in-memory parsed hierarchy (UiNode.packageName)
10
+ * with 0 ms additional ADB overhead in normal operation.
11
+ */
12
+ import type { UiNode } from './crawler.js';
13
+ import type { AndroidDriver } from './adb.js';
14
+ /**
15
+ * Registry of known Android OEM launchers and desktop interfaces.
16
+ */
17
+ export declare const KNOWN_LAUNCHER_PACKAGES: readonly string[];
18
+ /**
19
+ * Registry of allowed Android system overlay and framework packages
20
+ * (keyboards, permission dialogs, autofill, package installer).
21
+ */
22
+ export declare const KNOWN_SYSTEM_PACKAGES: readonly string[];
23
+ /**
24
+ * Checks if a given package name corresponds to a known launcher or desktop interface.
25
+ */
26
+ export declare function isLauncherPackage(pkg: string): boolean;
27
+ /**
28
+ * Checks if a given package name corresponds to an allowed Android system overlay.
29
+ */
30
+ export declare function isAllowedSystemPackage(pkg: string): boolean;
31
+ /**
32
+ * Determines whether a specific node's package belongs to the target app under test
33
+ * or an allowed system overlay (keyboard, permission prompt, system UI).
34
+ *
35
+ * @param nodePackage - The package name of the UI node being interacted with.
36
+ * @param targetPackage - The target application package under test (if specified).
37
+ */
38
+ export declare function isTargetOrAllowedSystem(nodePackage?: string, targetPackage?: string): boolean;
39
+ /**
40
+ * Extracts the primary / foreground package from a flattened list of UiNodes.
41
+ * Evaluates non-system, interactive nodes first, then falls back to the root node.
42
+ */
43
+ export declare function detectScreenPackage(nodes: UiNode[]): string | null;
44
+ /**
45
+ * Guards the app boundary: If an escape is detected (screen belongs to a launcher or foreign app),
46
+ * it blocks further actions, prevents unsafe back-presses on launchers, and brings the target app
47
+ * back to the foreground.
48
+ *
49
+ * @param nodes - In-memory UI nodes from current hierarchy dump.
50
+ * @param targetPackage - The expected target application package.
51
+ * @param driver - Android driver instance for containment recovery.
52
+ * @param serial - Optional target device serial.
53
+ * @returns Details on whether an escape occurred and whether recovery succeeded.
54
+ */
55
+ export declare function guardAppBoundary(nodes: UiNode[], targetPackage: string, driver: AndroidDriver, serial?: string): Promise<{
56
+ escaped: boolean;
57
+ recovered: boolean;
58
+ currentPackage: string | null;
59
+ }>;
@@ -19,6 +19,7 @@ export declare class PromptRunner {
19
19
  private lastHierarchy;
20
20
  private screenDimensions;
21
21
  private navBarHeight;
22
+ private targetPackage;
22
23
  /**
23
24
  * Initializes a new instance of the PromptRunner.
24
25
  * Uses dependency injection for core services, defaulting to singletons if omitted.
@@ -0,0 +1,44 @@
1
+ /**
2
+ * @module quiescence
3
+ * @description Dynamic Screen Quiescence & Idling Sentinel for PromptTest.
4
+ *
5
+ * Implements self-synchronizing stability detection inspired by Playwright's networkidle
6
+ * and Android Espresso's IdlingResource. Replaces fragile static timers with dual-sample
7
+ * layout stabilization and active in-flight loading sentinel checks.
8
+ */
9
+ import type { UiNode } from './crawler.js';
10
+ import type { ExecutionContext } from './step-handlers.js';
11
+ export interface QuiescenceResult {
12
+ /** Whether the screen reached verified stability before the timeout */
13
+ settled: boolean;
14
+ /** Milliseconds elapsed during stabilization */
15
+ durationMs: number;
16
+ /** Final structural hash of the quiescent screen */
17
+ finalHash: string;
18
+ /** Flattened UI nodes of the quiescent screen */
19
+ flat: UiNode[];
20
+ }
21
+ /**
22
+ * Inspects a flattened UI hierarchy to detect active in-flight network spinners,
23
+ * shimmers, progress bars, or server-waiting indicators.
24
+ *
25
+ * @param nodes - Flattened array of UiNodes
26
+ * @returns Status indicating whether loading is in progress and why.
27
+ */
28
+ export declare function isLoadingInProgress(nodes: UiNode[]): {
29
+ loading: boolean;
30
+ reason?: string;
31
+ };
32
+ /**
33
+ * Dynamically waits until the application screen achieves 100% quiescence:
34
+ * 1. All active loading indicators and spinners have vanished.
35
+ * 2. Two consecutive UI hierarchy snapshots yield matching structural hashes.
36
+ *
37
+ * Exits immediately once settled, saving seconds on fast devices while providing
38
+ * dynamic patience for slow network requests or cloud emulators.
39
+ *
40
+ * @param ctx - Execution context providing driver and hierarchy access.
41
+ * @param timeoutMs - Maximum wait ceiling (defaults to 4500ms).
42
+ * @param sampleIntervalMs - Interval between stability probe samples (defaults to 150ms).
43
+ */
44
+ export declare function waitForQuiescence(ctx: ExecutionContext, timeoutMs?: number, sampleIntervalMs?: number): Promise<QuiescenceResult>;
@@ -21,6 +21,8 @@ import { PromptStep } from './prompt-runner.js';
21
21
  * It encapsulates the driver instance, state cache, and common UI interaction logic.
22
22
  */
23
23
  export interface ExecutionContext {
24
+ /** Target application package name under test (used for jail guard containment). */
25
+ targetPackage?: string;
24
26
  /** The ADB-based Android UI driver used to interact with the device. */
25
27
  driver: AndroidDriver;
26
28
  /** Memory engine used to record and learn resilient locators over time. */
@@ -6,7 +6,7 @@ PromptTest is a zero-code, AI-assisted Android test automation and visual QA fra
6
6
 
7
7
  ## 1. High-Level Architecture Overview
8
8
 
9
- PromptTest operates on a modular layered architecture where test scripts written in natural language or AST-driven directives are compiled, bound to live or synthetic data, executed against connected devices via an isolated locking layer, and reported with visual diffs and interactive HTML reports.
9
+ PromptTest operates on a modular layered architecture where test scripts written in natural language or AST-driven directives are compiled, bound to live or synthetic data, executed against connected devices via an isolated locking layer, and reported with visual diffs and interactive HTML reports. Current release boundaries are tracked in [PRODUCT_STATUS.md](PRODUCT_STATUS.md).
10
10
 
11
11
  ```
12
12
  ┌───────────────────────────────────────────────────────────┐
@@ -35,14 +35,14 @@ PromptTest operates on a modular layered architecture where test scripts written
35
35
  │ │ │
36
36
  ┌──────▼──────────────────────▼──────────────────────▼──────┐
37
37
  │ Visual Testing & Diagnostics │
38
- │ lib/baseline.ts (SSIM / Perceptual Diff) │
38
+ │ lib/baseline.ts (pixelmatch / PNG Diff) │
39
39
  │ lib/explorer.ts (Autonomous Crawler) │
40
40
  └─────────────────────────────┬─────────────────────────────┘
41
41
  │
42
42
  ┌─────────────────────────────▼─────────────────────────────┐
43
43
  │ Reporting & Observability │
44
44
  │ lib/reporter.ts (HTML, JSON, JUnit XML, Markdown) │
45
- │ lib/live-server.ts (SSE / WebSocket Live Dash) │
45
+ │ lib/live-server.ts (HTTP / SSE Live Monitor) │
46
46
  └───────────────────────────────────────────────────────────┘
47
47
  ```
48
48
 
@@ -100,7 +100,7 @@ Instead of an unwieldy switch-case statement, execution is delegated to decouple
100
100
 
101
101
  ### 2.6 Visual Baselines & Computer Vision (`lib/baseline.ts`, `lib/explorer.ts`)
102
102
 
103
- - Pixel-by-pixel perceptual diffing and Structural Similarity Index Measure (SSIM) algorithms.
103
+ - Pixel-level image diffing using `pixelmatch` and PNG decoding via `pngjs`.
104
104
  - Configurable failure thresholds (e.g. `diffThreshold: 0.02` for 2% variance).
105
105
  - Automatic bounding box masking for dynamic regions (e.g. status bar clocks, live feeds) to eliminate false positives in visual regression testing.
106
106
 
@@ -0,0 +1,131 @@
1
+ # PromptTest CLI Contract
2
+
3
+ **Contract date:** 2026-09-22
4
+ **Package:** `prompttest` 1.3.5
5
+
6
+ This document describes the currently shipped CLI and SDK integration surfaces. It is intended for CI consumers and desktop clients. It does not promise functionality that is only planned or hardware-unvalidated.
7
+
8
+ ## Compatibility Rules
9
+
10
+ - Existing public exports and command behavior are additive-compatible within a major version.
11
+ - Consumers should use exported types and documented fields rather than parsing terminal text.
12
+ - Report fields may gain optional properties. Existing required fields will not be removed without a breaking release.
13
+ - Hardware operations require ADB and a reachable Android device or emulator.
14
+
15
+ ## Exit Codes
16
+
17
+ | Code | Name | Meaning |
18
+ | ---: | -------------------- | ----------------------------------------------------------------- |
19
+ | 0 | `SUCCESS` | Run completed successfully. |
20
+ | 1 | `TEST_FAILED` | An assertion, locator, or test step failed. |
21
+ | 2 | `DEVICE_UNAVAILABLE` | ADB/device connection or device availability failed. |
22
+ | 3 | `CONFIG_ERROR` | A spec, configuration, argument, or include reference is invalid. |
23
+ | 4 | `LOCKED` | The target device is locked by another PromptTest process. |
24
+
25
+ CI integrations should use these codes instead of matching human-readable output.
26
+
27
+ ## Report Model
28
+
29
+ The canonical report is JSON with this shape:
30
+
31
+ ```ts
32
+ interface QaReportData {
33
+ packageName: string;
34
+ deviceId: string;
35
+ startTime: string;
36
+ endTime: string;
37
+ durationSeconds: number;
38
+ totalSteps: number;
39
+ passedSteps: number;
40
+ failedSteps: number;
41
+ skippedSteps: number;
42
+ warnSteps: number;
43
+ errorsDetected: string[];
44
+ videoPath?: string;
45
+ steps: AuditStep[];
46
+ }
47
+
48
+ interface AuditStep {
49
+ stepIndex: number;
50
+ timestamp: string;
51
+ phase: string;
52
+ description: string;
53
+ action: string;
54
+ status: 'PASS' | 'WARN' | 'FAIL' | 'SKIPPED';
55
+ target?: string;
56
+ screenshotPath?: string;
57
+ durationMs?: number;
58
+ triageBundle?: string;
59
+ videoOffsetMs?: number;
60
+ details?: Record<string, unknown>;
61
+ }
62
+ ```
63
+
64
+ Supported serializers are JSON, Markdown, JUnit XML, and HTML. JSON is the machine-readable source of truth; the other formats are views of the same run.
65
+
66
+ ## Output Conventions
67
+
68
+ The default output directory is `output/` in the working directory. Common artifacts include:
69
+
70
+ - `<name>-results.json`
71
+ - `<name>-report.md`
72
+ - `<name>-report.html`
73
+ - `<name>-junit.xml`
74
+ - Failure triage folders or ZIPs containing screenshots, hierarchy XML, logcat, and match scores
75
+ - Baselines under `output/baselines/`
76
+ - Checkpoints under `output/checkpoints/`
77
+ - Optional MP4 recordings referenced by `videoPath`
78
+
79
+ Paths in report fields are relative to the report/output directory unless the producer explicitly returns an absolute path.
80
+
81
+ ## Live Engine Protocol
82
+
83
+ The live server is started with `prompttest serve [port]` or `startLiveServer()`. Defaults:
84
+
85
+ - Host: `127.0.0.1`
86
+ - Port: `4040`
87
+ - Transport: HTTP plus Server-Sent Events
88
+ - Event stream: `GET /events`
89
+ - Report view: `GET /` or `GET /report`
90
+
91
+ The current server has no authentication and sends permissive CORS headers. It must be treated as a local trusted-process interface. Do not bind it to a non-loopback host or expose it outside a controlled machine without adding authentication and access controls.
92
+
93
+ ### Stable read endpoints
94
+
95
+ - `GET /api/devices` returns connected device metadata.
96
+ - `GET /api/screen-state?deviceId=<id>&hierarchy=true|false` returns a screenshot, foreground package, and optional flattened UI elements.
97
+ - `GET /api/specs` lists local `.txt` and `.spec` files.
98
+ - `GET /api/specs/<name>` reads a spec file.
99
+ - `GET /api/reports` lists generated reports.
100
+ - `GET /api/reports/<name>` reads a report detail.
101
+ - `GET /api/baselines` lists available baselines.
102
+ - `GET /api/baselines/masks` lists baseline masks.
103
+ - `GET /api/report` and `GET /results.json` return the current report data.
104
+
105
+ ### Mutating endpoints
106
+
107
+ These endpoints control a connected device or write local files and should only be called by an explicit local user action:
108
+
109
+ - `POST /api/tap` with `{ x, y, deviceId? }`
110
+ - `POST /api/type` with `{ text, deviceId? }`
111
+ - `POST /api/key` with `{ keyCode, deviceId? }`
112
+ - `POST /api/wake` with `{ deviceId? }`
113
+ - `POST /api/specs` with `{ name, content }`
114
+ - `POST /api/action` with `{ step, deviceId? }`
115
+ - `POST /api/run` with `{ specName? | spec?, deviceId?, saveBaseline?, compareBaseline? }`
116
+ - Baseline approval, comparison, upload, mask, clear, and delete endpoints under `/api/baselines/*`
117
+
118
+ ### SSE events
119
+
120
+ Clients connecting to `/events` may receive:
121
+
122
+ - `step`: step number, instruction, status, duration, and optional error
123
+ - `complete`: final `QaReportData`
124
+ - `error`: an error message payload
125
+ - Heartbeat comments every 15 seconds
126
+
127
+ Clients must reconnect and tolerate duplicate or missing transient events; the final report is authoritative.
128
+
129
+ ## Studio Boundary
130
+
131
+ The CLI owns device control, test execution, report generation, and artifact storage. A Studio client owns presentation, browsing, device selection, and desktop packaging. Studio must not duplicate locator resolution or report aggregation logic. Any new endpoint or event should be added here with a compatibility note and an integration test.
@@ -0,0 +1,123 @@
1
+ # CLI and Studio Contract
2
+
3
+ **Contract version:** `1`
4
+ **CLI package:** `prompttest` 1.3.5
5
+ **Status:** Current implementation contract, verified 2026-09-22
6
+
7
+ This document defines the integration surface between the PromptTest CLI engine and a desktop or browser Studio client. The CLI remains independently usable; Studio consumes this contract and must not reimplement device control, test execution, report generation, or baseline comparison.
8
+
9
+ ## Engine Startup
10
+
11
+ Start the engine with:
12
+
13
+ ```bash
14
+ prompttest serve [port]
15
+ ```
16
+
17
+ The default port is `4040`. The server binds to `127.0.0.1` by default. Programmatic consumers can call `startLiveServer({ port, host, outputDir, specsDir })`.
18
+
19
+ The server is not an authenticated remote control plane. It exposes device actions, screenshots, UI hierarchy data, specifications, reports, and baselines. Keep the default loopback binding; do not expose the server on a network interface without adding authentication and authorization first.
20
+
21
+ ## Exit Codes
22
+
23
+ | Code | Meaning |
24
+ | ---: | --------------------------------------------- |
25
+ | `0` | Successful execution |
26
+ | `1` | Test assertion or locator failure |
27
+ | `2` | Device unavailable, offline, or unauthorized |
28
+ | `3` | Invalid configuration, spec, or CLI arguments |
29
+ | `4` | Device lock contention |
30
+
31
+ Studio should display the error message and preserve the numeric exit code when it starts the CLI as a child process.
32
+
33
+ ## Report Contract
34
+
35
+ The canonical report is JSON. The current `QaReportData` shape is:
36
+
37
+ ```json
38
+ {
39
+ "packageName": "com.example.app",
40
+ "deviceId": "emulator-5554",
41
+ "startTime": "2026-09-22T00:00:00.000Z",
42
+ "endTime": "2026-09-22T00:00:03.000Z",
43
+ "durationSeconds": 3,
44
+ "totalSteps": 2,
45
+ "passedSteps": 1,
46
+ "failedSteps": 1,
47
+ "skippedSteps": 0,
48
+ "warnSteps": 0,
49
+ "errorsDetected": [],
50
+ "videoPath": "login-recording.mp4",
51
+ "steps": []
52
+ }
53
+ ```
54
+
55
+ Each step can contain `stepIndex`, `timestamp`, `phase`, `description`, `action`, `target`, `status`, `screenshotPath`, `durationMs`, `triageBundle`, `videoOffsetMs`, and `details`.
56
+
57
+ Supported report serializers are `json`, `md`, `junit`, and `html`. JSON is the integration source of truth. File names use the selected base name:
58
+
59
+ - `<name>-results.json`
60
+ - `<name>-report.md`
61
+ - `<name>-junit.xml`
62
+ - `<name>-report.html`
63
+
64
+ ## SSE Events
65
+
66
+ Connect to `GET /events` with an `EventSource`. Events currently emitted are:
67
+
68
+ - `step`: step number, instruction, status, duration, and optional error
69
+ - `complete`: final `QaReportData`
70
+ - `error`: an error message object
71
+
72
+ Clients should tolerate unknown future events and reconnect after a dropped connection. The server sends comment heartbeats approximately every 15 seconds.
73
+
74
+ ## HTTP Routes
75
+
76
+ All successful JSON responses include `success: true` where applicable. Errors use `success: false` and an `error` string.
77
+
78
+ ### Devices and live device state
79
+
80
+ - `GET /api/devices`: connected device list
81
+ - `GET /api/screen-state?deviceId=<id>&hierarchy=true|false`: screenshot, foreground package, and UI elements
82
+ - `POST /api/tap`: `{ "x": number, "y": number, "deviceId"?: string }`
83
+ - `POST /api/type`: `{ "text": string, "deviceId"?: string }`
84
+ - `POST /api/key`: `{ "keyCode": number|string, "deviceId"?: string }`
85
+ - `POST /api/wake`: `{ "deviceId"?: string }`
86
+
87
+ ### Specifications and execution
88
+
89
+ - `GET /api/specs`: list `.txt` and `.spec` files
90
+ - `GET /api/specs/<name>` or `GET /api/spec?name=<name>`: read a specification
91
+ - `POST /api/specs` or `POST /api/spec`: `{ "name": string, "content": string }`
92
+ - `POST /api/action`: `{ "step": string, "deviceId"?: string }`
93
+ - `POST /api/run`: `{ "specName"?: string, "spec"?: string, "deviceId"?: string, "saveBaseline"?: boolean, "compareBaseline"?: boolean }`
94
+
95
+ `/api/run` starts execution asynchronously. Use SSE and report endpoints to observe completion.
96
+
97
+ ### Reports and baselines
98
+
99
+ - `GET /api/reports`: list available report artifacts
100
+ - `GET /api/reports/<name>` or `GET /api/report-detail?name=<name>`: retrieve report details
101
+ - `GET /api/report` or `GET /results.json`: retrieve the current JSON report
102
+ - `GET /api/baselines`: list baseline images and metadata
103
+ - `POST /api/baselines/compare`: compare an image against a baseline
104
+ - `POST /api/baselines/approve`: approve a baseline result
105
+ - `POST /api/baselines/clear`: clear baseline data
106
+ - `POST /api/baselines/delete-step`: remove a step baseline
107
+ - `POST /api/baselines/upload`: upload a baseline image
108
+ - `GET /api/baselines/masks`: list baseline masks
109
+ - `POST /api/baselines/masks`: create or update a baseline mask
110
+
111
+ The exact request bodies for baseline management should be generated from the CLI client implementation until a schema is added in a future contract version.
112
+
113
+ ## Artifact Rules
114
+
115
+ The CLI owns artifact creation under `output/` by default. Studio should display and download paths returned by reports rather than copying large screenshots or videos into JSON. Triage artifacts may include screenshots, hierarchy XML, logcat, match scores, a manifest, and a ZIP archive. A video path and failure offset do not imply that a clipped video snippet exists.
116
+
117
+ ## Compatibility Rules
118
+
119
+ - Existing public SDK exports and CLI flags are additive and semver-protected.
120
+ - Studio must tolerate missing optional fields and unknown JSON fields.
121
+ - Studio must not assume Android hardware is available during engine startup.
122
+ - A failed run is not an engine crash; inspect the report and exit code separately.
123
+ - Contract changes require a version update, documentation update, and package smoke-test coverage.
@@ -0,0 +1,71 @@
1
+ # PromptTest Performance Baseline
2
+
3
+ **Measured:** 2026-09-22
4
+ **Package:** `prompttest` 1.3.5
5
+ **Environment:** macOS, Node.js 24.13.0, no ADB device required
6
+
7
+ This is a reproducible CI and package-size baseline. Synthetic timings are health signals, not physical-device performance claims.
8
+
9
+ ## Current Measurements
10
+
11
+ - Offline test suite: 39 test files, 337 tests passing
12
+ - Test wall time: approximately 14 seconds in the current environment
13
+ - CLI bundle: approximately 332 KB
14
+ - SDK bundle: approximately 316 KB
15
+ - Engine bundle: approximately 316 KB
16
+ - Runtime dependencies: `chalk`, `pixelmatch`, `pngjs`
17
+ - Supported Node.js range: `>=18.0.0`
18
+
19
+ ## Existing Benchmark Coverage
20
+
21
+ `tests/performance-benchmark.test.ts` and `lib/benchmark.ts` cover:
22
+
23
+ - Synthetic screenshot allocation timing
24
+ - Synthetic UI hierarchy timing
25
+ - Matching latency across a synthetic 1,000-node hierarchy
26
+ - Percentile calculation and summary metrics
27
+ - Hierarchy hashing and structural diffing
28
+ - Verify-handler cache reuse
29
+ - Reporter screenshot-buffer cleanup
30
+ - PromptRunner benchmark mode
31
+
32
+ Run the offline benchmark tests with:
33
+
34
+ ```bash
35
+ npx vitest run tests/performance-benchmark.test.ts
36
+ ```
37
+
38
+ Run the complete offline suite with:
39
+
40
+ ```bash
41
+ npm test
42
+ ```
43
+
44
+ ## Required Hardware Measurements
45
+
46
+ The following values must be collected on representative Android emulator and physical-device configurations before making performance promises:
47
+
48
+ - ADB screenshot p50/p95 latency
49
+ - `uiautomator dump` p50/p95 latency
50
+ - Hierarchy dumps per visited screen
51
+ - Screenshots per exploration state
52
+ - Peak RSS for a 50-screen exploration
53
+ - HTML report size with and without embedded screenshots
54
+ - Triage ZIP size
55
+ - Aggregation time for large output directories
56
+ - Video-enabled versus video-disabled run time
57
+
58
+ Hardware results should include device model, Android version, screen size, Node version, iteration count, and whether the device is USB, Wi-Fi, or emulator. Do not compare synthetic benchmark timings directly with ADB measurements.
59
+
60
+ ## Release Guardrails
61
+
62
+ Any new runtime dependency or report-embedding behavior should be evaluated against:
63
+
64
+ - Published tarball size
65
+ - Installed unpacked size
66
+ - Cold CLI startup time
67
+ - Peak memory during report generation
68
+ - Existing offline test duration
69
+ - Node 18 compatibility
70
+
71
+ The npm package should remain independently usable without Studio or a running server.
@@ -0,0 +1,56 @@
1
+ # PromptTest Product Status
2
+
3
+ **Status date:** 2026-09-22
4
+ **Package:** `prompttest` 1.3.5
5
+
6
+ This document is the current status index for the PromptTest CLI. Older implementation plans and the internal SRS are historical context, not release requirements.
7
+
8
+ ## Status Definitions
9
+
10
+ - **Released:** Implemented, tested in the repository, and included in the published package.
11
+ - **Implemented:** Present in the CLI, but physical-device or cross-platform validation may still be incomplete.
12
+ - **Planned:** Intended work that is not part of the current release contract.
13
+ - **Studio:** Requires the separate desktop application; that source is not part of this repository.
14
+
15
+ ## Current CLI Contract
16
+
17
+ ### Released or implemented
18
+
19
+ - Android execution through ADB
20
+ - Natural-language prompt parsing and structured flow control
21
+ - Device locking and device-pool support
22
+ - Data-driven runs and dynamic test data
23
+ - Visual baselines using `pixelmatch` and PNG decoding
24
+ - Markdown, JSON, JUnit, and HTML reporting
25
+ - Failure triage artifacts and optional video recording
26
+ - Autonomous exploration with DFS traversal and safety modes
27
+ - Optional AI providers, disabled unless configured
28
+ - HTTP/SSE live monitoring server
29
+ - CI workflow generation through `init-ci`
30
+ - Committed offline quality CI and scheduled emulator CI workflows
31
+ - npm package and standalone distribution build paths
32
+
33
+ ### Implemented with validation limits
34
+
35
+ - iOS driver interfaces require external Apple tooling and need a documented real-device acceptance matrix.
36
+ - Autonomous exploration is unit-tested and fixture-tested; coverage across physical applications is not implied.
37
+ - The emulator workflow is configured but requires a successful GitHub Actions run before hardware support is release-validated.
38
+ - Live-server consumers must use the documented loopback and artifact-access contract.
39
+ - Hardware tests require ADB, an emulator, or a connected device and are not part of the offline default test guarantee.
40
+
41
+ ### Planned or deferred
42
+
43
+ - Performance benchmark evidence and output-retention limits
44
+ - Stronger subprocess isolation for untrusted JavaScript specifications
45
+ - Locale packs and broader non-English resolution
46
+ - Spec scaffolding and a dedicated profiler
47
+ - Offline licensing and report watermarking
48
+ - Hosted collaboration, cloud reporting, and multi-device orchestration
49
+
50
+ ## CLI and Studio Boundary
51
+
52
+ The CLI owns device control, execution, comparison, report generation, and artifact contracts. A separate Studio application may consume those contracts for device selection, live views, report browsing, packaging, and desktop workflows. Studio release status must be tracked in its own repository and must not be represented as CLI completion.
53
+
54
+ ## npm Compatibility Policy
55
+
56
+ Changes to the CLI must preserve existing public exports and behavior unless a versioned breaking change is explicitly planned. Every release should pass typecheck, lint, offline tests, package verification, tarball installation smoke tests, and the supported Node.js compatibility checks before publication.
@@ -584,7 +584,7 @@ PromptTest can capture visual baselines and perform perceptual diffing to detect
584
584
 
585
585
  - `--save-baseline`: Saves the current screen layout as a visual baseline for future comparisons.
586
586
  - `--compare-baseline`: Compares the current screen layout against the saved baseline and fails if differences exceed the threshold.
587
- - `--baseline-threshold=N`: Adjusts the perceptual diffing tolerance (e.g., `--baseline-threshold=5` for 5%).
587
+ - `--baseline-threshold=N`: Adjusts the diffing tolerance as a ratio from `0` to `1` (e.g., `--baseline-threshold=0.05` for 5%).
588
588
 
589
589
  ### Custom Reporters
590
590
 
@@ -615,7 +615,7 @@ PromptTest includes a cross-process, file-based mutex lock (`lib/lock.ts`) that
615
615
  PromptTest exposes a clean programmatic API using factory functions and a unified `DriverInterface`. Both Android and iOS drivers share the identical driver abstraction:
616
616
 
617
617
  ```typescript
618
- import { createPromptRunner, createAndroidDriver, createIosDriver } from 'prompttest';
618
+ import { createPromptRunner, createAndroidDriver, IosDriver } from 'prompttest';
619
619
 
620
620
  async function runCrossPlatformTests() {
621
621
  // 🤖 Android Execution
@@ -624,9 +624,8 @@ async function runCrossPlatformTests() {
624
624
  const androidReport = await androidRunner.runSpec('specs/login.txt', 'com.example.app');
625
625
 
626
626
  // 🍎 iOS Execution (Simulator or Physical Device)
627
- const iosDriver = createIosDriver({
627
+ const iosDriver = new IosDriver({
628
628
  udid: '00008101-001234567890',
629
- bundleId: 'com.example.app',
630
629
  });
631
630
  const iosRunner = createPromptRunner(iosDriver);
632
631
  const iosReport = await iosRunner.runSpec('specs/login.txt', 'com.example.app');
@@ -760,6 +759,6 @@ PromptTest is a **fast UI-tree automation engine over ADB**. It "sees" what `uia
760
759
 
761
760
  ## 💡 Support & Contribution
762
761
 
763
- - **Repository**: [github.com/shriramsingh/promptTest](https://github.com/shriramsingh/promptTest)
764
- - **Issues & Requests**: [File an Issue](https://github.com/shriramsingh/promptTest/issues)
765
- - **License**: MIT
762
+ - **Community Repository**: [github.com/shriramsingh/prompttest-community](https://github.com/shriramsingh/prompttest-community)
763
+ - **Issues & Requests**: [File an Issue](https://github.com/shriramsingh/prompttest-community/issues)
764
+ - **License**: Business Source License 1.1 (BSL 1.1)