@aarwitz/tapp 0.17.9 → 0.17.11

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "tapp",
3
3
  "description": "Give Claude hands and eyes on iOS, Android, and web apps, with exploration, replayable flows, evidence, and deterministic CI gates.",
4
- "version": "0.17.9",
4
+ "version": "0.17.11",
5
5
  "author": {
6
6
  "name": "Aaron Horowitz",
7
7
  "url": "https://github.com/aarwitz"
@@ -24,7 +24,7 @@
24
24
  "command": "npx",
25
25
  "args": [
26
26
  "-y",
27
- "@aarwitz/tapp@0.17.9",
27
+ "@aarwitz/tapp@0.17.11",
28
28
  "mcp"
29
29
  ],
30
30
  "cwd": "${CLAUDE_PROJECT_DIR}"
@@ -387,7 +387,8 @@ class ExplorerTests: XCTestCase {
387
387
  app.coordinate(withNormalizedOffset: .zero).withOffset(CGVector(dx: x, dy: y)).tap()
388
388
  } else { status = "bad_args" }
389
389
  case "type":
390
- status = sessionType(cmd["text"] as? String ?? "", id: cmd["id"] as? String) ? "ok" : "not_found"
390
+ status = sessionType(cmd["text"] as? String ?? "", id: cmd["id"] as? String) ? "ok" : (lastTypeFailure.isEmpty ? "not_found" : "focus_failed")
391
+ if status == "focus_failed" { loginDetail = lastTypeFailure }
391
392
  case "swipe":
392
393
  switch (cmd["direction"] as? String ?? "up") {
393
394
  case "down": app.swipeDown(); case "left": app.swipeLeft(); case "right": app.swipeRight(); default: app.swipeUp()
@@ -578,10 +579,12 @@ class ExplorerTests: XCTestCase {
578
579
  // credential the server rejects with no visible cause.
579
580
  if let id = id, !id.isEmpty {
580
581
  for field in [app.textFields[id], app.secureTextFields[id], app.textViews[id]] where field.exists {
581
- replaceText(on: field, with: text); lastTypedInto = fieldDesc(field); return true
582
+ guard replaceText(on: field, with: text) else { return false }
583
+ lastTypedInto = fieldDesc(field); return true
582
584
  }
583
585
  if let field = resolveFieldByHint(id) {
584
- replaceText(on: field, with: text); lastTypedInto = fieldDesc(field); return true
586
+ guard replaceText(on: field, with: text) else { return false }
587
+ lastTypedInto = fieldDesc(field); return true
585
588
  }
586
589
  return false
587
590
  }
@@ -590,8 +593,7 @@ class ExplorerTests: XCTestCase {
590
593
  if app.keyboards.firstMatch.exists {
591
594
  if let focused = focusedField() {
592
595
  lastTypedInto = fieldDesc(focused)
593
- replaceText(on: focused, with: text)
594
- return true
596
+ return replaceText(on: focused, with: text)
595
597
  }
596
598
  app.typeText(text)
597
599
  lastTypedInto = "focused field"
@@ -599,7 +601,8 @@ class ExplorerTests: XCTestCase {
599
601
  }
600
602
  // Nothing focused and no usable id — last resort: the first text field or text view.
601
603
  for first in [app.textFields.firstMatch, app.textViews.firstMatch] where first.exists {
602
- replaceText(on: first, with: text); lastTypedInto = fieldDesc(first); return true
604
+ guard replaceText(on: first, with: text) else { return false }
605
+ lastTypedInto = fieldDesc(first); return true
603
606
  }
604
607
  return false
605
608
  }
@@ -809,7 +812,7 @@ class ExplorerTests: XCTestCase {
809
812
  if status == "fail" { detail = "could not tap ‘\(target)’" }
810
813
  case "type":
811
814
  status = sessionType(value, id: target.isEmpty ? nil : target) ? "pass" : "fail"
812
- if status == "fail" { detail = "no field ‘\(target)’ to type into" }
815
+ if status == "fail" { detail = lastTypeFailure.isEmpty ? "no field ‘\(target)’ to type into" : lastTypeFailure }
813
816
  case "login":
814
817
  let email = subst((step["email"] as? String) ?? "$TEST_EMAIL")
815
818
  let password = subst((step["password"] as? String) ?? "$TEST_PASSWORD")
@@ -1082,11 +1085,23 @@ class ExplorerTests: XCTestCase {
1082
1085
 
1083
1086
  let testEmail = resolve("OCQA_TEST_EMAIL")
1084
1087
  let testPassword = resolve("OCQA_TEST_PASSWORD")
1085
- if resolve("OCQA_CREDENTIALS_EXPLICIT") == "1" {
1086
- // Presence only: never print, persist, or expose credential values. Report rebuilding
1087
- // needs this durable marker to distinguish "not supplied" from "supplied but unused".
1088
+ // Presence only: never print, persist, or expose credential values. Report rebuilding
1089
+ // needs this durable marker to distinguish "not supplied" from "supplied but unused".
1090
+ // Decided by what actually reached the run config, not by a separate flag the run config
1091
+ // never carried (feedback #3: every iOS run reported credentialsProvided=false).
1092
+ if resolve("OCQA_CREDENTIALS_EXPLICIT") == "1" || !testEmail.isEmpty || !testPassword.isEmpty {
1088
1093
  print("OCQA_STATE:credentials_supplied")
1089
1094
  }
1095
+ // Capture conditions shared by every screenshot in this run, so a baseline diff across a
1096
+ // different device/scale is visibly a layout comparison (feedback #5: iOS reports carried
1097
+ // captureContext: null while the changelog promised it).
1098
+ do {
1099
+ let env = ProcessInfo.processInfo.environment
1100
+ let device = env["SIMULATOR_DEVICE_NAME"] ?? env["SIMULATOR_MODEL_IDENTIFIER"] ?? "iOS Simulator"
1101
+ let bounds = app.windows.firstMatch.exists ? app.windows.firstMatch.frame : app.frame
1102
+ let scale = UIScreen.main.scale
1103
+ print("OCQA_CONTEXT:{\"device\":\"\(escapeJSON(device))\",\"viewport\":{\"width\":\(Int(bounds.width.rounded())),\"height\":\(Int(bounds.height.rounded()))},\"deviceScaleFactor\":\(scale)}")
1104
+ }
1090
1105
 
1091
1106
  // --- Explicit login replay (config-driven): a recorded type/tap/wait sequence for custom
1092
1107
  // login UIs the heuristic preamble below can't parse. When configured it takes precedence. ---
@@ -4706,9 +4721,62 @@ class ExplorerTests: XCTestCase {
4706
4721
  print("OCQA_PROGRESS:{\"action\":\(action),\"max\":\(maxActions),\"states\":\(states)}")
4707
4722
  }
4708
4723
 
4709
- // Replace existing field contents to avoid repeatedly appending test text.
4710
- private func replaceText(on element: XCUIElement, with text: String) {
4711
- element.tap()
4724
+ /// Why the last replaceText could not type — surfaced as the step's detail so a flow says
4725
+ /// "the field never took keyboard focus" instead of dying on an XCTest assertion.
4726
+ private var lastTypeFailure = ""
4727
+
4728
+ /// Give a field keyboard focus and PROVE it before typing. `element.tap()` returning is not
4729
+ /// proof: SwiftUI fields inside a ScrollView regularly report an accessibility frame that
4730
+ /// lags the visible layout (content above them resolved and shifted), so the first tap lands
4731
+ /// on empty space, `typeText` then fails XCTest's "Neither element nor any descendant has
4732
+ /// keyboard focus" assertion and the whole run dies. Strategy: tap, poll for focus, and on a
4733
+ /// miss re-resolve the element (fresh snapshot) and try a coordinate tap at its centre, then
4734
+ /// a tap nudged to the top of the frame. Three attempts, ~4 s worst case.
4735
+ private func focusForTyping(_ element: XCUIElement) -> Bool {
4736
+ func focused() -> Bool { (element.value(forKey: "hasKeyboardFocus") as? Bool) ?? false }
4737
+ func settleUntilFocused() -> Bool {
4738
+ let deadline = Date().addingTimeInterval(1.2)
4739
+ repeat {
4740
+ if focused() { return true }
4741
+ Thread.sleep(forTimeInterval: 0.15)
4742
+ } while Date() < deadline
4743
+ return focused()
4744
+ }
4745
+ for attempt in 0..<3 {
4746
+ switch attempt {
4747
+ case 0:
4748
+ if element.isHittable { element.tap() } else { element.coordinate(withNormalizedOffset: CGVector(dx: 0.5, dy: 0.5)).tap() }
4749
+ case 1:
4750
+ element.coordinate(withNormalizedOffset: CGVector(dx: 0.5, dy: 0.5)).tap()
4751
+ default:
4752
+ // Frame may be stale by a few rows: tap just inside its top edge, which stays
4753
+ // within the visible control when content shifted up.
4754
+ element.coordinate(withNormalizedOffset: CGVector(dx: 0.35, dy: 0.2)).tap()
4755
+ }
4756
+ if settleUntilFocused() {
4757
+ if attempt > 0 { print("OCQA_STATE:type_focus_recovered attempt=\(attempt + 1)") }
4758
+ return true
4759
+ }
4760
+ let f = element.frame
4761
+ print("OCQA_STATE:type_focus_miss attempt=\(attempt + 1) frame=\(Int(f.minX)),\(Int(f.minY)),\(Int(f.width))x\(Int(f.height)) keyboard=\(app.keyboards.count) focusedElsewhere=\(focusedField().map { fieldDesc($0) } ?? "none")")
4762
+ }
4763
+ return false
4764
+ }
4765
+
4766
+ // Replace existing field contents to avoid repeatedly appending test text. Returns false
4767
+ // (and records `lastTypeFailure`) instead of letting XCTest abort the run when the field
4768
+ // cannot be focused.
4769
+ @discardableResult
4770
+ private func replaceText(on element: XCUIElement, with text: String) -> Bool {
4771
+ lastTypeFailure = ""
4772
+ guard focusForTyping(element) else {
4773
+ let f = element.frame
4774
+ lastTypeFailure = "‘\(fieldDesc(element))’ was found (frame \(Int(f.minX)),\(Int(f.minY)) \(Int(f.width))×\(Int(f.height))) but never took keyboard focus after 3 taps"
4775
+ + (app.keyboards.count == 0 ? " — no keyboard appeared" : "")
4776
+ + (focusedField().map { " — focus is on ‘\(fieldDesc($0))’" } ?? "")
4777
+ print("OCQA_STATE:type_focus_failed detail=\(escapeJSON(lastTypeFailure))")
4778
+ return false
4779
+ }
4712
4780
 
4713
4781
  if let existing = element.value as? String,
4714
4782
  !existing.isEmpty,
@@ -4718,5 +4786,6 @@ class ExplorerTests: XCTestCase {
4718
4786
  }
4719
4787
 
4720
4788
  element.typeText(text)
4789
+ return true
4721
4790
  }
4722
4791
  }
package/README.md CHANGED
@@ -199,6 +199,7 @@ one by hand:
199
199
 
200
200
  ```bash
201
201
  npx -y @aarwitz/tapp@latest flow example
202
+ npx -y @aarwitz/tapp@latest flow steps # the step vocabulary: target semantics and pass condition per step
202
203
  npx -y @aarwitz/tapp@latest flow validate .tapp/flows/smoke.yml
203
204
  npx -y @aarwitz/tapp@latest flow run .tapp/flows/smoke.yml
204
205
  ```
package/bin/tapp.js CHANGED
@@ -270,7 +270,7 @@ function safeCommandUsage(verb) {
270
270
  shot: "tapp shot [--out FILE]",
271
271
  apps: "tapp apps",
272
272
  build: "tapp build [repo] [--scheme NAME] [--configuration NAME]",
273
- flow: "tapp flow example\ntapp flow validate FILE [--platform PLATFORM] [--map FILE]\ntapp flow run FILE [--actor NAME] [--email VALUE] [--password VALUE] [--device \"iPhone 13\"] [--viewport 390x844]\n Exit codes: 0 replay passed · 1 replay failed · 2 infrastructure/usage error",
273
+ flow: "tapp flow example\ntapp flow steps [--json]\ntapp flow validate FILE [--platform PLATFORM] [--map FILE]\ntapp flow run FILE [--actor NAME] [--email VALUE] [--password VALUE] [--device \"iPhone 13\"] [--viewport 390x844]\n Exit codes: 0 replay passed · 1 replay failed · 2 infrastructure/usage error",
274
274
  task: "tapp task validate FILE [--platform PLATFORM] [--map FILE]\ntapp task compile FILE --platform PLATFORM [--inputs JSON] [--out FILE]\ntapp task run FILE --platform PLATFORM [--url URL|--bundle-id ID|--app-id ID] [--inputs JSON]",
275
275
  contract: "tapp contract validate FILE [--platform PLATFORM] [--map FILE]\ntapp contract compile FILE --platform PLATFORM [--out FILE]\ntapp contract run FILE --platform PLATFORM [--url URL|--bundle-id ID|--app-id ID]\n Exit codes: 0 contract held · 1 contract failed · 2 infrastructure/usage error",
276
276
  scenario: "tapp scenario validate FILE [--project-dir DIR]\ntapp scenario run FILE --platform web --url URL [--project-dir DIR]\n Exit codes: 0 scenario passed · 1 scenario failed · 2 infrastructure/usage error",
@@ -1133,8 +1133,16 @@ switch (command) {
1133
1133
  console.log(`# Tapp Flow — deterministic, keyless replay\nname: sign-in-smoke\nplatform: web\nurl: https://example.test/login\nsteps:\n - login:\n email: $TEST_EMAIL\n password: $TEST_PASSWORD\n - wait_for: Dashboard\n - assert_screen: Dashboard\n`);
1134
1134
  break;
1135
1135
  }
1136
+ if (verb === "steps") {
1137
+ const { FLOW_ACTIONS } = await import(path.join(packageRoot, "mcp-server", "src", "flow-runtime.js"));
1138
+ if (flags.json === true) { console.log(JSON.stringify(FLOW_ACTIONS, null, 2)); break; }
1139
+ console.log("Flow step vocabulary — identical on ios, android, and web. Anything else fails `tapp flow validate`.\n");
1140
+ for (const a of FLOW_ACTIONS) console.log(` ${a.action.padEnd(14)} target: ${a.target}\n ${"".padEnd(14)} passes: ${a.passes}\n`);
1141
+ console.log("Targets are labels / accessibility ids / visible text, never coordinates. `assert_screen` checks the detected screen TITLE; use `assert_exists` for \"this text is on screen\".");
1142
+ break;
1143
+ }
1136
1144
  if (!["run", "validate"].includes(verb) || !flowPath) {
1137
- console.error("usage: tapp flow example\n tapp flow run <flow.yml> [--platform ios|android|web] [--actor NAME] [--email VALUE] [--password VALUE] [--url URL] [--app-id ID] [--apk FILE] [--serial ID]\n tapp flow validate <flow.yml>");
1145
+ console.error("usage: tapp flow example\n tapp flow steps [--json]\n tapp flow run <flow.yml> [--platform ios|android|web] [--actor NAME] [--email VALUE] [--password VALUE] [--url URL] [--app-id ID] [--apk FILE] [--serial ID]\n tapp flow validate <flow.yml>");
1138
1146
  process.exit(2);
1139
1147
  }
1140
1148
  const absolute = path.resolve(flowPath);
@@ -1142,7 +1150,7 @@ switch (command) {
1142
1150
  console.error(`❌ Flow not found: ${absolute}`);
1143
1151
  process.exit(2);
1144
1152
  }
1145
- const { loadFlowFile } = await import(path.join(packageRoot, "mcp-server", "src", "flow-runtime.js"));
1153
+ const { loadFlowFile, validateFlowSteps } = await import(path.join(packageRoot, "mcp-server", "src", "flow-runtime.js"));
1146
1154
  let flow;
1147
1155
  try { flow = loadFlowFile(absolute); } catch (error) {
1148
1156
  console.error(`❌ Invalid Flow: ${error.message}`);
@@ -1157,6 +1165,13 @@ switch (command) {
1157
1165
  console.error(`❌ Unsupported Flow platform: ${platform}`);
1158
1166
  process.exit(2);
1159
1167
  }
1168
+ // Static step checks run before validate AND run: a Flow outside the vocabulary (or a
1169
+ // coordinate tap) can never replay, so it must not reach a simulator (feedback #2).
1170
+ const stepErrors = validateFlowSteps(flow, platform);
1171
+ if (stepErrors.length) {
1172
+ console.error(`❌ Invalid ${platform} Flow — ${flow.name || path.basename(absolute)}:\n${stepErrors.map((e) => ` • ${e}`).join("\n")}\n Vocabulary: tapp flow steps`);
1173
+ process.exit(2);
1174
+ }
1160
1175
  if (verb === "validate") {
1161
1176
  const taskCount = Array.isArray(flow.taskPlan) ? flow.taskPlan.length : 0;
1162
1177
  console.log(`✅ Valid ${platform} Flow — ${flow.name} (${flow.steps.length} deterministic steps${taskCount ? ` compiled from ${taskCount} Task call${taskCount === 1 ? "" : "s"}` : ""})`);
package/docs/scenarios.md CHANGED
@@ -74,7 +74,7 @@ tapp ci --platform web --url http://127.0.0.1:4180 \
74
74
  GitHub Action:
75
75
 
76
76
  ```yaml
77
- - uses: aarwitz/tapp@v0.17.9 # or pin the reviewed release commit SHA
77
+ - uses: aarwitz/tapp@v0.17.10 # or pin the reviewed release commit SHA
78
78
  with:
79
79
  platform: web
80
80
  url: http://127.0.0.1:4180
@@ -86,12 +86,19 @@ export function ghStatus(ghBin = process.env.TAPP_GH_BIN || "gh") {
86
86
  }
87
87
 
88
88
  export function submitFeedbackViaGh(issue, ghBin = process.env.TAPP_GH_BIN || "gh") {
89
- const r = spawnSync(ghBin, [
89
+ // Labels can only be set by accounts with triage rights on the repo. Anyone else (which is
90
+ // every real user) gets the issue filed without them; the footer already carries the kind
91
+ // and the maintainer applies labels on triage. Try with labels first, then without.
92
+ const attempt = (withLabels) => spawnSync(ghBin, [
90
93
  "issue", "create", "--repo", FEEDBACK_REPO,
91
- "--title", issue.title, "--body-file", "-", "--label", issue.labels.join(","),
94
+ "--title", issue.title, "--body-file", "-",
95
+ ...(withLabels ? ["--label", issue.labels.join(",")] : []),
92
96
  ], { encoding: "utf8", input: issue.body });
97
+ let r = attempt(true);
98
+ let labelsApplied = r.status === 0;
99
+ if (r.status !== 0 && /label/i.test(`${r.stderr || ""}${r.stdout || ""}`)) { r = attempt(false); labelsApplied = false; }
93
100
  const out = (r.stdout || "").trim();
94
101
  const err = (r.stderr || "").trim();
95
102
  const url = (out.match(/https:\/\/github\.com\/\S+/) || [])[0] || null;
96
- return { ok: r.status === 0 && Boolean(url), url, detail: r.status === 0 ? out : (err || out) };
103
+ return { ok: r.status === 0 && Boolean(url), url, labelsApplied, detail: r.status === 0 ? out : (err || out) };
97
104
  }
@@ -36,6 +36,60 @@ export function normalizeFlowStep(raw) {
36
36
  return { action: key.toLowerCase(), target: body === true ? "" : String(body ?? ""), value: body === true ? "" : String(body ?? ""), params: {}, ...(raw.__tappTask?.name ? { task: raw.__tappTask.name } : {}) };
37
37
  }
38
38
 
39
+ // The committed Flow step vocabulary. Every driver (XCUITest, Android, browser) implements exactly
40
+ // this table; `tapp flow steps` prints it and `tapp flow validate` rejects anything outside it, so
41
+ // a Flow that validates can actually replay (feedback #2: coordinate taps and `click:` used to
42
+ // validate and then fail at runtime).
43
+ export const FLOW_ACTIONS = Object.freeze([
44
+ { action: "tap", target: "a visible label / accessibility id", passes: "the control was found and tapped; coordinates are not accepted — use the session's tap-by-point to learn the label" },
45
+ { action: "type", target: "{field, value}", passes: "the field was found and now holds the value ($TEST_EMAIL/$TEST_PASSWORD substitute)" },
46
+ { action: "login", target: "{email, password} (defaults to $TEST_EMAIL/$TEST_PASSWORD)", passes: "credentials were entered and submitted and the login form went away" },
47
+ { action: "swipe", target: "up | down | left | right", passes: "the gesture was performed" },
48
+ { action: "back", target: "(none)", passes: "the platform back navigation was performed" },
49
+ { action: "wait", target: "milliseconds (fixed pause; prefer wait_for)", passes: "always" },
50
+ { action: "wait_for", target: "label / text (+ timeoutMs)", passes: "the element appeared before the timeout" },
51
+ { action: "assert_screen", target: "the detected SCREEN TITLE (navigation bar / heading), not arbitrary text", passes: "the current screen's title equals the target" },
52
+ { action: "assert_exists", target: "label / text", passes: "an element with that text or id is present" },
53
+ { action: "assert_absent", target: "label / text", passes: "no element with that text or id is present" },
54
+ { action: "assert_text", target: "{of, contains}", passes: "the element's text contains the substring" },
55
+ { action: "assert_ai", target: "a natural-language expectation", passes: "the vision judge agrees (needs ANTHROPIC_API_KEY; advisory)" },
56
+ ]);
57
+ const FLOW_ACTION_NAMES = new Set(FLOW_ACTIONS.map((a) => a.action));
58
+ const ALIASES = { click: "tap", press: "tap", fill: "type", input: "type", sleep: "wait", wait_for_text: "wait_for", assert_visible: "assert_exists", expect: "assert_exists" };
59
+
60
+ // Static checks a Flow must pass before any runtime is launched. Returns human-readable errors;
61
+ // an empty array means every step is in the vocabulary and shaped so a driver can execute it.
62
+ export function validateFlowSteps(flow, platform = inferFlowPlatform(flow)) {
63
+ const errors = [];
64
+ const steps = Array.isArray(flow?.steps) ? flow.steps : [];
65
+ steps.forEach((raw, i) => {
66
+ const step = normalizeFlowStep(raw);
67
+ const n = i + 1;
68
+ if (step.action === "noop") { errors.push(`step ${n}: empty step`); return; }
69
+ if (!FLOW_ACTION_NAMES.has(step.action)) {
70
+ const alias = ALIASES[step.action];
71
+ errors.push(`step ${n}: unknown action '${step.action}'${alias ? ` — did you mean '${alias}'? (web flows use tap:, not click:)` : ""}; run \`tapp flow steps\` for the vocabulary`);
72
+ return;
73
+ }
74
+ if (step.action === "tap" && /^\s*-?\d+(\.\d+)?\s*,\s*-?\d+(\.\d+)?\s*$/.test(step.target)) {
75
+ errors.push(`step ${n}: tap target '${step.target.trim()}' is a coordinate; Flow taps are label-only on ${platform} (use tapp_session_act tap {x,y} to learn the label, then record it)`);
76
+ }
77
+ if (["tap", "wait_for", "assert_screen", "assert_exists", "assert_absent"].includes(step.action) && !step.target.trim()) {
78
+ errors.push(`step ${n}: ${step.action} needs a target`);
79
+ }
80
+ if (step.action === "type" && (!step.target.trim() || !("value" in (step.params || {})))) {
81
+ errors.push(`step ${n}: type needs {field, value}`);
82
+ }
83
+ if (step.action === "assert_text" && (!step.target.trim() || !step.value)) {
84
+ errors.push(`step ${n}: assert_text needs {of, contains}`);
85
+ }
86
+ if (step.action === "swipe" && step.target && !["up", "down", "left", "right"].includes(step.target.trim().toLowerCase())) {
87
+ errors.push(`step ${n}: swipe direction must be up|down|left|right`);
88
+ }
89
+ });
90
+ return errors;
91
+ }
92
+
39
93
  export function flowVariables(flow, overrides = {}) {
40
94
  return {
41
95
  TEST_EMAIL: process.env.OCQA_TEST_EMAIL || "test@example.com",
@@ -907,10 +907,39 @@ function recordStep(cmd, result) {
907
907
  if (newScreen) activeSession.lastScreen = newScreen;
908
908
  }
909
909
 
910
+ // Argument shapes tapp_session_act accepts, per action. A malformed call is answered here with
911
+ // the accepted shape and never reaches the driver, so it can neither time out nor disturb the
912
+ // session (feedback #7: `{action:"wait", seconds:3}` used to wait on an empty target).
913
+ const SESSION_ACT_ARGS = Object.freeze({
914
+ tap: "{id: <label|accessibility id>} or {x, y} (points)",
915
+ type: "{id|label: <field>, text: <value>}",
916
+ wait: "{text|id: <label to wait for>, timeoutMs?: <default 5000, max 60000>} — there is no fixed sleep; wait for something",
917
+ login: "{email?, password?} (defaults to the session's credentials)",
918
+ swipe: "{direction: up|down|left|right}",
919
+ back: "{}",
920
+ tree: "{verbose?: true}",
921
+ screenshot: "{label?}",
922
+ });
923
+ export function sessionActUsageError(cmd = {}) {
924
+ const action = String(cmd.action || "");
925
+ if (!SESSION_ACT_ARGS[action]) return `Unknown action '${action || "(none)"}'. Accepted: ${Object.keys(SESSION_ACT_ARGS).join(", ")}.`;
926
+ const has = (k) => cmd[k] !== undefined && cmd[k] !== null && String(cmd[k]).trim() !== "";
927
+ const bad = (why) => `${why}. ${action} takes ${SESSION_ACT_ARGS[action]}.`;
928
+ if (action === "wait" && !has("text") && !has("id")) return bad(`wait needs a target${cmd.seconds !== undefined || cmd.ms !== undefined ? " (seconds/ms are not arguments)" : ""}`);
929
+ if (action === "wait" && cmd.timeoutMs !== undefined && !(Number.isFinite(Number(cmd.timeoutMs)) && Number(cmd.timeoutMs) > 0)) return bad("timeoutMs must be a positive number of milliseconds");
930
+ if (action === "tap" && !has("id") && !has("label") && !(Number.isFinite(cmd.x) && Number.isFinite(cmd.y))) return bad("tap needs an id/label or both x and y");
931
+ if (action === "type" && !has("id") && !has("label")) return bad("type needs the field's id/label");
932
+ if (action === "type" && cmd.text === undefined) return bad("type needs text");
933
+ if (action === "swipe" && cmd.direction !== undefined && !["up", "down", "left", "right"].includes(String(cmd.direction))) return bad("direction must be up|down|left|right");
934
+ return null;
935
+ }
936
+
910
937
  async function sessionAct(cmd) {
911
938
  const startedAt = Date.now();
912
939
  const done = (result) => ({ ...result, durationMs: Date.now() - startedAt });
913
940
  if (!activeSession || activeSession.ended) return done({ error: "No active session. Call tapp_session_start first." });
941
+ const usage = sessionActUsageError(cmd);
942
+ if (usage) return done({ status: "usage", detail: usage, ...treeSnapshot(), recordedSteps: activeSession.recording.length });
914
943
  let coordinateResolvedTarget = "";
915
944
  if (cmd.action === "tap" && !cmd.id && Number.isFinite(cmd.x) && Number.isFinite(cmd.y)) {
916
945
  coordinateResolvedTarget = semanticTargetAtPoint(activeSession.latestTree?.elements, cmd.x, cmd.y);
@@ -298,6 +298,7 @@ export function buildQaReport(markersFilePath, { platform = "ios", target = null
298
298
  // 40-action campaign. Drivers signal the real cause in COMPLETE.stop; captures from drivers
299
299
  // that predate the field keep the old inference.
300
300
  const driverStop = base.complete && typeof base.complete === "object" ? base.complete.stop : null;
301
+ const nativeOutcome = base.complete && typeof base.complete === "object" && typeof base.complete.outcome === "string" ? base.complete.outcome : null;
301
302
  const stopReason = unexercisedLoginWall ? (credentialsProvided ? "login-wall-credentials-unused" : "login-wall-no-credentials")
302
303
  : timeBudgetExhausted ? "time-budget-exhausted"
303
304
  : !coverageFloorMet ? "coverage-floor-not-met"
@@ -306,6 +307,11 @@ export function buildQaReport(markersFilePath, { platform = "ios", target = null
306
307
  // Any other driver-signalled cause (navigation-trap, app-crashed, stuck-no-progress …)
307
308
  // passes through verbatim: "completed" is ONLY the exhausted action budget.
308
309
  : driverStop && driverStop !== "action-budget" && driverStop !== "time-budget" ? String(driverStop)
310
+ // XCUITest predates COMPLETE.stop and signals the cause through COMPLETE.outcome instead
311
+ // (feedback #5: a run whose marker said limited_surface must not read as "completed").
312
+ : nativeOutcome === "limited_surface" ? "limited-surface"
313
+ : nativeOutcome === "timeout" ? "time-budget-exhausted"
314
+ : nativeOutcome && nativeOutcome.startsWith("crash") ? "app-crashed"
309
315
  : "completed";
310
316
 
311
317
  const headline = timeBudgetExhausted
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@aarwitz/tapp",
3
- "version": "0.17.9",
3
+ "version": "0.17.11",
4
4
  "mcpName": "io.github.aarwitz/tapp",
5
5
  "description": "Let coding agents verify UI changes on real iOS, Android, and web surfaces, then enforce reviewed proof in deterministic CI.",
6
6
  "license": "MIT",
@@ -89,7 +89,7 @@
89
89
  "mobile"
90
90
  ],
91
91
  "scripts": {
92
- "test": "node --test tests/report.test.js tests/regression.test.js tests/engine.test.js tests/project-config.test.js tests/application-model.test.js tests/ui-map.test.js tests/focused-navigation.test.js tests/task-runtime.test.js tests/release-contract.test.js tests/pr-selection.test.js tests/flow-runtime.test.js tests/web-explorer.test.js tests/web-session.test.js tests/web-link-audit.test.js tests/web-flow.test.js tests/scenario-runtime.test.js tests/android-driver.test.js tests/android-explorer.test.js tests/android-flow.test.js tests/android-primitives-protocol.test.js tests/managed-web.test.js tests/product-operations.test.js tests/browser-product.test.js tests/browser-onboarding.test.js tests/managed-operation.test.js tests/cloud-runner.test.js tests/ci-setup.test.js tests/ci-install.test.js tests/cli.test.js tests/mcp-workspace.test.js tests/action.test.js tests/package-surface.test.js tests/agent-surface.test.js tests/presentation-contract.test.js tests/landing-brand.test.js tests/ci-gate.test.js tests/ci-report.test.js tests/desktop-protocol.test.js tests/ios-flow-protocol.test.js vscode-extension/test/bridge.test.js",
92
+ "test": "node --test tests/report.test.js tests/regression.test.js tests/engine.test.js tests/project-config.test.js tests/application-model.test.js tests/ui-map.test.js tests/focused-navigation.test.js tests/task-runtime.test.js tests/release-contract.test.js tests/pr-selection.test.js tests/flow-runtime.test.js tests/web-explorer.test.js tests/web-session.test.js tests/web-link-audit.test.js tests/web-flow.test.js tests/scenario-runtime.test.js tests/android-driver.test.js tests/android-explorer.test.js tests/android-flow.test.js tests/android-primitives-protocol.test.js tests/managed-web.test.js tests/product-operations.test.js tests/browser-product.test.js tests/browser-onboarding.test.js tests/managed-operation.test.js tests/cloud-runner.test.js tests/ci-setup.test.js tests/ci-install.test.js tests/cli.test.js tests/mcp-workspace.test.js tests/action.test.js tests/package-surface.test.js tests/agent-surface.test.js tests/presentation-contract.test.js tests/landing-brand.test.js tests/ci-gate.test.js tests/ci-report.test.js tests/desktop-protocol.test.js tests/ios-flow-protocol.test.js tests/feedback-triage-0919.test.js vscode-extension/test/bridge.test.js",
93
93
  "test:browser-journey": "node --test tests/browser-journey.test.js",
94
94
  "test:browser-native": "TAPP_RUN_NATIVE_BROWSER=1 node --test tests/browser-native-journey.test.js"
95
95
  }
@@ -63,12 +63,36 @@ def raw_json(path):
63
63
  print(json.dumps(load_flow(path) or {}))
64
64
 
65
65
 
66
+ ABORT_PATTERNS = [
67
+ r"Failed to synthesize event: [^\n]*",
68
+ r"Neither element nor any descendant has keyboard focus[^\n]*",
69
+ r"Test Case '[^']*' failed \([^)]*\)",
70
+ r"Error Domain=[^\n]*",
71
+ r"Unable to (?:find|launch|boot)[^\n]*",
72
+ r"App state is [^\n]*",
73
+ r"Timed out [^\n]*",
74
+ r"Testing failed:[^\n]*",
75
+ r"error: [^\n]*",
76
+ r"\*\* TEST (?:EXECUTE )?FAILED \*\*",
77
+ ]
78
+
79
+
80
+ def harness_abort_reason(log):
81
+ """First XCTest/xcodebuild line that explains why a run ended before step 1."""
82
+ for pat in ABORT_PATTERNS:
83
+ m = re.search(pat, log)
84
+ if m:
85
+ return m.group(0).strip()[:300]
86
+ return None
87
+
88
+
66
89
  def report(path, as_json=False):
67
90
  log = open(path, encoding="utf-8", errors="replace").read()
68
91
  steps = []
69
92
  result = None
70
93
  name = "flow"
71
94
  kind = "flow"
95
+ declared_total = None
72
96
  for line in log.splitlines():
73
97
  line = line.strip()
74
98
  if line.startswith("OCQA_FLOW_STEP:"):
@@ -83,6 +107,9 @@ def report(path, as_json=False):
83
107
  km = re.search(r"\bkind=([^ ]+)", line)
84
108
  if km:
85
109
  kind = km.group(1)
110
+ tm = re.search(r"\btotal=(\d+)", line)
111
+ if tm:
112
+ declared_total = int(tm.group(1))
86
113
  elif line.startswith("OCQA_FLOW_RESULT:{"):
87
114
  try:
88
115
  result = json.loads(line[len("OCQA_FLOW_RESULT:"):])
@@ -97,8 +124,29 @@ def report(path, as_json=False):
97
124
  executed = (result or {}).get("executed", len(steps))
98
125
  passed_steps = sum(1 for s in steps if s.get("status") == "pass")
99
126
 
127
+ # A run that died before step 1 used to print `0 passed · 0 failed · 0/0 executed` and
128
+ # nothing else; the XCTest reason lived only in flow.log (feedback #1). Surface it as the
129
+ # failure of the first step so the scoreboard says why instead of looking like a crash.
130
+ abort_reason = None
131
+ if not steps and not passed:
132
+ abort_reason = harness_abort_reason(log)
133
+ steps.append({"index": 1, "action": "harness", "target": "", "status": "fail",
134
+ "detail": abort_reason or "the harness exited before the first step; see flow.log"})
135
+ failed = max(failed, 1)
136
+ elif result is None and steps:
137
+ # The harness died MID-run (an XCTest assertion, the test time budget, a crash): there is
138
+ # no final OCQA_FLOW_RESULT line. This used to be reported as "PASSED 16/16" because
139
+ # `total` silently became the number of steps that happened to run before the death.
140
+ abort_reason = harness_abort_reason(log)
141
+ total = declared_total or total
142
+ steps.append({"index": len(steps) + 1, "action": "harness", "target": "", "status": "fail",
143
+ "detail": (abort_reason or "the harness exited") + f" — run ended after step {executed} of {total}"})
144
+ failed += 1
145
+ passed = False
146
+
100
147
  if as_json:
101
- print(json.dumps({"name": name, "kind": kind, "passed": passed, "total": total, "executed": executed, "failed": failed, "steps": steps}))
148
+ print(json.dumps({"name": name, "kind": kind, "passed": passed, "total": total, "executed": executed, "failed": failed,
149
+ **({"abortReason": abort_reason} if abort_reason else {}), "steps": steps}))
102
150
  return 0 if passed else 1
103
151
 
104
152
  icon = {"pass": "✅", "fail": "❌", "skip": "⚪️"}
@@ -184,7 +184,8 @@ run_harness_test() {
184
184
  "OCQA_MAX_ACTIONS": "$max_actions",
185
185
  "OCQA_TIMEOUT_SECONDS": "$timeout_secs",
186
186
  "OCQA_TEST_EMAIL": "${OCQA_TEST_EMAIL:-}",
187
- "OCQA_TEST_PASSWORD": "${OCQA_TEST_PASSWORD:-}"$interactive_line$overrides_line$launch_args_line$launch_env_line$login_steps_line$pr_target_line$visual_ready_line$recording_started_line
187
+ "OCQA_TEST_PASSWORD": "${OCQA_TEST_PASSWORD:-}",
188
+ "OCQA_CREDENTIALS_EXPLICIT": "${OCQA_CREDENTIALS_EXPLICIT:-}"$interactive_line$overrides_line$launch_args_line$launch_env_line$login_steps_line$pr_target_line$visual_ready_line$recording_started_line
188
189
  }
189
190
  CONF
190
191
 
@@ -101,6 +101,8 @@ ask the user rather than pretending the explored surface was complete.
101
101
  Flow YAML belongs under `.tapp/flows/` and can replay without a model or API key:
102
102
 
103
103
  ```bash
104
+ npx -y @aarwitz/tapp@latest flow steps # vocabulary: assert_screen = screen TITLE, assert_exists = text present; taps are label-only
105
+ npx -y @aarwitz/tapp@latest flow validate .tapp/flows/smoke.yml
104
106
  npx -y @aarwitz/tapp@latest flow run .tapp/flows/smoke.yml
105
107
  npx -y @aarwitz/tapp@latest ci
106
108
  ```