@aarwitz/tapp 0.17.11 → 0.17.13

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "tapp",
3
3
  "description": "Give Claude hands and eyes on iOS, Android, and web apps, with exploration, replayable flows, evidence, and deterministic CI gates.",
4
- "version": "0.17.11",
4
+ "version": "0.17.13",
5
5
  "author": {
6
6
  "name": "Aaron Horowitz",
7
7
  "url": "https://github.com/aarwitz"
@@ -24,7 +24,7 @@
24
24
  "command": "npx",
25
25
  "args": [
26
26
  "-y",
27
- "@aarwitz/tapp@0.17.11",
27
+ "@aarwitz/tapp@0.17.13",
28
28
  "mcp"
29
29
  ],
30
30
  "cwd": "${CLAUDE_PROJECT_DIR}"
@@ -552,6 +552,21 @@ class ExplorerTests: XCTestCase {
552
552
  }
553
553
  }
554
554
 
555
+ /// What WAS on screen when a wait/assert missed — so a failure report points at the fix
556
+ /// (wrong screen? renamed label? error state?) without a separate `tapp tree` run (#20/#22).
557
+ private func visibleLabelsHint(limit: Int = 8) -> String {
558
+ let labels = readUITree(app)
559
+ .filter { isStaticTextType($0.type) || isInteractable($0.type) }
560
+ .map { normalizeVisibleText($0.label) }
561
+ .filter { $0.count >= 2 }
562
+ var seen = Set<String>(); var top: [String] = []
563
+ for label in labels where !seen.contains(label) {
564
+ seen.insert(label); top.append(label)
565
+ if top.count >= limit { break }
566
+ }
567
+ return top.isEmpty ? "" : " — visible: \(top.joined(separator: " · "))"
568
+ }
569
+
555
570
  private func sessionWaitFor(_ target: String, timeoutMs: Int) -> Bool {
556
571
  guard !target.isEmpty else { return false }
557
572
  let deadline = Date().addingTimeInterval(Double(timeoutMs) / 1000.0)
@@ -767,6 +782,10 @@ class ExplorerTests: XCTestCase {
767
782
  print("OCQA_FLOW_RESULT:{\"passed\":false,\"total\":0,\"failed\":0,\"error\":\"no OCQA_FLOW_JSON with steps\"}")
768
783
  return
769
784
  }
785
+ // Per-flow wait default (field issue #20): a splash that prefetches for ~8s makes the
786
+ // fixed 6s wait_for fail on a healthy app. Steps may still override individually.
787
+ let flowDefaultTimeoutMs = (flow["timeoutMs"] as? Int) ?? (flow["timeout"] as? Int) ?? 6000
788
+
770
789
  // Variable substitution: $TEST_EMAIL/$TEST_PASSWORD from creds, plus any OCQA_FLOW_VARS.
771
790
  var vars: [String: String] = ["TEST_EMAIL": resolve("OCQA_TEST_EMAIL", fallback: "test@example.com"),
772
791
  "TEST_PASSWORD": resolve("OCQA_TEST_PASSWORD", fallback: "TestPass123!")]
@@ -802,7 +821,7 @@ class ExplorerTests: XCTestCase {
802
821
  let (action, step) = normalizeFlowStep(raw)
803
822
  let target = subst((step["target"] as? String) ?? "")
804
823
  let value = subst((step["value"] as? String) ?? "")
805
- let timeoutMs = (step["timeoutMs"] as? Int) ?? 6000
824
+ let timeoutMs = (step["timeoutMs"] as? Int) ?? (step["timeout"] as? Int) ?? flowDefaultTimeoutMs
806
825
  var status = "pass"
807
826
  var detail = ""
808
827
 
@@ -818,7 +837,14 @@ class ExplorerTests: XCTestCase {
818
837
  let password = subst((step["password"] as? String) ?? "$TEST_PASSWORD")
819
838
  let result = sessionLogin(email: email, password: password)
820
839
  status = result.status == "ok" ? "pass" : "fail"
821
- if status == "fail" { detail = result.detail.isEmpty ? result.status : result.detail }
840
+ if status == "fail" {
841
+ detail = result.detail.isEmpty ? result.status : result.detail
842
+ // Firebase's "error accessing the keychain" on a simulator is stale keychain
843
+ // state, not bad credentials — name the one-line fix (field issue #21).
844
+ if detail.lowercased().contains("accessing the keychain") {
845
+ detail += " — simulator keychain is stale; `xcrun simctl erase <udid>` (or Device ▸ Erase All Content and Settings) usually fixes Firebase keychain errors"
846
+ }
847
+ }
822
848
  case "swipe":
823
849
  switch target.lowercased() { case "down": app.swipeDown(); case "left": app.swipeLeft(); case "right": app.swipeRight(); default: app.swipeUp() }
824
850
  case "back":
@@ -827,7 +853,7 @@ class ExplorerTests: XCTestCase {
827
853
  Thread.sleep(forTimeInterval: Double(timeoutMs) / 1000.0)
828
854
  case "wait_for":
829
855
  status = sessionWaitFor(target, timeoutMs: timeoutMs) ? "pass" : "fail"
830
- if status == "fail" { detail = "‘\(target)’ never appeared within \(timeoutMs)ms" }
856
+ if status == "fail" { detail = "‘\(target)’ never appeared within \(timeoutMs)ms\(visibleLabelsHint())" }
831
857
  case "assert_screen":
832
858
  let ok = pollUntil(timeoutMs: timeoutMs) { (detectTitle(readUITree(app)) ?? "").caseInsensitiveCompare(value.isEmpty ? target : value) == .orderedSame }
833
859
  status = ok ? "pass" : "fail"
@@ -835,7 +861,7 @@ class ExplorerTests: XCTestCase {
835
861
  case "assert_exists":
836
862
  let ok = sessionWaitFor(target, timeoutMs: timeoutMs)
837
863
  status = ok ? "pass" : "fail"
838
- if !ok { detail = "‘\(target)’ not found" }
864
+ if !ok { detail = "‘\(target)’ not found\(visibleLabelsHint())" }
839
865
  case "assert_absent":
840
866
  waitForUIStability(timeout: 1.5)
841
867
  let present = elementPresent(target)
@@ -1110,8 +1136,24 @@ class ExplorerTests: XCTestCase {
1110
1136
  // --- Login preamble: if credentials are provided and login fields are visible, log in first ---
1111
1137
  if !ranExplicitLogin, !testEmail.isEmpty, !testPassword.isEmpty {
1112
1138
  waitForUIStability(timeout: 2.0) // let app fully settle
1113
- let allTextFields = app.textFields.allElementsBoundByIndex.filter { $0.exists && $0.frame.width > 0 }
1114
- let allSecureFields = app.secureTextFields.allElementsBoundByIndex.filter { $0.exists && $0.frame.width > 0 }
1139
+ var allTextFields = app.textFields.allElementsBoundByIndex.filter { $0.exists && $0.frame.width > 0 }
1140
+ var allSecureFields = app.secureTextFields.allElementsBoundByIndex.filter { $0.exists && $0.frame.width > 0 }
1141
+
1142
+ if allSecureFields.count > 1 {
1143
+ let existingAccountLabels = ["Back to login", "Back to Login", "Already have an account?", "Sign in instead", "Log in instead"]
1144
+ for label in existingAccountLabels {
1145
+ let control = app.buttons[label].exists ? app.buttons[label] : app.staticTexts[label]
1146
+ if control.exists && control.isHittable {
1147
+ control.tap()
1148
+ print("OCQA_STATE:login_preamble_switched_from_signup")
1149
+ Thread.sleep(forTimeInterval: 0.8)
1150
+ waitForUIStability(timeout: 2.0)
1151
+ allTextFields = app.textFields.allElementsBoundByIndex.filter { $0.exists && $0.frame.width > 0 }
1152
+ allSecureFields = app.secureTextFields.allElementsBoundByIndex.filter { $0.exists && $0.frame.width > 0 }
1153
+ break
1154
+ }
1155
+ }
1156
+ }
1115
1157
  print("OCQA_STATE:login_preamble_fields textFields=\(allTextFields.count) secureFields=\(allSecureFields.count)")
1116
1158
 
1117
1159
  let emailField = allTextFields.first { f in
package/bin/tapp.js CHANGED
@@ -860,6 +860,14 @@ switch (command) {
860
860
  if (!snap.settled) console.error("⚠️ Page still showed a loading or changing state when the bounded wait ended.");
861
861
  } catch (error) {
862
862
  console.error(`❌ ${error.message || String(error)}`);
863
+ const evidence = error.timeoutEvidence;
864
+ if (evidence?.image) {
865
+ const out = typeof flags.out === "string" ? path.resolve(flags.out) : path.join(tappHome, "shots", `web-timeout-${Date.now()}.png`);
866
+ fs.mkdirSync(path.dirname(out), { recursive: true });
867
+ fs.writeFileSync(out, evidence.image);
868
+ console.error(`📸 Screenshot at timeout: ${out}`);
869
+ }
870
+ if (evidence?.visible?.length) console.error(`👀 Visible instead: ${evidence.visible.join(" · ")}`);
863
871
  process.exit(1);
864
872
  }
865
873
  break;
@@ -1473,6 +1481,13 @@ switch (command) {
1473
1481
  const python = run("python3", ["--version"]);
1474
1482
  report.python3 = { ok: python.code === 0, version: python.code === 0 ? python.stdout : null };
1475
1483
  python.code === 0 ? sayOk("python3", `${python.stdout} (used by Flows)`) : sayBad("python3", "not found — Flow replay needs python3 + pyyaml (everything else works)");
1484
+ // Hosted macOS runners ship python3 WITHOUT PyYAML (field issue #16) — check the module,
1485
+ // not just the interpreter, so `doctor` catches it before a Flow dies mid-CI.
1486
+ if (python.code === 0) {
1487
+ const pyyaml = run("python3", ["-c", "import yaml"]);
1488
+ report.pyyaml = { ok: pyyaml.code === 0 };
1489
+ pyyaml.code === 0 ? sayOk("PyYAML", "importable (YAML Flow replay)") : sayBad("PyYAML", "missing — `python3 -m pip install pyyaml` (YAML Flow replay needs it; JSON Flows work without)");
1490
+ }
1476
1491
 
1477
1492
  const { storagePreflight } = await import(path.join(packageRoot, "mcp-server", "src", "environment-preflight.js"));
1478
1493
  const storage = storagePreflight(tappHome);
@@ -6,7 +6,8 @@
6
6
  // summary (when GITHUB_STEP_SUMMARY is set), a machine-readable JSON report, and — the point —
7
7
  // an exit code CI can gate a merge on.
8
8
  //
9
- // node src/ci-report.js --markers <ocqa-markers.txt>
9
+ // node src/ci-report.js --markers <ocqa-markers.txt> # or --flows-only: no exploration ran
10
+ // # (deliberate; suites are the whole surface)
10
11
  // [--baseline <baseline.json>] # prior run's findings[] (or a full report)
11
12
  // [--flow-log <log> ...] # run-flow.sh logs (repeatable)
12
13
  // [--json-out <report.json>] # full report incl. findings for the next baseline
@@ -31,7 +32,7 @@
31
32
  import fs from "fs";
32
33
  import path from "node:path";
33
34
  import { execSync } from "node:child_process";
34
- import { buildQaReport, computeRegression, computeContentCollapse, computeReachabilityLoss, evaluateGate, GATE_EXIT } from "./report.js";
35
+ import { buildQaReport, buildFlowsOnlyReport, computeRegression, computeContentCollapse, computeReachabilityLoss, evaluateGate, GATE_EXIT } from "./report.js";
35
36
  import { writeHtmlReport } from "./html-report.js";
36
37
  import { buildUiMapFromMarkers, writeUiMap } from "./ui-map.js";
37
38
  import { proposeSelectorMaintenance, validateWebMaintenanceProposal } from "./maintenance-proposal.js";
@@ -55,13 +56,14 @@ function parseArgs(argv) {
55
56
  else if (a === "--pr-plan") args.prPlan = argv[++i];
56
57
  else if (a === "--project-dir") args.projectDir = argv[++i];
57
58
  else if (a === "--maintenance-url") args.maintenanceUrl = argv[++i];
59
+ else if (a === "--flows-only") args.flowsOnly = true;
58
60
  else {
59
61
  console.error(`Unknown argument: ${a}`);
60
62
  process.exit(2);
61
63
  }
62
64
  }
63
- if (!args.markers) {
64
- console.error("Required: --markers <ocqa-markers.txt>");
65
+ if (!args.markers && !args.flowsOnly) {
66
+ console.error("Required: --markers <ocqa-markers.txt> (or --flows-only for a gate with no exploration)");
65
67
  process.exit(2);
66
68
  }
67
69
  if (!["gate", "absolute", "any", "high", "medium"].includes(args.failOn)) {
@@ -126,7 +128,7 @@ function parseFlowLog(logPath) {
126
128
  const name = logPath.split("/").pop().replace(/\.log$/, "");
127
129
  if (!fs.existsSync(logPath)) return { name, passed: false, total: 0, failed: 0, steps: [], missing: true, modelObserved: false, deterministicFailed: false };
128
130
  const steps = [];
129
- let total = 0, executed = 0, failed = 0, passed = false, sawResult = false, flowName = null, kind = "flow", contract = "", criticality = "";
131
+ let total = 0, executed = 0, failed = 0, passed = false, sawResult = false, flowName = null, kind = "flow", contract = "", criticality = "", url = "";
130
132
  for (const raw of fs.readFileSync(logPath, "utf8").split(/\r?\n/)) {
131
133
  const line = raw.trim();
132
134
  if (line.startsWith("OCQA_FLOW_STEP:{")) {
@@ -145,6 +147,7 @@ function parseFlowLog(logPath) {
145
147
  if (o.kind) kind = o.kind;
146
148
  if (o.contract) contract = o.contract;
147
149
  if (o.criticality) criticality = o.criticality;
150
+ if (o.url) url = o.url; // the page the flow actually opened (diagnosable from the report)
148
151
  sawResult = true;
149
152
  } catch { /* ignore malformed */ }
150
153
  }
@@ -160,7 +163,7 @@ function parseFlowLog(logPath) {
160
163
  // default deterministic gate. evaluateGate reads these flags, never the raw action string.
161
164
  const modelObserved = steps.some((s) => s.action === "assert_ai");
162
165
  const deterministicFailed = steps.some((s) => s.action !== "assert_ai" && s.status === "fail");
163
- return { name: flowName || name, kind, ...(contract ? { contract, criticality } : {}), passed, total, executed, failed, steps, modelObserved, deterministicFailed };
166
+ return { name: flowName || name, kind, ...(contract ? { contract, criticality } : {}), ...(url ? { url } : {}), passed, total, executed, failed, steps, modelObserved, deterministicFailed };
164
167
  }
165
168
 
166
169
  function loadBaseline(baselinePath) {
@@ -459,6 +462,7 @@ function renderMarkdown(report, regression, flows, scenarios, contracts, prPlan,
459
462
  for (const f of flows) {
460
463
  const firstFail = f.steps.find((s) => s.status === "fail");
461
464
  lines.push(`- ${f.passed ? "✅" : "❌"} **${f.name}** — ${f.steps.filter((s) => s.status === "pass").length}/${f.total} steps` +
465
+ (f.url ? ` — \`${f.url}\`` : "") +
462
466
  (firstFail ? ` — failed at \`${firstFail.action} ${firstFail.target}\`${firstFail.detail ? `: ${firstFail.detail}` : ""}` : "") +
463
467
  (f.missing ? " — log missing (flow did not run)" : ""));
464
468
  }
@@ -509,7 +513,9 @@ function renderMarkdown(report, regression, flows, scenarios, contracts, prPlan,
509
513
  }
510
514
 
511
515
  const args = parseArgs(process.argv.slice(2));
512
- const report = buildQaReport(args.markers, { platform: args.platform || "ios", target: args.label || null });
516
+ const report = args.flowsOnly
517
+ ? buildFlowsOnlyReport({ platform: args.platform || "ios", target: args.label || null })
518
+ : buildQaReport(args.markers, { platform: args.platform || "ios", target: args.label || null });
513
519
  if (!report) {
514
520
  // Required evidence could not be obtained — this is inconclusive (fails closed), not a gate FAIL
515
521
  // and not a usage error. See the outcome model in report.js (GATE_EXIT).
@@ -517,7 +523,7 @@ if (!report) {
517
523
  process.exit(GATE_EXIT.inconclusive);
518
524
  }
519
525
  let currentUiMap = null;
520
- if (args.htmlDir) {
526
+ if (args.htmlDir && !args.flowsOnly) {
521
527
  try {
522
528
  const map = buildUiMapFromMarkers({ markersPath: args.markers, platform: args.platform || "ios", target: args.label || "", runId: path.basename(args.htmlDir) });
523
529
  currentUiMap = map;
@@ -545,7 +551,10 @@ if (baseline?.targetKey && baseline.targetKey !== args.targetKey) {
545
551
  if (args.targetKey) report.targetKey = args.targetKey;
546
552
  // Content-collapse findings are cross-run by nature — merge them into the current findings
547
553
  // BEFORE the regression diff so they count as new-vs-baseline and drive the gate normally.
548
- const collapsed = [
554
+ // A flows-only run carries no exploration evidence: cross-run collapse/reachability and the
555
+ // findings regression are exploration comparisons and would read "everything resolved" — skip
556
+ // them rather than lie.
557
+ const collapsed = args.flowsOnly ? [] : [
549
558
  ...computeContentCollapse(report.screenElementCounts, baseline?.screenElementCounts),
550
559
  ...computeReachabilityLoss(report, baseline),
551
560
  ];
@@ -560,7 +569,7 @@ if (collapsed.length) {
560
569
  // No score/verdict to mutate — exploration is scoreless; the gate renders the outcome.
561
570
  report.headline = `${collapsed.length} screen(s) regressed vs. baseline (content collapsed or became unreachable).`;
562
571
  }
563
- const regression = computeRegression(report.findings, baseline?.findings ?? null);
572
+ const regression = args.flowsOnly ? null : computeRegression(report.findings, baseline?.findings ?? null);
564
573
  // A baseline captured at a different device/viewport is a layout comparison, not a regression
565
574
  // signal: a phone run legitimately hides desktop nav links, so its "resolved" list lies. Keep
566
575
  // the diff (new findings still gate) but stamp the mismatch so every consumer can see it.
@@ -47,7 +47,7 @@ export const FLOW_ACTIONS = Object.freeze([
47
47
  { action: "swipe", target: "up | down | left | right", passes: "the gesture was performed" },
48
48
  { action: "back", target: "(none)", passes: "the platform back navigation was performed" },
49
49
  { action: "wait", target: "milliseconds (fixed pause; prefer wait_for)", passes: "always" },
50
- { action: "wait_for", target: "label / text (+ timeoutMs)", passes: "the element appeared before the timeout" },
50
+ { action: "wait_for", target: "label / text (+ per-step timeoutMs/timeout; flow-level timeoutMs sets the default)", passes: "the element appeared before the timeout" },
51
51
  { action: "assert_screen", target: "the detected SCREEN TITLE (navigation bar / heading), not arbitrary text", passes: "the current screen's title equals the target" },
52
52
  { action: "assert_exists", target: "label / text", passes: "an element with that text or id is present" },
53
53
  { action: "assert_absent", target: "label / text", passes: "no element with that text or id is present" },
@@ -172,10 +172,12 @@ export class FlowLog {
172
172
  }
173
173
  }
174
174
 
175
- finish() {
175
+ finish(extra = {}) {
176
176
  const passed = this.failed === 0 && this.executed > 0;
177
- // Keep `passed` first for marker consumers that stream-match the payload.
178
- this.emit(`OCQA_FLOW_RESULT:${JSON.stringify({ passed, name: this.flow.name || "flow", kind: this.kind, ...(this.contract ? { contract: this.contract, criticality: this.criticality } : {}), total: this.flow.steps.length, executed: this.executed, failed: this.failed })}`);
179
- return { passed, name: this.flow.name || "flow", kind: this.kind, ...(this.contract ? { contract: this.contract, criticality: this.criticality } : {}), total: this.flow.steps.length, executed: this.executed, failed: this.failed, logPath: this.logPath, lines: this.lines };
177
+ // Keep `passed` first for marker consumers that stream-match the payload. `extra` carries
178
+ // run context worth diagnosing from the log alone (e.g. the URL the flow actually opened —
179
+ // field issue #15 was six flows silently replayed against the wrong page).
180
+ this.emit(`OCQA_FLOW_RESULT:${JSON.stringify({ passed, name: this.flow.name || "flow", kind: this.kind, ...(this.contract ? { contract: this.contract, criticality: this.criticality } : {}), total: this.flow.steps.length, executed: this.executed, failed: this.failed, ...extra })}`);
181
+ return { passed, name: this.flow.name || "flow", kind: this.kind, ...(this.contract ? { contract: this.contract, criticality: this.criticality } : {}), total: this.flow.steps.length, executed: this.executed, failed: this.failed, ...extra, logPath: this.logPath, lines: this.lines };
180
182
  }
181
183
  }
@@ -84,6 +84,9 @@ export function writeHtmlReport(captureDir, { report, label = "", recordingWarni
84
84
  ${esc(f.title)}${f.screen ? ` <span class="dim">— on ${esc(f.screen)}</span>` : ""}${f.url ? ` <span class="dim">— ${esc(f.url)}</span>` : ""}
85
85
  ${f.aiAnalysis ? `<div class="ai">why: ${esc(f.aiAnalysis)}</div>` : ""}
86
86
  ${f.suggestedFix ? `<div class="ai">fix: ${esc(f.suggestedFix)}</div>` : ""}
87
+ ${f.evidence && fs.existsSync(path.join(captureDir, f.evidence))
88
+ ? `<a class="evidence" href="${esc(f.evidence)}"><img loading="lazy" src="${esc(f.evidence)}" alt="The specific control named above, scrolled into view"></a>`
89
+ : ""}
87
90
  </li>`
88
91
  )
89
92
  .join("\n")
@@ -129,6 +132,8 @@ ${r.conditionsNotReached?.length ? `<div><h2>Conditions not reached</h2><ul>${sc
129
132
  table.trace th, table.trace td { text-align: left; padding: 0.25rem 0.7rem 0.25rem 0; border-bottom: 1px solid #eceef1; }
130
133
  table.trace th { color: #57606a; font-weight: 600; }
131
134
  .ai { color: #57606a; font-size: 0.88rem; margin: 0.15rem 0 0 0.2rem; }
135
+ a.evidence { display: block; margin: 0.4rem 0 0 0.2rem; max-width: 360px; }
136
+ a.evidence img { width: 100%; border: 1px solid #d0d7de; border-radius: 6px; }
132
137
  .dim { color: #57606a; }
133
138
  .grid { display: grid; grid-template-columns: repeat(auto-fill, minmax(200px, 1fr)); gap: 0.8rem; }
134
139
  figure { margin: 0; } figure img { width: 100%; border: 1px solid #d0d7de; border-radius: 6px; }
@@ -31,6 +31,7 @@ export function parseOcqaMarkers(markersFilePath) {
31
31
  const issues = [];
32
32
  let complete = null;
33
33
  let context = null;
34
+ let requestedMaxActions = null;
34
35
 
35
36
  for (const line of lines) {
36
37
  if (!line.startsWith("OCQA_")) continue;
@@ -59,6 +60,7 @@ export function parseOcqaMarkers(markersFilePath) {
59
60
  if (category === "ISSUE") issues.push(parsed);
60
61
  if (category === "COMPLETE") complete = parsed;
61
62
  if (category === "CONTEXT") context = parsed;
63
+ if (category === "PROGRESS" && parsed && typeof parsed === "object" && Number.isFinite(parsed.max)) requestedMaxActions = parsed.max;
62
64
  }
63
65
 
64
66
  return {
@@ -75,6 +77,7 @@ export function parseOcqaMarkers(markersFilePath) {
75
77
  ),
76
78
  complete,
77
79
  context,
80
+ requestedMaxActions,
78
81
  actions,
79
82
  recentActions: actions.slice(-5),
80
83
  recentTransitions: transitions.slice(-5),
@@ -172,7 +175,10 @@ export function buildQaReport(markersFilePath, { platform = "ios", target = null
172
175
  // same screen are two findings, and fixing one while breaking another is a regression.
173
176
  const target = (typeof o.control === "string" && o.control) || (typeof o.target === "string" && o.target) || null;
174
177
  const url = typeof o.url === "string" && o.url.trim() ? o.url.trim() : null;
175
- rawIssues.push({ type: o.type, severity: sev, title: o.title, screen: o.screen || null, target, url, step: o.step ?? null });
178
+ // Element-scoped screenshot proving the specific control the finding names is actually
179
+ // visible — set only for web findings that name one (anchor_missing, placeholder_link).
180
+ const evidence = typeof o.evidence === "string" && o.evidence.trim() ? o.evidence.trim() : null;
181
+ rawIssues.push({ type: o.type, severity: sev, title: o.title, screen: o.screen || null, target, url, step: o.step ?? null, evidence });
176
182
  } catch {
177
183
  /* ignore malformed */
178
184
  }
@@ -288,9 +294,16 @@ export function buildQaReport(markersFilePath, { platform = "ios", target = null
288
294
  // placeholder anchors, and visible controls do not require a second route. Native exploration
289
295
  // retains the stronger multi-screen/action floor. A credentialless single-screen login remains
290
296
  // inconclusive so a login wall can never turn into a clean pass.
291
- const coverageFloorMet = platform === "web"
292
- ? screensExplored >= 1 && actionsPerformed >= 1
293
- : screensExplored >= 2 && actionsPerformed >= 3;
297
+ // An EXPLICITLY small budget scales the floor to the request (issue #18): a --actions 1 run
298
+ // that performed its one action saw exactly what was asked — that is conclusive evidence of
299
+ // one action, not "couldn't see enough". A run that undershot even its tiny request (crash at
300
+ // launch: 0 of 1) stays inconclusive, so the floor's crash-detection job survives.
301
+ const requestedMax = Number.isFinite(base.requestedMaxActions) ? base.requestedMaxActions : null;
302
+ const floorActions = platform === "web" ? 1 : 3;
303
+ const floorScreens = platform === "web" ? 1 : 2;
304
+ const effectiveFloorActions = requestedMax != null ? Math.min(floorActions, requestedMax) : floorActions;
305
+ const effectiveFloorScreens = requestedMax != null && requestedMax < floorActions ? 1 : floorScreens;
306
+ const coverageFloorMet = screensExplored >= effectiveFloorScreens && actionsPerformed >= effectiveFloorActions;
294
307
  const unexercisedLoginWall = anySecure && !loginAttempted && screensExplored <= 1;
295
308
  const inconclusive = !coverageFloorMet || unexercisedLoginWall || timeBudgetExhausted;
296
309
  // "completed" is reserved for a run that exhausted its action budget; a drained frontier is
@@ -572,7 +585,47 @@ export const GATE_EXIT = { pass: 0, fail: 1, error: 2, inconclusive: 3 };
572
585
  // finding at or above that severity, and the CLI defaults web targets to `medium` — a 404 in the
573
586
  // nav is the release blocker on a website, and a field-tested green PASS over six deterministic
574
587
  // findings was exactly the dishonest verdict this product refuses to render.
575
- export const GATE_POLICY_VERSION = "4";
588
+ export const GATE_POLICY_VERSION = "5";
589
+
590
+ // A gate run with exploration deliberately not requested (--actions 0): reviewed Flows,
591
+ // scenarios and contracts are the whole deterministic surface. The stub is shaped like an
592
+ // ExplorationRun so every consumer (gate, markdown, HTML, JSON) renders it without special
593
+ // cases, and it says plainly that exploration was not run rather than pretending coverage.
594
+ export function buildFlowsOnlyReport({ platform = "ios", target = null } = {}) {
595
+ return {
596
+ kind: "tapp-exploration-run",
597
+ schemaVersion: 1,
598
+ runStatus: "not-run",
599
+ stopReason: "exploration-not-requested",
600
+ headline: "Exploration not requested (flows-only gate): reviewed suites are the entire deterministic surface of this run.",
601
+ inconclusive: false,
602
+ explorationRequested: false,
603
+ coverage: { screensExplored: 0, actionsPerformed: 0, screens: [] },
604
+ trace: [],
605
+ evidence: { markers: null },
606
+ captureContext: null,
607
+ uiMap: null,
608
+ comparison: null,
609
+ checkedFor: [],
610
+ notChecked: ["autonomous exploration (not requested: --actions 0 — the gate replays reviewed suites only)"],
611
+ conditionsNotReached: [],
612
+ platform,
613
+ target: typeof target === "string" && target.trim() ? target.trim() : null,
614
+ screensExplored: 0,
615
+ actionsPerformed: 0,
616
+ findingCounts: { critical: 0, high: 0, medium: 0, low: 0, total: 0 },
617
+ deterministicFindingCounts: { critical: 0, high: 0, medium: 0, low: 0, total: 0 },
618
+ sampledFindingCounts: { critical: 0, high: 0, medium: 0, low: 0, total: 0 },
619
+ findings: [],
620
+ screens: [],
621
+ screenElementCounts: {},
622
+ inputFieldsEncountered: [],
623
+ loginEncountered: false,
624
+ credentialsProvided: false,
625
+ credentialsUsed: false,
626
+ credentialWarning: false,
627
+ };
628
+ }
576
629
 
577
630
  // Pure gate evaluator: frozen evidence + policy → a GateRun decision. Extracted verbatim from the
578
631
  // former inline logic in ci-report.js so the `[char]` characterization tests keep passing — the
@@ -617,7 +670,14 @@ export function evaluateGate({ report, regression = null, flows = [], scenarios
617
670
  if (prPlan?.execution?.notRun) inconclusive(`${prPlan.execution.notRun} selected release contract(s) did not run`);
618
671
  if (prPlan?.execution?.explorationFailed) inconclusive(`${prPlan.execution.explorationFailed} planned PR exploration target(s) failed or were not reached`);
619
672
 
620
- if (failOn === "any") {
673
+ if (report.explorationRequested === false) {
674
+ // Flows-only gate (v5): exploration was deliberately not requested, so no exploration-
675
+ // derived policy applies — the reviewed suites above are the entire decision. An empty
676
+ // selection proves nothing and must not pass.
677
+ if (!flows.length && !scenarios.length && !contracts.length) {
678
+ inconclusive("flows-only gate (--actions 0) selected no flows, scenarios, or contracts — nothing was verified");
679
+ }
680
+ } else if (failOn === "any") {
621
681
  if (report.findingCounts.total > 0) fail(`${report.findingCounts.total} finding(s) (fail-on: any)`);
622
682
  // "any" is the strictest policy — an inconclusive run (evidence not obtained) must never pass it.
623
683
  if (report.inconclusive) inconclusive("run was inconclusive (coverage floor not met)");
@@ -25,6 +25,13 @@ const NAV_TIMEOUT_MS = 15_000;
25
25
  const BUTTONS_PER_PAGE = 4;
26
26
  const OUTBOUND_LINK_LIMIT = 10;
27
27
  const WATCH_ACTION_DELAY_MS = 350;
28
+ // A page's visible signature (waitForWebStability's basis) can go quiet before an async-injected
29
+ // widget (a third-party tour scheduler, a chat launcher, ...) has actually finished mounting its
30
+ // target element — nothing about that wait shows a spinner. One snapshot cannot tell "genuinely
31
+ // dead" from "hasn't finished loading"; give a real anchor-missing candidate a second look before
32
+ // treating it as a confirmed defect.
33
+ const ANCHOR_RECHECK_MS = 4_000;
34
+ const ANCHOR_RECHECK_INTERVAL_MS = 400;
28
35
  const ERROR_TEXT_RE = /\b(something went wrong|internal server error|an error occurred|failed to load|unhandled exception)\b/i;
29
36
  const STANDALONE_ERROR_TEXT_RE = /^(something went wrong|internal server error|an error occurred|failed to load|unhandled exception)(?:[.!:]|\s|$)/i;
30
37
 
@@ -82,6 +89,53 @@ export function webPageAppearsBlank({ textLen = 0, controlCount = 0, visualConte
82
89
  return Number(textLen) === 0 && Number(controlCount) === 0 && Number(visualContentCount) === 0;
83
90
  }
84
91
 
92
+ function webAnchorIdMissing(rawId) {
93
+ try {
94
+ const esc = window.CSS && CSS.escape ? CSS.escape(rawId) : rawId;
95
+ return !document.getElementById(rawId) && !document.querySelector(`a[name="${esc}"]`);
96
+ } catch {
97
+ return !document.getElementById(rawId);
98
+ }
99
+ }
100
+
101
+ // A single DOM snapshot cannot distinguish "this anchor target will never exist" from "the
102
+ // script that creates it hasn't run yet". Poll for up to timeoutMs before accepting the miss —
103
+ // cheap when the target is genuinely absent (every poll agrees), and it only spends the extra
104
+ // time on pages that actually have a candidate anchor_missing finding.
105
+ export async function webAnchorStillMissing(page, anchor, { timeoutMs = ANCHOR_RECHECK_MS, intervalMs = ANCHOR_RECHECK_INTERVAL_MS } = {}) {
106
+ const id = decodeURIComponent(String(anchor || "").slice(1));
107
+ if (!id) return true;
108
+ const deadline = Date.now() + Math.max(0, Number(timeoutMs) || 0);
109
+ for (;;) {
110
+ const missing = await page.evaluate(webAnchorIdMissing, id).catch(() => true);
111
+ if (!missing) return false;
112
+ if (Date.now() >= deadline) return true;
113
+ await page.waitForTimeout(intervalMs);
114
+ }
115
+ }
116
+
117
+ // Evidence for a finding that names a specific control is worthless if the screenshot never
118
+ // actually shows that control — a generic per-screen shot proves nothing about where the element
119
+ // is on the page. Scroll it into view and shoot it directly; fall back to a viewport shot (still
120
+ // scrolled to the element) if the element itself can't be screenshotted (zero-size, clipped).
121
+ export async function captureElementEvidence(page, locator, outDir, name) {
122
+ try {
123
+ const target = locator.first();
124
+ if ((await target.count()) === 0) return null;
125
+ await target.scrollIntoViewIfNeeded({ timeout: 2_000 });
126
+ await page.waitForTimeout(120);
127
+ const evidencePath = path.join(outDir, name);
128
+ await target.screenshot({ path: evidencePath }).catch(() => page.screenshot({ path: evidencePath }));
129
+ return path.basename(evidencePath);
130
+ } catch {
131
+ return null;
132
+ }
133
+ }
134
+
135
+ function cssAttrEscape(value) {
136
+ return String(value).replace(/["\\]/g, (c) => "\\" + c);
137
+ }
138
+
85
139
  async function installWebListenerTracking(context) {
86
140
  await context.addInitScript(() => {
87
141
  const key = Symbol.for("tapp.clickListeners");
@@ -460,12 +514,35 @@ export async function inspectWebPage({ url, timeoutMs = NAV_TIMEOUT_MS, screensh
460
514
  try {
461
515
  await page.getByText(requested, { exact: false }).first().waitFor({ state: "visible", timeout: boundedTimeout });
462
516
  } catch {
463
- throw new Error(`Timed out waiting for visible text “${requested}”`);
517
+ // The timeout is exactly when the screenshot matters most (field issue #22): the page
518
+ // may have rendered an error state. Capture what IS there and hand it to the caller.
519
+ const error = new Error(`Timed out waiting for visible text “${requested}”`);
520
+ error.timeoutEvidence = {
521
+ image: screenshot ? await page.screenshot({ type: "png", fullPage: !!fullPage }).catch(() => null) : null,
522
+ visible: await page.locator("body").innerText({ timeout: 1000 })
523
+ .then((text) => [...new Set(String(text).split(/\n+/).map((l) => l.trim()).filter((l) => l.length >= 2 && l.length <= 60))].slice(0, 8))
524
+ .catch(() => []),
525
+ url: page.url(),
526
+ };
527
+ throw error;
464
528
  }
465
529
  stability = await waitForWebStability(page, { timeoutMs: Math.min(5_000, boundedTimeout) });
466
530
  }
467
531
  const observed = await page.evaluate(() => {
468
532
  const visible = (element) => element.offsetParent !== null;
533
+ // A control's name is what a screen reader would read: ALL descendant text, shadow roots
534
+ // included, with spaces between the pieces. Custom card buttons render their title/price
535
+ // inside nested divs (or a shadow root) — plain textContent ran the pieces together and
536
+ // came back empty for shadow DOM, so `assert_exists: "$2,000"` had nothing to match
537
+ // (field issue #19).
538
+ const accessibleText = (node) => {
539
+ let out = "";
540
+ for (const child of (node.shadowRoot || node).childNodes) {
541
+ if (child.nodeType === Node.TEXT_NODE) out += child.textContent + " ";
542
+ else if (child.nodeType === Node.ELEMENT_NODE) out += accessibleText(child) + " ";
543
+ }
544
+ return out;
545
+ };
469
546
  const controls = [...document.querySelectorAll("button, a[href], input, textarea, select, summary, [role=button], [role=tab], [role=checkbox], [role=switch]")]
470
547
  .filter((element) => element.type !== "hidden" && visible(element))
471
548
  .slice(0, 80)
@@ -474,7 +551,7 @@ export async function inspectWebPage({ url, timeoutMs = NAV_TIMEOUT_MS, screensh
474
551
  const field = ["input", "textarea", "select"].includes(tag);
475
552
  const secure = element.type === "password";
476
553
  const role = element.getAttribute("role") || (tag === "a" ? "link" : tag === "button" || tag === "summary" ? "button" : "");
477
- const label = (element.labels?.[0]?.textContent || element.getAttribute("aria-label") || element.textContent || element.placeholder || element.name || element.id || "").trim().slice(0, 120);
554
+ const label = (element.labels?.[0]?.textContent || element.getAttribute("aria-label") || accessibleText(element).replace(/\s+/g, " ").trim() || element.placeholder || element.name || element.id || "").trim().slice(0, 120);
478
555
  const box = element.getBoundingClientRect();
479
556
  return {
480
557
  // `type` stays faithful so an agent follows links and presses buttons, not vice versa.
@@ -589,13 +666,13 @@ export async function exploreWeb({ url, maxActions = 40, timeoutSec = 300, outDi
589
666
 
590
667
  const deadline = Date.now() + timeoutSec * 1000;
591
668
  const issues = []; // emitted immediately; kept for counting only
592
- const issue = (type, severity, title, screen, target, sourceUrl) => {
669
+ const issue = (type, severity, title, screen, target, sourceUrl, evidence) => {
593
670
  issues.push(type);
594
671
  // Findings belong to the page that CARRIED the defect. Async detectors default to the
595
672
  // current page; the post-crawl outbound audit passes the link's source page explicitly so
596
673
  // a bad footer link is never attributed to whatever page happened to be visited last.
597
674
  const pageUrl = sourceUrl || page.url();
598
- emit("ISSUE", { type, severity, title, screen, ...(target ? { target } : {}), ...(pageUrl && pageUrl !== "about:blank" ? { url: pageUrl } : {}) });
675
+ emit("ISSUE", { type, severity, title, screen, ...(target ? { target } : {}), ...(pageUrl && pageUrl !== "about:blank" ? { url: pageUrl } : {}), ...(evidence ? { evidence } : {}) });
599
676
  };
600
677
 
601
678
  // Async defect listeners: attribute to whatever screen is current when they fire.
@@ -782,14 +859,33 @@ export async function exploreWeb({ url, maxActions = 40, timeoutSec = 300, outDi
782
859
  for (const finding of webPlaceholderLinkFindings(info.placeholderLinks)) {
783
860
  if (placeholderLinksSeen.has(finding.target)) continue;
784
861
  placeholderLinksSeen.add(finding.target);
785
- issue(finding.type, finding.severity, finding.title, screen, finding.target);
862
+ // Only a labeled link's evidence can be located with any confidence — an unlabeled
863
+ // link's `target` is a synthetic fingerprint, not text a locator can find on the page.
864
+ const evidence = finding.target.startsWith("unlabeled:")
865
+ ? null
866
+ : await captureElementEvidence(
867
+ page,
868
+ page.locator("a[href]").filter({ hasText: finding.target }),
869
+ outDir,
870
+ `evidence_${screenshotFor.size}_${slug(screen)}_${slug(finding.target)}.png`
871
+ );
872
+ issue(finding.type, finding.severity, finding.title, screen, finding.target, null, evidence);
786
873
  }
787
- // Anchor links pointing at ids that do not exist are deterministic dead navigation.
874
+ // Anchor links pointing at ids that do not exist are deterministic dead navigation — but
875
+ // one snapshot can't tell "genuinely dead" from "the script that creates the target hasn't
876
+ // finished running yet" (an async-mounted widget shows no spinner while it loads).
788
877
  for (const anchor of info.missingAnchors || []) {
789
878
  const anchorTarget = `${key}${anchor}`;
790
879
  if (missingAnchorsSeen.has(anchorTarget)) continue;
791
880
  missingAnchorsSeen.add(anchorTarget);
792
- issue("anchor_missing", "medium", `Anchor link "${anchor}" has no matching element on the page`, screen, anchorTarget);
881
+ if (!(await webAnchorStillMissing(page, anchor))) continue;
882
+ const evidence = await captureElementEvidence(
883
+ page,
884
+ page.locator(`a[href="${cssAttrEscape(anchor)}"]`),
885
+ outDir,
886
+ `evidence_${screenshotFor.size}_${slug(screen)}_${slug(anchor)}.png`
887
+ );
888
+ issue("anchor_missing", "medium", `Anchor link "${anchor}" has no matching element on the page`, screen, anchorTarget, null, evidence);
793
889
  }
794
890
  }
795
891
  return { key, screen, info };
@@ -7,6 +7,16 @@ import { loadPlaywright, webContextOptions } from "./web-explorer.js";
7
7
 
8
8
  const DEFAULT_TIMEOUT = 6000;
9
9
 
10
+ // What WAS visible when a wait missed — points a failure report at the fix (renamed label,
11
+ // error state, wrong page) without a separate `tapp tree` run (field issue #20).
12
+ async function visibleLabelsHint(page, limit = 8) {
13
+ try {
14
+ const text = await page.locator("body").innerText({ timeout: 1000 });
15
+ const labels = [...new Set(String(text).split(/\n+/).map((l) => l.trim()).filter((l) => l.length >= 2 && l.length <= 60))].slice(0, limit);
16
+ return labels.length ? ` — visible: ${labels.join(" · ")}` : "";
17
+ } catch { return ""; }
18
+ }
19
+
10
20
  async function firstVisible(candidates) {
11
21
  for (const locator of candidates) {
12
22
  try {
@@ -115,12 +125,27 @@ export async function runWebRequestStep({ step, startUrl, vars = {}, timeout = D
115
125
  return { action: "request", target: `${method} ${targetUrl.pathname}`, status: "pass", detail: "" };
116
126
  }
117
127
 
128
+ // Playwright's timeout errors bury the actual cause ("<div id=…> intercepts pointer events",
129
+ // "element is not visible") in a multi-line retry log; single-line consumers kept only
130
+ // "Timeout 6000ms exceeded" and the report read like the element did not exist (field issue
131
+ // #17). Keep the first line AND name the diagnosable cause on it.
132
+ export function distillPlaywrightFailure(message) {
133
+ const text = String(message || "");
134
+ const firstLine = text.split("\n", 1)[0].trim();
135
+ const cause = text.match(/(<[^>\n]{1,120}>[^\n]{0,80}intercepts pointer events)/)
136
+ || text.match(/element is (?:not visible|outside of the viewport|not enabled|not stable)[^\n]*/)
137
+ || text.match(/waiting for element to be visible, enabled and stable[^\n]*/);
138
+ if (!cause || firstLine.includes(cause[0])) return firstLine;
139
+ const reason = cause[0].includes("intercepts pointer events") ? `click intercepted by ${cause[0].replace(/\s*intercepts pointer events.*/, "").trim()}` : cause[0].trim();
140
+ return `${firstLine} — ${reason}`;
141
+ }
142
+
118
143
  export async function executeWebFlowStep({ page, step, vars = {}, defaultTimeout = DEFAULT_TIMEOUT }) {
119
144
  const raw = normalizeFlowStep(step);
120
145
  const action = raw.action;
121
146
  const target = substituteFlowValue(raw.target, vars);
122
147
  const value = substituteFlowValue(raw.value, vars);
123
- const timeout = Number(raw.params.timeoutMs) || defaultTimeout;
148
+ const timeout = Number(raw.params.timeoutMs ?? raw.params.timeout) || defaultTimeout;
124
149
  let status = "pass";
125
150
  let detail = "";
126
151
  try {
@@ -154,7 +179,7 @@ export async function executeWebFlowStep({ page, step, vars = {}, defaultTimeout
154
179
  } else if (action === "back") {
155
180
  await page.goBack({ waitUntil: "domcontentloaded" });
156
181
  } else if (action === "wait_for") {
157
- if (!await waitForScreen(page, target, timeout)) throw new Error(`‘${target}’ never appeared within ${timeout}ms`);
182
+ if (!await waitForScreen(page, target, timeout)) throw new Error(`‘${target}’ never appeared within ${timeout}ms${await visibleLabelsHint(page)}`);
158
183
  } else if (action === "wait") {
159
184
  await page.waitForTimeout(timeout);
160
185
  } else if (action === "assert_screen") {
@@ -187,13 +212,34 @@ export async function executeWebFlowStep({ page, step, vars = {}, defaultTimeout
187
212
  await settle(page);
188
213
  } catch (error) {
189
214
  status = "fail";
190
- detail = error.message || String(error);
215
+ detail = distillPlaywrightFailure(error.message || String(error));
191
216
  }
192
217
  return { action, target: action === "login" ? "sign-in form" : target || value, status, detail, task: raw.task };
193
218
  }
194
219
 
195
- export async function runWebFlow({ flow, url, logPath, screenshotDir, playwright, device = "", viewport = "" }) {
196
- const startUrl = url || flow.url || (/^https?:\/\//i.test(flow.app || "") ? flow.app : "");
220
+ // The gate targets ONE deployment; a Flow's `url:` names a PAGE of the app, not a deployment.
221
+ // Under a gate url, keep the Flow's path/query/hash but rebase a differing origin onto the
222
+ // gate's: committed Flows carry the origin they were recorded against (a dev port, the prod
223
+ // domain) while the gate may run an ephemeral server — and six flows replaying the gate
224
+ // homepage instead of their own pages was field issue #15.
225
+ export function resolveGateFlowUrl(declared, gateUrl) {
226
+ if (!declared) return gateUrl || "";
227
+ if (!gateUrl) return declared;
228
+ try {
229
+ const page = new URL(declared);
230
+ const gate = new URL(gateUrl);
231
+ if (page.origin === gate.origin) return declared;
232
+ return new URL(page.pathname + page.search + page.hash, gate.origin).href;
233
+ } catch {
234
+ return declared;
235
+ }
236
+ }
237
+
238
+ export async function runWebFlow({ flow, url, logPath, screenshotDir, playwright, device = "", viewport = "", urlIsFallback = false }) {
239
+ // urlIsFallback = the caller is a gate: the Flow's own page wins, rebased onto the gate's
240
+ // deployment (see resolveGateFlowUrl). An explicit caller url (CLI positional) still overrides.
241
+ const declaredUrl = flow.url || (/^https?:\/\//i.test(flow.app || "") ? flow.app : "");
242
+ const startUrl = urlIsFallback ? resolveGateFlowUrl(declaredUrl, url) : (url || declaredUrl);
197
243
  if (!startUrl) throw new Error("Web Flow needs `url:` (or an http(s) `app:` value)");
198
244
  if (logPath) fs.rmSync(logPath, { force: true });
199
245
  const setup = flow.setup || [];
@@ -204,7 +250,7 @@ export async function runWebFlow({ flow, url, logPath, screenshotDir, playwright
204
250
  const browser = await pw.chromium.launch({ headless: true });
205
251
  const context = await browser.newContext(webContextOptions({ device, viewport, devices: pw.devices }));
206
252
  const page = await context.newPage();
207
- const timeout = Number(flow.timeoutMs) || DEFAULT_TIMEOUT;
253
+ const timeout = Number(flow.timeoutMs ?? flow.timeout) || DEFAULT_TIMEOUT;
208
254
  page.setDefaultTimeout(timeout);
209
255
  const vars = flowVariables(flow);
210
256
  let index = 0;
@@ -250,5 +296,5 @@ export async function runWebFlow({ flow, url, logPath, screenshotDir, playwright
250
296
  await requestPhase(teardown);
251
297
  await browser.close().catch(() => {});
252
298
  }
253
- return log.finish();
299
+ return log.finish({ url: startUrl });
254
300
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@aarwitz/tapp",
3
- "version": "0.17.11",
3
+ "version": "0.17.13",
4
4
  "mcpName": "io.github.aarwitz/tapp",
5
5
  "description": "Let coding agents verify UI changes on real iOS, Android, and web surfaces, then enforce reviewed proof in deterministic CI.",
6
6
  "license": "MIT",
@@ -16,7 +16,8 @@
16
16
  # tapp ci # in an initialized repo: reads .tapp/application-model.json for platform/target
17
17
  # # omit --url with --project-dir to detect/build/start/stop one owned web target
18
18
  # # bundle id is detected from the .app when omitted
19
- # [--actions N] # exploration budget (default 40)
19
+ # [--actions N] # exploration budget (default 40); 0 = flows-only
20
+ # # gate (replay reviewed suites, no exploration)
20
21
  # [--timeout S] # exploration watchdog (default 600)
21
22
  # [--flows <glob>] # Flow YAMLs to replay (default: <app repo>/.tapp/flows/*.yml if --project-dir given)
22
23
  # [--scenarios <glob>] # Multi-actor Scenario YAMLs (web; default: <app repo>/.tapp/scenarios/*.yml)
@@ -83,7 +84,7 @@ while [[ $# -gt 0 ]]; do
83
84
  done
84
85
  [[ "$PLATFORM" == "ios" || "$PLATFORM" == "android" || "$PLATFORM" == "web" ]] || { echo "❌ --platform must be ios|android|web" >&2; exit 2; }
85
86
  [[ "$PLATFORM" == "ios" && -n "$VIEWPORT" ]] && { echo "❌ --viewport applies to web gates only; on ios --device selects the simulator" >&2; exit 2; }
86
- [[ "$ACTIONS" =~ ^[1-9][0-9]*$ ]] || { echo "❌ --actions must be a positive integer" >&2; exit 2; }
87
+ [[ "$ACTIONS" =~ ^[0-9]+$ ]] || { echo "❌ --actions must be a non-negative integer (0 = flows-only gate: replay reviewed suites, no exploration)" >&2; exit 2; }
87
88
  [[ "$TIMEOUT" =~ ^[1-9][0-9]*$ ]] || { echo "❌ --timeout must be a positive integer" >&2; exit 2; }
88
89
  [[ "$FAIL_ON" == "gate" || "$FAIL_ON" == "absolute" || "$FAIL_ON" == "any" || "$FAIL_ON" == "high" || "$FAIL_ON" == "medium" ]] || { echo "❌ --fail-on must be gate|absolute|any|high|medium" >&2; exit 2; }
89
90
  if [[ -n "$PROJECT_DIR" ]]; then
@@ -328,6 +329,11 @@ if [[ "${#CONTRACT_FILES[@]}" -gt 0 ]]; then
328
329
  done
329
330
  fi
330
331
 
332
+ if [[ "$ACTIONS" == "0" && "${#FLOW_FILES[@]}" -eq 0 && "${#SCENARIO_FILES[@]}" -eq 0 && "${#CONTRACT_COMPILED_FILES[@]}" -eq 0 ]]; then
333
+ echo "❌ --actions 0 is a flows-only gate, but no Flows, Scenarios, or Contracts were selected — nothing would be verified. Commit suites under .tapp/ or pass --flows/--scenarios/--contracts." >&2
334
+ exit 2
335
+ fi
336
+
331
337
  [[ -n "$APP_PATH" && -d "$APP_PATH" ]] || { echo "❌ Required: --app <path/to/App.app> (a simulator build)" >&2; exit 2; }
332
338
  if [[ -z "$BUNDLE_ID" ]]; then
333
339
  BUNDLE_ID="$(/usr/libexec/PlistBuddy -c 'Print :CFBundleIdentifier' "$APP_PATH/Info.plist" 2>/dev/null || true)"
@@ -366,16 +372,23 @@ step "Install $BUNDLE_ID"
366
372
  xcrun simctl install "$UDID" "$APP_PATH" || { echo "❌ simctl install failed — is $APP_PATH a SIMULATOR build?" >&2; exit 1; }
367
373
 
368
374
  # ── Autonomous exploration (quick-capture builds the harness itself if needed).
369
- step "Explore ($ACTIONS actions, ${TIMEOUT}s watchdog)"
370
- set +e
375
+ # --actions 0 = flows-only: the reviewed suites are the whole deterministic surface. No
376
+ # explorer touches the app — the point, for targets wired to production backends (#11).
371
377
  CAPTURE_ROOT="${TAPP_HOME:-$ROOT}/captures"
372
378
  mkdir -p "$CAPTURE_ROOT"
373
379
  CAPTURE_DIR="$(mktemp -d "$CAPTURE_ROOT/ci.XXXXXX")"
374
- TAPP_CAPTURE_DIR="$CAPTURE_DIR" OCQA_PR_TARGET_JSON="$IOS_PR_TARGET_JSON" "$ROOT/scripts/quick-capture.sh" explore "$BUNDLE_ID" --actions "$ACTIONS" --timeout "$TIMEOUT"
375
- set -e
376
- MARKERS="$CAPTURE_DIR/ocqa-markers.txt"
377
- [[ -f "$MARKERS" ]] || { echo "❌ Exploration produced no markers ($MARKERS)" >&2; exit 1; }
378
- echo "Markers: $MARKERS"
380
+ MARKERS=""
381
+ if [[ "$ACTIONS" == "0" ]]; then
382
+ step "Explore — skipped (flows-only gate)"
383
+ else
384
+ step "Explore ($ACTIONS actions, ${TIMEOUT}s watchdog)"
385
+ set +e
386
+ TAPP_CAPTURE_DIR="$CAPTURE_DIR" OCQA_PR_TARGET_JSON="$IOS_PR_TARGET_JSON" "$ROOT/scripts/quick-capture.sh" explore "$BUNDLE_ID" --actions "$ACTIONS" --timeout "$TIMEOUT"
387
+ set -e
388
+ MARKERS="$CAPTURE_DIR/ocqa-markers.txt"
389
+ [[ -f "$MARKERS" ]] || { echo "❌ Exploration produced no markers ($MARKERS)" >&2; exit 1; }
390
+ echo "Markers: $MARKERS"
391
+ fi
379
392
 
380
393
  # ── Replay committed Flows (each failure becomes a gate reason).
381
394
  FLOW_LOG_ARGS=()
@@ -417,6 +430,8 @@ PR_PLAN_ARGS=()
417
430
  [[ -n "$PR_PLAN_PATH" ]] && PR_PLAN_ARGS=(--pr-plan "$PR_PLAN_PATH")
418
431
  TARGET_KEY_ARGS=()
419
432
  [[ -n "$TARGET_KEY" ]] && TARGET_KEY_ARGS=(--target-key "$TARGET_KEY")
420
- node "$ROOT/mcp-server/src/ci-report.js" --markers "$MARKERS" --fail-on "$FAIL_ON" \
433
+ MODE_ARGS=(--markers "$MARKERS")
434
+ [[ "$ACTIONS" == "0" ]] && MODE_ARGS=(--flows-only)
435
+ node "$ROOT/mcp-server/src/ci-report.js" ${MODE_ARGS[@]+"${MODE_ARGS[@]}"} --fail-on "$FAIL_ON" \
421
436
  --html-dir "$CAPTURE_DIR" --label "$BUNDLE_ID" \
422
437
  ${BASELINE_ARGS[@]+"${BASELINE_ARGS[@]}"} ${JSON_ARGS[@]+"${JSON_ARGS[@]}"} ${MD_ARGS[@]+"${MD_ARGS[@]}"} ${PR_PLAN_ARGS[@]+"${PR_PLAN_ARGS[@]}"} ${TARGET_KEY_ARGS[@]+"${TARGET_KEY_ARGS[@]}"} ${FLOW_LOG_ARGS[@]+"${FLOW_LOG_ARGS[@]}"}
@@ -18,7 +18,14 @@ def load_flow(path):
18
18
  raw = open(path, encoding="utf-8").read()
19
19
  if path.endswith(".json"):
20
20
  return json.loads(raw)
21
- import yaml
21
+ try:
22
+ import yaml
23
+ except ModuleNotFoundError:
24
+ # Hosted macOS runners ship python3 without PyYAML (field issue #16); a raw
25
+ # ModuleNotFoundError reads like a tapp crash instead of a one-line fix.
26
+ sys.exit("tapp flow replay needs PyYAML to read YAML Flows: run `python3 -m pip install pyyaml` "
27
+ "(on GitHub macOS runners: `pip3 install pyyaml`), or commit the Flow as .json. "
28
+ "`tapp doctor` checks this.")
22
29
  return yaml.safe_load(raw)
23
30
 
24
31
 
@@ -9,7 +9,7 @@ import { fileURLToPath } from "node:url";
9
9
  import { runQaAndroid, runQaWeb, startManagedWebTarget, stopManagedWebTarget } from "../mcp-server/src/index.js";
10
10
  import { runAndroidFlow } from "../mcp-server/src/android-flow.js";
11
11
  import { inferFlowPlatform, loadFlowFile } from "../mcp-server/src/flow-runtime.js";
12
- import { runWebFlow } from "../mcp-server/src/web-flow.js";
12
+ import { runWebFlow, resolveGateFlowUrl } from "../mcp-server/src/web-flow.js";
13
13
  import { runWebScenario, validateScenario } from "../mcp-server/src/scenario-runtime.js";
14
14
  import { compileReleaseContract, loadReleaseContractFile } from "../mcp-server/src/release-contract.js";
15
15
  import { prExplorationTargetsFromPlan } from "../mcp-server/src/pr-selection.js";
@@ -126,8 +126,16 @@ try {
126
126
  }
127
127
  if (args.platform === "web" && !args.url) {
128
128
  exitCode = 2;
129
+ } else if (args.actions === 0 && !selectedFlows.length && !selectedScenarios.length && !selectedContracts.length) {
130
+ console.error("❌ --actions 0 is a flows-only gate, but no Flows, Scenarios, or Contracts were selected — nothing would be verified.");
131
+ exitCode = 2;
129
132
  } else {
130
- const qa = args.platform === "web"
133
+ // --actions 0: flows-only — no explorer touches the target (it may be production, #11).
134
+ // The reviewed suites are the whole deterministic surface; ci-report gets --flows-only.
135
+ const flowsOnly = args.actions === 0;
136
+ const qa = flowsOnly
137
+ ? { structured: { capture: { path: fs.mkdtempSync(path.join(os.tmpdir(), `tapp-ci-${args.platform}-flows-only-`)) } } }
138
+ : args.platform === "web"
131
139
  ? await runQaWeb({ url: args.url, maxActions: args.actions, timeout: args.timeout, testEmail: process.env.OCQA_TEST_EMAIL, testPassword: process.env.OCQA_TEST_PASSWORD, seedTargets: prExplorationTargets, device: args.device || "", viewport: args.viewport || "" })
132
140
  : await runQaAndroid({ appId: args.appId, apkPath: args.apk, serial: args.serial, maxActions: args.actions, timeout: args.timeout,
133
141
  testEmail: process.env.OCQA_TEST_EMAIL, testPassword: process.env.OCQA_TEST_PASSWORD, seedTargets: prExplorationTargets });
@@ -142,7 +150,7 @@ try {
142
150
  const logPath = path.join(os.tmpdir(), `tapp-ci-${args.platform}-${path.basename(flowPath).replace(/\.ya?ml$/i, "")}-${Date.now()}.log`);
143
151
  const evidenceDir = path.join(captureDir, "flows", path.basename(flowPath).replace(/\.ya?ml$/i, ""));
144
152
  try {
145
- if (args.platform === "web") await runWebFlow({ flow, url: args.url, logPath, screenshotDir: evidenceDir });
153
+ if (args.platform === "web") await runWebFlow({ flow, url: args.url, urlIsFallback: true, logPath, screenshotDir: evidenceDir });
146
154
  else await runAndroidFlow({ flow, appId: args.appId, apkPath: undefined, serial: args.serial, logPath, screenshotDir: evidenceDir });
147
155
  } catch (error) {
148
156
  fs.writeFileSync(logPath, `OCQA_FLOW_RESULT:${JSON.stringify({ passed: false, total: flow.steps.length, failed: 1, error: error.message || String(error) })}\n`);
@@ -153,7 +161,7 @@ try {
153
161
  const logPath = path.join(os.tmpdir(), `tapp-ci-scenario-${path.basename(scenarioPath).replace(/\.ya?ml$/i, "")}-${Date.now()}.log`);
154
162
  const evidenceDir = path.join(captureDir, "scenarios", path.basename(scenarioPath).replace(/\.ya?ml$/i, ""));
155
163
  try {
156
- await runWebScenario({ scenario, url: args.url, logPath, screenshotDir: evidenceDir });
164
+ await runWebScenario({ scenario, url: resolveGateFlowUrl(scenario.url, args.url), logPath, screenshotDir: evidenceDir });
157
165
  } catch (error) {
158
166
  fs.writeFileSync(logPath, `OCQA_FLOW_RESULT:${JSON.stringify({ passed: false, name: scenario.name, kind: "scenario", total: scenario.steps.length, executed: 0, failed: 1, error: error.message || String(error) })}\n`);
159
167
  }
@@ -164,8 +172,8 @@ try {
164
172
  const logPath = path.join(os.tmpdir(), `tapp-ci-contract-${stem}-${Date.now()}.log`);
165
173
  const evidenceDir = path.join(captureDir, "contracts", stem);
166
174
  try {
167
- if (execution.kind === "scenario") await runWebScenario({ scenario: execution, url: args.url, logPath, screenshotDir: evidenceDir });
168
- else if (args.platform === "web") await runWebFlow({ flow: execution, url: args.url, logPath, screenshotDir: evidenceDir });
175
+ if (execution.kind === "scenario") await runWebScenario({ scenario: execution, url: resolveGateFlowUrl(execution.url, args.url), logPath, screenshotDir: evidenceDir });
176
+ else if (args.platform === "web") await runWebFlow({ flow: execution, url: args.url, urlIsFallback: true, logPath, screenshotDir: evidenceDir });
169
177
  else await runAndroidFlow({ flow: execution, appId: args.appId, apkPath: undefined, serial: args.serial, logPath, screenshotDir: evidenceDir });
170
178
  } catch (error) {
171
179
  fs.writeFileSync(logPath, `OCQA_FLOW_RESULT:${JSON.stringify({ passed: false, name: contract.title, kind: "release-contract", contract: contract.name, criticality: contract.criticality, total: execution.steps.length, executed: 0, failed: 1, error: error.message || String(error) })}\n`);
@@ -173,7 +181,8 @@ try {
173
181
  flowLogs.push({ kind: "contract", path: logPath });
174
182
  }
175
183
 
176
- const reportArgs = [path.join(root, "mcp-server", "src", "ci-report.js"), "--markers", markers, "--platform", args.platform,
184
+ const reportArgs = [path.join(root, "mcp-server", "src", "ci-report.js"),
185
+ ...(flowsOnly ? ["--flows-only"] : ["--markers", markers]), "--platform", args.platform,
177
186
  "--fail-on", args.failOn, "--html-dir", captureDir, "--label", args.url || args.appId];
178
187
  if (args.targetKey) reportArgs.push("--target-key", args.targetKey);
179
188
  if (args.baseline) reportArgs.push("--baseline", args.baseline);