browser_review_gate 0.1.1 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,27 +1,23 @@
1
1
  require "json"
2
2
 
3
3
  module BrowserReviewGate
4
- # Runs around an agent's shell commands.
5
- #
6
- # After a command that opened a PR, the browser run saved for this commit is published.
7
- # Before a command that opens a PR or asks for review, the agent is told when a browser run is still
8
- # owed, so the request is not made only to be taken back. It never fails the agent's session.
4
+ # Runs after an agent's shell command. When the command opened a PR, the browser run saved for this
5
+ # commit is published. It never holds a command back and never fails the agent: running the
6
+ # verification is the author's choice, and the gate on review requests lives in CI.
9
7
  class Hook
10
8
  OPEN_PR = "gh pr create"
11
- REVIEW_REQUEST = /gh pr (create|edit)\b.*(--reviewer|--add-reviewer|-r )|requested_reviewers/
12
- SKIP = "BROWSER_REVIEW_GATE_SKIP=1"
13
9
 
14
- def initialize(publisher_factory:, status_factory: nil, config: Config.new, shell: Shell.new)
10
+ def initialize(publisher_factory:, shell: Shell.new)
15
11
  @publisher_factory = publisher_factory
16
- @status_factory = status_factory
17
- @config = config
18
12
  @shell = shell
19
13
  end
20
14
 
21
15
  # `payload` is the raw hook input. Returns a message for the agent, or nil when nothing was done.
22
16
  def after(payload)
23
17
  return unless payload.to_s.include?(OPEN_PR)
24
- return unless saved_run?
18
+
19
+ sha = @shell.call("git", "rev-parse", "HEAD").strip
20
+ return unless SavedReport.new(sha: sha, shell: @shell).exist?
25
21
 
26
22
  verified = @publisher_factory.call.publish
27
23
  verified ? "Saved browser verification published; human review can be requested." :
@@ -30,58 +26,12 @@ module BrowserReviewGate
30
26
  "Saved browser verification was not published: #{error.message}"
31
27
  end
32
28
 
33
- # Returns the reason the command should wait, or nil to let it run.
34
- def before(payload)
35
- text = payload.to_s
36
- return if text.include?(SKIP)
37
-
38
- reason = review_request_reason(text) if text.match?(REVIEW_REQUEST)
39
- reason ||= open_pr_reason if text.include?(OPEN_PR)
40
- reason && "#{reason} If the person asked to go ahead anyway, run the same command prefixed with #{SKIP}."
41
- rescue Error
42
- nil
43
- end
44
-
45
- # Claude Code and Codex read a JSON object; Cursor reads its own shape.
46
- def self.render(message, agent, before: false)
29
+ # Claude Code and Codex read a JSON object; other agents show plain text.
30
+ def self.render(message, agent)
47
31
  return if message.nil?
48
- return JSON.generate(permission: "deny", agentMessage: message, userMessage: message) if before && agent == "cursor"
49
32
  return message unless %w[claude codex].include?(agent)
50
- return JSON.generate(hookSpecificOutput: { hookEventName: "PostToolUse", additionalContext: message }) unless before
51
-
52
- JSON.generate(hookSpecificOutput: { hookEventName: "PreToolUse", permissionDecision: "deny", permissionDecisionReason: message })
53
- end
54
-
55
- private
56
-
57
- def saved_run?
58
- SavedReport.new(sha: @shell.call("git", "rev-parse", "HEAD").strip, shell: @shell).exist?
59
- end
60
-
61
- # No PR exists yet, so CI has not decided; only the changed paths can tell.
62
- def open_pr_reason
63
- return if saved_run? || changed_files.all? { |path| @config.ignored?(path) }
64
-
65
- "No browser run is saved for this commit, and the change may alter browser behavior. " \
66
- "Run #{@config.command} first: the run is published when the PR is opened and review requests stay."
67
- end
68
-
69
- def review_request_reason(text)
70
- status = @status_factory.call.to_h(text[/gh pr edit\s+(\d+)/, 1]&.to_i)
71
- case status["next"]
72
- when "publish_saved" then "A browser run is saved but not published. Run `browser-review-gate publish` first, or the review request will be taken back."
73
- when "run_and_publish", "ask_to_start_app" then "This PR needs a browser run (#{@config.command} #{status["pull_request"]}); a review request made now will be taken back."
74
- end
75
- end
76
33
 
77
- # What the PR would contain. When no base branch can be found the change counts as unknown.
78
- def changed_files
79
- %w[origin/HEAD origin/main origin/master].each do |base|
80
- return @shell.call("git", "diff", "--name-only", "#{base}...HEAD").lines.map(&:chomp)
81
- rescue Error
82
- next
83
- end
84
- [ "unknown" ]
34
+ JSON.generate(hookSpecificOutput: { hookEventName: "PostToolUse", additionalContext: message })
85
35
  end
86
36
  end
87
37
  end
@@ -110,25 +110,23 @@ module BrowserReviewGate
110
110
 
111
111
  def install_claude
112
112
  managed(".claude/skills/browser-pr-verification/SKILL.md", render("claude_skill.md"))
113
- add_hooks(".claude/settings.json", "claude") do |hooks, command, before|
114
- (hooks[before ? "PreToolUse" : "PostToolUse"] ||= []) <<
115
- { "matcher" => "Bash", "hooks" => [ { "type" => "command", "command" => command, "timeout" => 60 } ] }
113
+ add_hook(".claude/settings.json", "claude", %w[PreToolUse]) do |hooks, command|
114
+ (hooks["PostToolUse"] ||= []) << { "matcher" => "Bash", "hooks" => [ { "type" => "command", "command" => command, "timeout" => 60 } ] }
116
115
  end
117
116
  end
118
117
 
119
118
  def install_cursor
120
119
  managed(".cursor/commands/browser-pr-verification.md", render("cursor_command.md"))
121
- add_hooks(".cursor/hooks.json", "cursor", "version" => 1) do |hooks, command, before|
122
- (hooks[before ? "beforeShellExecution" : "afterShellExecution"] ||= []) << { "command" => command }
120
+ add_hook(".cursor/hooks.json", "cursor", %w[beforeShellExecution], "version" => 1) do |hooks, command|
121
+ (hooks["afterShellExecution"] ||= []) << { "command" => command }
123
122
  end
124
123
  end
125
124
 
126
125
  # Codex runs a project hook only after the person trusts it in /hooks, so the AGENTS.md section also
127
126
  # tells the agent to publish after it opens the PR.
128
127
  def install_codex
129
- add_hooks(".codex/hooks.json", "codex") do |hooks, command, before|
130
- (hooks[before ? "PreToolUse" : "PostToolUse"] ||= []) <<
131
- { "matcher" => "^Bash$", "hooks" => [ { "type" => "command", "command" => command, "timeout" => 60 } ] }
128
+ add_hook(".codex/hooks.json", "codex", %w[PreToolUse]) do |hooks, command|
129
+ (hooks["PostToolUse"] ||= []) << { "matcher" => "^Bash$", "hooks" => [ { "type" => "command", "command" => command, "timeout" => 60 } ] }
132
130
  end
133
131
  path = "AGENTS.md"
134
132
  current = exist?(path) ? File.read(File.join(@root, path)) : nil
@@ -139,8 +137,8 @@ module BrowserReviewGate
139
137
 
140
138
  # A machine without the gem (a cloud agent, a fresh container) must not see a failing hook after
141
139
  # every command.
142
- def hook_command(agent, before)
143
- "command -v browser-review-gate >/dev/null 2>&1 && #{HOOK} #{agent}#{" --before" if before} || true"
140
+ def hook_command(agent)
141
+ "command -v browser-review-gate >/dev/null 2>&1 && #{HOOK} #{agent} || true"
144
142
  end
145
143
 
146
144
  def render(name)
@@ -161,14 +159,22 @@ module BrowserReviewGate
161
159
  write(path, content, current)
162
160
  end
163
161
 
164
- # Adds the after-command and before-command hook entries the file does not have yet.
165
- def add_hooks(path, agent, defaults = {})
162
+ # Adds the after-command hook entry when the file lacks it, and removes the before-command entries
163
+ # version 0.1.1 put under `stale_events`: nothing holds a command back any more.
164
+ def add_hook(path, agent, stale_events, defaults = {})
166
165
  current = exist?(path) ? File.read(File.join(@root, path)) : nil
167
- missing = [ false, true ].reject { |before| current.to_s.match?(before ? /hook #{agent} --before/ : /hook #{agent}(?! --before)/) }
168
- return @log.puts(" unchanged #{path}") if missing.empty?
166
+ present = current.to_s.match?(/hook #{agent}(?! --before)/)
167
+ return @log.puts(" unchanged #{path}") if present && !current.include?("hook #{agent} --before")
169
168
 
170
169
  settings = defaults.merge(current ? JSON.parse(current) : {})
171
- missing.each { |before| yield(settings["hooks"] ||= {}, hook_command(agent, before), before) }
170
+ hooks = (settings["hooks"] ||= {})
171
+ stale_events.each do |event|
172
+ next unless hooks[event]
173
+
174
+ hooks[event].reject! { |entry| JSON.generate(entry).include?("hook #{agent} --before") }
175
+ hooks.delete(event) if hooks[event].empty?
176
+ end
177
+ yield(hooks, hook_command(agent)) unless present
172
178
  write(path, "#{JSON.pretty_generate(settings)}\n", current)
173
179
  end
174
180
 
@@ -186,7 +192,7 @@ module BrowserReviewGate
186
192
  def next_steps
187
193
  <<~TEXT
188
194
  Next:
189
- 1. Add a CLAUDE_CODE_OAUTH_TOKEN or ANTHROPIC_API_KEY secret to the repository.
195
+ 1. Add one secret to the repository: CLAUDE_CODE_OAUTH_TOKEN, ANTHROPIC_API_KEY, OPENAI_API_KEY or GEMINI_API_KEY.
190
196
  2. Fill start_command, url and sign_in in #{Config::PATH}.
191
197
  3. Describe what this application always checks in #{Prompts::RULES}.
192
198
  4. Commit and merge to the default branch: the workflow runs from there.
@@ -0,0 +1,108 @@
1
+ require "json"
2
+ require "net/http"
3
+ require "uri"
4
+
5
+ module BrowserReviewGate
6
+ # One request to a model provider's HTTP API, with no tools and the answer bound to a JSON schema.
7
+ # For projects that assess with an API key instead of the Claude Code CLI.
8
+ class ModelApi
9
+ PROVIDERS = {
10
+ "anthropic" => { key: "ANTHROPIC_API_KEY", model: "claude-opus-5-5" },
11
+ "openai" => { key: "OPENAI_API_KEY" },
12
+ "gemini" => { key: "GEMINI_API_KEY" }
13
+ }.freeze
14
+ MAX_TOKENS = 16_000
15
+
16
+ # `transport` takes the URL, the headers and the request body, and returns [HTTP status, body].
17
+ def initialize(provider:, model: nil, transport: nil)
18
+ settings = PROVIDERS.fetch(provider) { raise Error, "Unknown model provider #{provider.inspect} (known: #{PROVIDERS.keys.join(", ")})" }
19
+ @provider = provider
20
+ @key_name = settings.fetch(:key)
21
+ @default_model = settings[:model]
22
+ @model = model.to_s.empty? ? @default_model : model
23
+ @transport = transport || method(:post)
24
+ raise Error, "Set `model` in #{Config::PATH} to assess with #{provider}" unless @model
25
+ end
26
+
27
+ # Returns the model's answer as JSON text. Raises Error when the call fails or gives no answer.
28
+ def complete(system:, user:, schema:)
29
+ key = ENV[@key_name].to_s
30
+ raise Error, "The model gave no answer: #{@key_name} is not set" if key.empty?
31
+
32
+ url, headers, body = send("#{@provider}_request", key, system, user, schema)
33
+ status, text = @transport.call(url, headers, body)
34
+ response = parse(text)
35
+ raise Error, "The model gave no answer: HTTP #{status} #{failure(response, text)}" unless status.to_i == 200
36
+
37
+ answer = parse(send("#{@provider}_answer", response))
38
+ return JSON.generate(answer) unless answer.empty?
39
+
40
+ raise Error, "The model gave no answer: #{failure(response, text)}"
41
+ end
42
+
43
+ private
44
+
45
+ # A request the provider's safeguards decline is retried on the provider's recommended fallback.
46
+ # That is asked for only with the built-in model: a model the project chose may not accept it.
47
+ def anthropic_request(key, system, user, schema)
48
+ headers = { "x-api-key" => key, "anthropic-version" => "2023-06-01" }
49
+ body = { model: @model, max_tokens: MAX_TOKENS, system: system, messages: [ { role: "user", content: user } ],
50
+ output_config: { format: { type: "json_schema", schema: schema } } }
51
+ if @model == @default_model
52
+ headers["anthropic-beta"] = "server-side-fallback-2026-07-01"
53
+ body[:fallbacks] = "default"
54
+ end
55
+ [ "https://api.anthropic.com/v1/messages", headers, body ]
56
+ end
57
+
58
+ def anthropic_answer(response)
59
+ return if response["stop_reason"] == "refusal"
60
+
61
+ Array(response["content"]).find { |block| block.is_a?(Hash) && block["type"] == "text" }&.fetch("text", nil)
62
+ end
63
+
64
+ def openai_request(key, system, user, schema)
65
+ body = { model: @model, messages: [ { role: "system", content: system }, { role: "user", content: user } ],
66
+ response_format: { type: "json_schema", json_schema: { name: "browser_assessment", schema: schema, strict: true } } }
67
+ [ "https://api.openai.com/v1/chat/completions", { "Authorization" => "Bearer #{key}" }, body ]
68
+ end
69
+
70
+ def openai_answer(response)
71
+ response.dig("choices", 0, "message", "content")
72
+ end
73
+
74
+ def gemini_request(key, system, user, schema)
75
+ body = { systemInstruction: { parts: [ { text: system } ] }, contents: [ { role: "user", parts: [ { text: user } ] } ],
76
+ generationConfig: { responseMimeType: "application/json", responseJsonSchema: schema } }
77
+ [ "https://generativelanguage.googleapis.com/v1beta/models/#{@model}:generateContent", { "x-goog-api-key" => key }, body ]
78
+ end
79
+
80
+ def gemini_answer(response)
81
+ Array(response.dig("candidates", 0, "content", "parts")).filter_map { |part| part["text"] if part.is_a?(Hash) }.join
82
+ end
83
+
84
+ def parse(text)
85
+ data = JSON.parse(text.to_s)
86
+ data.is_a?(Hash) ? data : {}
87
+ rescue JSON::ParserError
88
+ {}
89
+ end
90
+
91
+ # Providers explain an error in `error.message`; a declined request has no text to return.
92
+ def failure(response, text)
93
+ message = response.dig("error", "message") if response["error"].is_a?(Hash)
94
+ message ||= "the request was declined" if response["stop_reason"] == "refusal" || response.dig("choices", 0, "message", "refusal")
95
+ (message || text.to_s.strip)[0, 300]
96
+ end
97
+
98
+ def post(url, headers, body)
99
+ uri = URI(url)
100
+ request = Net::HTTP::Post.new(uri, headers.merge("Content-Type" => "application/json"))
101
+ request.body = JSON.generate(body)
102
+ response = Net::HTTP.start(uri.host, uri.port, use_ssl: true, open_timeout: 10, read_timeout: 300) { |http| http.request(request) }
103
+ [ response.code, response.body ]
104
+ rescue SystemCallError, SocketError, Timeout::Error, OpenSSL::SSL::SSLError => error
105
+ raise Error, "The model gave no answer: #{error.message}"
106
+ end
107
+ end
108
+ end
@@ -9,6 +9,22 @@ module BrowserReviewGate
9
9
  NPX = %w[npx --yes @anthropic-ai/claude-code@2.1.288].freeze
10
10
  CREDENTIALS = %w[CLAUDE_CODE_OAUTH_TOKEN ANTHROPIC_API_KEY].freeze
11
11
 
12
+ # The client the project's settings and the credentials at hand call for. The Claude Code CLI stays
13
+ # the default; an OpenAI or Gemini key alone selects that provider's API.
14
+ def self.for(config, environment: ENV)
15
+ provider = config.provider || detect(config, environment)
16
+ return new(command: config.claude_command, model: config.model) if provider == "claude-cli"
17
+
18
+ ModelApi.new(provider: provider, model: config.model)
19
+ end
20
+
21
+ def self.detect(config, environment)
22
+ set = ->(name) { !environment[name].to_s.empty? }
23
+ return "claude-cli" if config.claude_command || CREDENTIALS.any?(&set)
24
+
25
+ ModelApi::PROVIDERS.keys.find { |provider| set.call(ModelApi::PROVIDERS.fetch(provider).fetch(:key)) } || "claude-cli"
26
+ end
27
+
12
28
  # `runner` takes the environment, argv and stdin, and returns [stdout, stderr, success?].
13
29
  def initialize(command: nil, model: nil, runner: nil)
14
30
  @command = command.to_s.empty? ? default_command : command.split
@@ -49,21 +49,29 @@ module BrowserReviewGate
49
49
 
50
50
  PROJECT_RULES_NOTE = <<~TEXT.strip.freeze
51
51
  The maintainers of this repository wrote these rules for their application. They refine the
52
- decision above and win where they disagree with it. Use the scenarios they name when you judge
53
- whether an earlier report covers the change.
52
+ decision above and win where they disagree with it. Include the scenarios they name for the areas
53
+ the change touches.
54
54
  TEXT
55
55
 
56
56
  FRAME_TAIL = <<~TEXT.strip.freeze
57
- ## Coverage
57
+ ## Scenarios
58
58
 
59
- `covered_by_report` is true only when `report_cases` is present and those cases already exercise
60
- every browser-visible behavior this pull request changes, including what the project rules require
61
- for the areas it touches. Otherwise it is false. It is false when the decision is `not-required`.
59
+ When the decision is `required`, `scenarios` lists what a person must do in the browser to exercise
60
+ every behavior this pull request changes, including what the project rules require for the areas
61
+ it touches. One entry per scenario, at most 12, the fewest that cover the change. `name` says what
62
+ to do and what must be seen, in one sentence. `id` is a short label such as `SIGN-IN-1`: letters,
63
+ digits and hyphens.
64
+
65
+ When `report_cases` is present, those are the cases of an earlier browser run. If one of them
66
+ already exercises a scenario, give that scenario the `id` of that case, exactly. Give every other
67
+ scenario an `id` no case uses. Never reuse the `id` of a case that does not exercise the scenario.
68
+
69
+ When the decision is `not-required`, `scenarios` is an empty list.
62
70
 
63
71
  ## Reason
64
72
 
65
73
  `reason` is one plain sentence of at most 30 words. No links, mentions, code, or HTML. No company
66
- or customer names.
74
+ or customer names. The same holds for scenario names.
67
75
  TEXT
68
76
 
69
77
  private
@@ -30,7 +30,7 @@ module BrowserReviewGate
30
30
  actor = @github.login
31
31
  data["verified_by"] = "AI browser agent (run by #{actor})"
32
32
 
33
- comments = @github.comments(number)
33
+ comments = @github.trusted_comments(number)
34
34
  ensure_browser_run_wanted!(number, comments, local_sha)
35
35
  report = merged_report(data, comments)
36
36
 
@@ -54,7 +54,7 @@ module BrowserReviewGate
54
54
  data = read_report_file
55
55
  ensure_report_sha!(data, local_sha)
56
56
  report = Report.new(data.merge("verified_by" => "pending"))
57
- raise Error, report.errors.join("; ") unless report.valid?
57
+ ensure_acceptable!(report)
58
58
 
59
59
  saved_report = SavedReport.new(sha: local_sha, shell: @shell)
60
60
  saved_report.write(data)
@@ -107,7 +107,7 @@ module BrowserReviewGate
107
107
 
108
108
  def merged_report(data, comments)
109
109
  submitted = Report.new(data)
110
- raise Error, submitted.errors.join("; ") unless submitted.valid?
110
+ ensure_acceptable!(submitted)
111
111
 
112
112
  prior = Report.latest_comment(comments)
113
113
  report = Report.merge(prior ? Report.data_from(prior["body"]) : {}, submitted.to_h)
@@ -116,6 +116,12 @@ module BrowserReviewGate
116
116
  report
117
117
  end
118
118
 
119
+ # A run is taken only with the evidence of each case: the page and what was observed there.
120
+ def ensure_acceptable!(report)
121
+ problems = report.valid? ? report.evidence_errors : report.errors
122
+ raise Error, problems.join("; ") if problems.any?
123
+ end
124
+
119
125
  def body(report, complete)
120
126
  return report.markdown if complete
121
127
 
@@ -1,3 +1,4 @@
1
+ require "digest"
1
2
  require "json"
2
3
 
3
4
  module BrowserReviewGate
@@ -7,6 +8,7 @@ module BrowserReviewGate
7
8
  SHA_PATTERN = /\A\h{7,40}\z/
8
9
  CASE_ID_PATTERN = /\A[A-Za-z0-9][A-Za-z0-9_.-]{0,63}\z/
9
10
  RESULTS = %w[pass fail].freeze
11
+ URL_PATTERN = %r{\A(?:https?://|/)\S*\z}
10
12
 
11
13
  attr_reader :errors
12
14
 
@@ -21,24 +23,32 @@ module BrowserReviewGate
21
23
  raise Error, "Existing browser verification report has a malformed data marker"
22
24
  end
23
25
 
24
- # The newest passing report on the PR, or nil.
25
- def self.latest_passing(comments)
26
+ # The newest report on the PR when it is well formed, or nil.
27
+ def self.latest(comments)
26
28
  comment = latest_comment(comments)
27
29
  report = comment && new(data_from(comment["body"]))
28
- report if report&.passed?
30
+ report if report&.valid?
29
31
  rescue Error
30
32
  nil
31
33
  end
32
34
 
33
- # Later results replace earlier ones case by case; cases that were not run again are kept.
35
+ # The newest report on the PR when it passes, or nil.
36
+ def self.latest_passing(comments)
37
+ report = latest(comments)
38
+ report if report&.passed?
39
+ end
40
+
41
+ # Later results replace earlier ones case by case; cases that were not run again are kept. The
42
+ # merged report counts the runs published on the PR and how many of them had a failing case.
34
43
  def self.merge(previous_data, latest_data)
35
44
  previous = new(previous_data)
36
45
  latest = new(latest_data)
37
- return latest unless previous.valid? && latest.valid?
46
+ return latest unless latest.valid?
38
47
 
39
- combined = previous.cases.to_h { |test_case| [ test_case["id"], test_case ] }
48
+ combined = previous.valid? ? previous.cases.to_h { |test_case| [ test_case["id"], test_case ] } : {}
40
49
  latest.cases.each { |test_case| combined[test_case["id"]] = test_case }
41
- new(latest.to_h.merge("cases" => combined.values))
50
+ failed = latest.cases.all? { |test_case| test_case["result"] == "pass" } ? 0 : 1
51
+ new(latest.to_h.merge("cases" => combined.values, "runs" => previous.runs + 1, "failed_runs" => previous.failed_runs + failed))
42
52
  end
43
53
 
44
54
  def initialize(data)
@@ -53,7 +63,28 @@ module BrowserReviewGate
53
63
  def verified_by = @data["verified_by"]
54
64
  def cases = @data["cases"].is_a?(Array) ? @data["cases"] : []
55
65
  def excluded = @data["excluded"].is_a?(Array) ? @data["excluded"] : []
66
+ def runs = @data["runs"].is_a?(Integer) ? @data["runs"] : 0
67
+ def failed_runs = @data["failed_runs"].is_a?(Integer) ? @data["failed_runs"] : 0
56
68
  def to_h = JSON.parse(JSON.generate(@data))
69
+ def passed_ids = cases.select { |test_case| test_case["result"] == "pass" }.map { |test_case| test_case["id"] }
70
+
71
+ # Changes whenever a case is added, re-run or changes its outcome.
72
+ def fingerprint
73
+ Digest::SHA256.hexdigest(JSON.generate(cases.map { |test_case| test_case.values_at("id", "result", "tested_sha") }))[0, 16]
74
+ end
75
+
76
+ # What a newly submitted run must say about each case: where it was exercised and what was seen.
77
+ # Cases carried over from reports published before this was required are not checked.
78
+ def evidence_errors
79
+ cases.each_with_index.flat_map do |test_case, index|
80
+ next [] unless test_case.is_a?(Hash)
81
+
82
+ found = []
83
+ found << "case #{index + 1} url must be the page that was exercised: an http(s) URL or a path starting with /" unless url?(test_case["url"])
84
+ found << "case #{index + 1} details must say what was observed" unless text?(test_case["details"], 500)
85
+ found
86
+ end
87
+ end
57
88
 
58
89
  def passed?
59
90
  valid? && @data["coverage_complete"] && cases.all? { |test_case| test_case["result"] == "pass" }
@@ -122,6 +153,7 @@ module BrowserReviewGate
122
153
  errors << "case #{position} result must be pass or fail" unless RESULTS.include?(test_case["result"])
123
154
  errors << "case #{position} tested_sha must be a 7–40 character hexadecimal commit SHA" unless sha?(test_case["tested_sha"])
124
155
  errors << "case #{position} verified_by must be a non-empty string of at most 100 characters" unless text?(test_case["verified_by"], 100)
156
+ errors << "case #{position} url must be a string of at most 300 characters" unless test_case["url"].nil? || url?(test_case["url"])
125
157
  details = test_case["details"]
126
158
  errors << "case #{position} details must be a string of at most 500 characters" unless details.nil? || (details.is_a?(String) && details.length <= 500)
127
159
  end
@@ -140,10 +172,12 @@ module BrowserReviewGate
140
172
 
141
173
  def sha?(value) = value.is_a?(String) && SHA_PATTERN.match?(value)
142
174
  def text?(value, limit) = value.is_a?(String) && !value.strip.empty? && value.length <= limit
175
+ def url?(value) = value.is_a?(String) && value.length <= 300 && URL_PATTERN.match?(value)
143
176
 
144
177
  def markdown_case(test_case)
145
178
  detail = test_case["details"].to_s.empty? ? "" : ": #{Markdown.escape(test_case["details"])}"
146
- "- #{test_case["result"].upcase} — #{Markdown.escape(test_case["name"])} (`#{test_case["id"]}`, tested " \
179
+ place = test_case["url"].to_s.empty? ? "" : " at #{Markdown.escape(test_case["url"])}"
180
+ "- #{test_case["result"].upcase} — #{Markdown.escape(test_case["name"])}#{place} (`#{test_case["id"]}`, tested " \
147
181
  "`#{test_case["tested_sha"]}` by #{Markdown.escape(test_case["verified_by"])})#{detail}"
148
182
  end
149
183
 
@@ -0,0 +1,34 @@
1
+ module BrowserReviewGate
2
+ # What the gate did on recent pull requests, read back from its own comments and labels.
3
+ class Stats
4
+ def initialize(github:, config: Config.new)
5
+ @github = github
6
+ @config = config
7
+ end
8
+
9
+ def to_h(since:)
10
+ counts = Hash.new(0)
11
+ @github.pull_requests_updated_since(since).each { |pull_request| count(pull_request, counts) }
12
+ %w[pull_requests assessed required verified waived owed found_a_failure].to_h { |key| [ key, counts[key] ] }
13
+ end
14
+
15
+ private
16
+
17
+ def count(pull_request, counts)
18
+ counts["pull_requests"] += 1
19
+ comments = @github.trusted_comments(pull_request.fetch("number"))
20
+ assessment = Assessment.latest(comments)
21
+ return unless assessment
22
+
23
+ counts["assessed"] += 1
24
+ return unless assessment.required?
25
+
26
+ labels = Array(pull_request["labels"]).map { |label| label["name"] }
27
+ counts["required"] += 1
28
+ counts["verified"] += 1 if labels.include?(@config.label)
29
+ counts["waived"] += 1 if labels.include?(@config.waiver_label)
30
+ counts["owed"] += 1 if pull_request["state"] == "open" && !labels.intersect?([ @config.label, @config.waiver_label ])
31
+ counts["found_a_failure"] += 1 if Report.latest(comments)&.failed_runs.to_i.positive?
32
+ end
33
+ end
34
+ end
@@ -8,7 +8,8 @@ module BrowserReviewGate
8
8
  RUN_STEPS = %w[run_and_publish run_and_save].freeze
9
9
 
10
10
  # `probe` takes the app URL and says whether something answers there.
11
- def initialize(github:, config: Config.new, shell: Shell.new, probe: nil)
11
+ def initialize(github:, config: Config.new, shell: Shell.new, probe: nil, prompts: Prompts.new)
12
+ @prompts = prompts
12
13
  @github = github
13
14
  @config = config
14
15
  @shell = shell
@@ -17,7 +18,7 @@ module BrowserReviewGate
17
18
 
18
19
  def to_h(number = nil)
19
20
  result = facts(number)
20
- result["project_rules"] = Prompts::RULES if Prompts.new.rules
21
+ result["project_rules"] = Prompts::RULES if @prompts.rules
21
22
  result["app"] = app
22
23
  # The agent never starts the app itself: a person decides what runs on their machine.
23
24
  result["next"] = "ask_to_start_app" if RUN_STEPS.include?(result["next"]) && result["app"]["running"] == false
@@ -33,16 +34,19 @@ module BrowserReviewGate
33
34
  return { "pull_request" => nil, "local_head" => local_sha, "saved_run" => saved, "next" => saved ? "open_pull_request" : "run_and_save" } unless number
34
35
 
35
36
  pull_request = @github.pull_request(number)
36
- comments = @github.comments(number)
37
+ comments = @github.trusted_comments(number)
37
38
  head_sha = pull_request.fetch("head").fetch("sha")
38
39
  assessment = Assessment.latest(comments)
39
40
  assessment = nil unless assessment&.for?(head_sha)
40
41
  labels = pull_request.fetch("labels").map { |label| label["name"] }
41
- verified = labels.include?(@config.label) && !Report.latest_passing(comments).nil?
42
+ report = Report.latest(comments)
43
+ verified = labels.include?(@config.label) && (assessment ? assessment.covered_by?(report) : report&.passed? == true)
42
44
  waived = labels.include?(@config.waiver_label)
43
45
 
44
46
  { "pull_request" => number, "pr_head" => head_sha, "local_head" => local_sha,
45
- "decision" => assessment&.decision, "reason" => assessment&.reason, "verified" => verified, "waived" => waived,
47
+ "decision" => assessment&.decision, "reason" => assessment&.reason,
48
+ # What the run must exercise; each id is the `id` of the case that reports it.
49
+ "scenarios" => assessment ? assessment.outstanding(report) : [], "verified" => verified, "waived" => waived,
46
50
  "saved_run" => saved, "next" => next_step(assessment, verified || waived, saved, head_sha == local_sha) }
47
51
  end
48
52
 
@@ -6,6 +6,12 @@ url:<%= url ? " #{url.inspect}" : "" %>
6
6
  # How to sign in locally, e.g. "use the seed account from db/seeds.rb". No real credentials here.
7
7
  sign_in:
8
8
 
9
+ # Who assesses a PR in CI. Unset: the Claude Code CLI with a CLAUDE_CODE_OAUTH_TOKEN or
10
+ # ANTHROPIC_API_KEY secret; with only an OPENAI_API_KEY or GEMINI_API_KEY secret, that provider's API.
11
+ # Set `provider` to claude-cli, anthropic, openai or gemini to choose; openai and gemini need `model`.
12
+ provider:
13
+ model:
14
+
9
15
  # Workflow file name under .github/workflows, used to request an assessment by hand.
10
16
  workflow: <%= workflow %>
11
17
 
@@ -15,10 +15,13 @@ This file is the playbook for the AI coding agent that does the run. The person
15
15
  - `open_pull_request`: a run is saved and waits for the PR. Tell the person to open it.
16
16
  - `ask_to_start_app`: the app is not running. Do not start it yourself: ask the person to start it (`app.start_command` says how) and stop until they have.
17
17
  - `run_and_publish` or `run_and_save`: continue with step 2.
18
- 2. Read the diff against the base branch and the earlier report in the PR, if any. Then read `<%= rules_path %>`: it says what this project always checks and which scenarios each area needs, and its rules are mandatory. List the scenarios for the behavior that changed, plus the ones the project rules require for the areas the change touches. Do not re-run behavior an earlier passing case still covers.
18
+ 2. Decide what to run.
19
+ - `scenarios` in the status output is the list CI wrote for this PR. Run every one of them, and give each case the `id` of its scenario. You may add cases of your own; you may not drop or rename one from the list.
20
+ - With no PR yet, or an empty list, choose the scenarios yourself: read the diff against the base branch, then `<%= rules_path %>`, which says what this project always checks and which scenarios each area needs. Its rules are mandatory. CI compares your cases with its own list once the PR is open and asks for what is missing.
21
+ - Do not re-run behavior an earlier passing case still covers.
19
22
  3. Open the running app in a browser tool. `app` in the status output gives the URL and a sign-in hint. Never start, restart or stop the app yourself: what runs on the person's machine is their decision. Use local test data and do not destroy existing data. If the app, the data, or a browser tool is missing, report the blocker. Reading code and running unit tests does not replace a browser run.
20
- 4. Run every scenario end to end. After key actions check the page state, console errors, and failed requests. Record what you observed. A skipped or blocked scenario is not a pass.
21
- 5. Write the report as JSON outside the repository (see the shape below). Set `coverage_complete` to true only when every scenario from step 2 ran.
23
+ 4. Run every scenario end to end. After key actions check the page state, console errors, and failed requests. For each case record the page it was exercised on (`url`) and what you observed there, including console errors and failed requests (`details`). A case without both is refused. A skipped or blocked scenario is not a pass.
24
+ 5. Write the report as JSON outside the repository (see the shape below). Set `coverage_complete` to true only when every scenario from step 2 ran. CI checks the report against its list by case `id`: a missing scenario takes the label away again.
22
25
  6. Publish or save, from the checkout that was tested, with everything committed and pushed:
23
26
  - PR exists: `browser-review-gate publish --report PATH` (add `--pr NUMBER` when the branch has several). It posts the cases with their outcomes. Only a complete all-pass report adds the `<%= label %>` label.
24
27
  - No PR yet: `browser-review-gate save --report PATH`. The run is published when the PR for this commit is opened. A new commit makes it stale.
@@ -36,7 +39,7 @@ Some scenarios cannot be exercised locally, for example a sign-in through an ext
36
39
  "tested_sha": "<full commit SHA that was tested>",
37
40
  "coverage_complete": true,
38
41
  "cases": [
39
- { "id": "BROWSER-1", "name": "The changed interaction completes", "result": "pass", "details": "What was observed." }
42
+ { "id": "BROWSER-1", "name": "The changed interaction completes", "result": "pass", "url": "/the/page/exercised", "details": "What was observed, with console errors and failed requests if any." }
40
43
  ],
41
44
  "excluded": [
42
45
  { "name": "Sign-in through an external provider", "reason": "Needs a real provider account; covered by integration tests." }
@@ -44,7 +47,7 @@ Some scenarios cannot be exercised locally, for example a sign-in through an ext
44
47
  }
45
48
  ```
46
49
 
47
- `result` is `pass` or `fail`. `excluded` is optional: list a scenario there only when the person decided it stays outside the local run. Never exclude a scenario on your own to make the run complete. Keep customer and real company names out of the report.
50
+ `result` is `pass` or `fail`. `url` is an http(s) URL or a path starting with `/`. `excluded` is optional: list a scenario there only when the person decided it stays outside the local run. Never exclude a scenario on your own to make the run complete. Keep customer and real company names out of the report.
48
51
 
49
52
  ## Rules
50
53