browser_review_gate 0.1.1 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +27 -0
- data/README.md +28 -9
- data/lib/browser_review_gate/assessment.rb +68 -16
- data/lib/browser_review_gate/assessor.rb +19 -19
- data/lib/browser_review_gate/cli.rb +39 -12
- data/lib/browser_review_gate/config.rb +2 -0
- data/lib/browser_review_gate/gate.rb +6 -6
- data/lib/browser_review_gate/github.rb +24 -0
- data/lib/browser_review_gate/hook.rb +10 -60
- data/lib/browser_review_gate/installer.rb +22 -16
- data/lib/browser_review_gate/model_api.rb +108 -0
- data/lib/browser_review_gate/model_client.rb +16 -0
- data/lib/browser_review_gate/prompts.rb +15 -7
- data/lib/browser_review_gate/publisher.rb +9 -3
- data/lib/browser_review_gate/report.rb +42 -8
- data/lib/browser_review_gate/stats.rb +34 -0
- data/lib/browser_review_gate/status.rb +9 -5
- data/lib/browser_review_gate/templates/config.yml.erb +6 -0
- data/lib/browser_review_gate/templates/playbook.md.erb +8 -5
- data/lib/browser_review_gate/templates/workflow.yml.erb +8 -4
- data/lib/browser_review_gate/version.rb +1 -1
- data/lib/browser_review_gate.rb +2 -0
- metadata +4 -2
|
@@ -1,27 +1,23 @@
|
|
|
1
1
|
require "json"
|
|
2
2
|
|
|
3
3
|
module BrowserReviewGate
|
|
4
|
-
# Runs
|
|
5
|
-
#
|
|
6
|
-
#
|
|
7
|
-
# Before a command that opens a PR or asks for review, the agent is told when a browser run is still
|
|
8
|
-
# owed, so the request is not made only to be taken back. It never fails the agent's session.
|
|
4
|
+
# Runs after an agent's shell command. When the command opened a PR, the browser run saved for this
|
|
5
|
+
# commit is published. It never holds a command back and never fails the agent: running the
|
|
6
|
+
# verification is the author's choice, and the gate on review requests lives in CI.
|
|
9
7
|
class Hook
|
|
10
8
|
OPEN_PR = "gh pr create"
|
|
11
|
-
REVIEW_REQUEST = /gh pr (create|edit)\b.*(--reviewer|--add-reviewer|-r )|requested_reviewers/
|
|
12
|
-
SKIP = "BROWSER_REVIEW_GATE_SKIP=1"
|
|
13
9
|
|
|
14
|
-
def initialize(publisher_factory:,
|
|
10
|
+
def initialize(publisher_factory:, shell: Shell.new)
|
|
15
11
|
@publisher_factory = publisher_factory
|
|
16
|
-
@status_factory = status_factory
|
|
17
|
-
@config = config
|
|
18
12
|
@shell = shell
|
|
19
13
|
end
|
|
20
14
|
|
|
21
15
|
# `payload` is the raw hook input. Returns a message for the agent, or nil when nothing was done.
|
|
22
16
|
def after(payload)
|
|
23
17
|
return unless payload.to_s.include?(OPEN_PR)
|
|
24
|
-
|
|
18
|
+
|
|
19
|
+
sha = @shell.call("git", "rev-parse", "HEAD").strip
|
|
20
|
+
return unless SavedReport.new(sha: sha, shell: @shell).exist?
|
|
25
21
|
|
|
26
22
|
verified = @publisher_factory.call.publish
|
|
27
23
|
verified ? "Saved browser verification published; human review can be requested." :
|
|
@@ -30,58 +26,12 @@ module BrowserReviewGate
|
|
|
30
26
|
"Saved browser verification was not published: #{error.message}"
|
|
31
27
|
end
|
|
32
28
|
|
|
33
|
-
#
|
|
34
|
-
def
|
|
35
|
-
text = payload.to_s
|
|
36
|
-
return if text.include?(SKIP)
|
|
37
|
-
|
|
38
|
-
reason = review_request_reason(text) if text.match?(REVIEW_REQUEST)
|
|
39
|
-
reason ||= open_pr_reason if text.include?(OPEN_PR)
|
|
40
|
-
reason && "#{reason} If the person asked to go ahead anyway, run the same command prefixed with #{SKIP}."
|
|
41
|
-
rescue Error
|
|
42
|
-
nil
|
|
43
|
-
end
|
|
44
|
-
|
|
45
|
-
# Claude Code and Codex read a JSON object; Cursor reads its own shape.
|
|
46
|
-
def self.render(message, agent, before: false)
|
|
29
|
+
# Claude Code and Codex read a JSON object; other agents show plain text.
|
|
30
|
+
def self.render(message, agent)
|
|
47
31
|
return if message.nil?
|
|
48
|
-
return JSON.generate(permission: "deny", agentMessage: message, userMessage: message) if before && agent == "cursor"
|
|
49
32
|
return message unless %w[claude codex].include?(agent)
|
|
50
|
-
return JSON.generate(hookSpecificOutput: { hookEventName: "PostToolUse", additionalContext: message }) unless before
|
|
51
|
-
|
|
52
|
-
JSON.generate(hookSpecificOutput: { hookEventName: "PreToolUse", permissionDecision: "deny", permissionDecisionReason: message })
|
|
53
|
-
end
|
|
54
|
-
|
|
55
|
-
private
|
|
56
|
-
|
|
57
|
-
def saved_run?
|
|
58
|
-
SavedReport.new(sha: @shell.call("git", "rev-parse", "HEAD").strip, shell: @shell).exist?
|
|
59
|
-
end
|
|
60
|
-
|
|
61
|
-
# No PR exists yet, so CI has not decided; only the changed paths can tell.
|
|
62
|
-
def open_pr_reason
|
|
63
|
-
return if saved_run? || changed_files.all? { |path| @config.ignored?(path) }
|
|
64
|
-
|
|
65
|
-
"No browser run is saved for this commit, and the change may alter browser behavior. " \
|
|
66
|
-
"Run #{@config.command} first: the run is published when the PR is opened and review requests stay."
|
|
67
|
-
end
|
|
68
|
-
|
|
69
|
-
def review_request_reason(text)
|
|
70
|
-
status = @status_factory.call.to_h(text[/gh pr edit\s+(\d+)/, 1]&.to_i)
|
|
71
|
-
case status["next"]
|
|
72
|
-
when "publish_saved" then "A browser run is saved but not published. Run `browser-review-gate publish` first, or the review request will be taken back."
|
|
73
|
-
when "run_and_publish", "ask_to_start_app" then "This PR needs a browser run (#{@config.command} #{status["pull_request"]}); a review request made now will be taken back."
|
|
74
|
-
end
|
|
75
|
-
end
|
|
76
33
|
|
|
77
|
-
|
|
78
|
-
def changed_files
|
|
79
|
-
%w[origin/HEAD origin/main origin/master].each do |base|
|
|
80
|
-
return @shell.call("git", "diff", "--name-only", "#{base}...HEAD").lines.map(&:chomp)
|
|
81
|
-
rescue Error
|
|
82
|
-
next
|
|
83
|
-
end
|
|
84
|
-
[ "unknown" ]
|
|
34
|
+
JSON.generate(hookSpecificOutput: { hookEventName: "PostToolUse", additionalContext: message })
|
|
85
35
|
end
|
|
86
36
|
end
|
|
87
37
|
end
|
|
@@ -110,25 +110,23 @@ module BrowserReviewGate
|
|
|
110
110
|
|
|
111
111
|
def install_claude
|
|
112
112
|
managed(".claude/skills/browser-pr-verification/SKILL.md", render("claude_skill.md"))
|
|
113
|
-
|
|
114
|
-
(hooks[
|
|
115
|
-
{ "matcher" => "Bash", "hooks" => [ { "type" => "command", "command" => command, "timeout" => 60 } ] }
|
|
113
|
+
add_hook(".claude/settings.json", "claude", %w[PreToolUse]) do |hooks, command|
|
|
114
|
+
(hooks["PostToolUse"] ||= []) << { "matcher" => "Bash", "hooks" => [ { "type" => "command", "command" => command, "timeout" => 60 } ] }
|
|
116
115
|
end
|
|
117
116
|
end
|
|
118
117
|
|
|
119
118
|
def install_cursor
|
|
120
119
|
managed(".cursor/commands/browser-pr-verification.md", render("cursor_command.md"))
|
|
121
|
-
|
|
122
|
-
(hooks[
|
|
120
|
+
add_hook(".cursor/hooks.json", "cursor", %w[beforeShellExecution], "version" => 1) do |hooks, command|
|
|
121
|
+
(hooks["afterShellExecution"] ||= []) << { "command" => command }
|
|
123
122
|
end
|
|
124
123
|
end
|
|
125
124
|
|
|
126
125
|
# Codex runs a project hook only after the person trusts it in /hooks, so the AGENTS.md section also
|
|
127
126
|
# tells the agent to publish after it opens the PR.
|
|
128
127
|
def install_codex
|
|
129
|
-
|
|
130
|
-
(hooks[
|
|
131
|
-
{ "matcher" => "^Bash$", "hooks" => [ { "type" => "command", "command" => command, "timeout" => 60 } ] }
|
|
128
|
+
add_hook(".codex/hooks.json", "codex", %w[PreToolUse]) do |hooks, command|
|
|
129
|
+
(hooks["PostToolUse"] ||= []) << { "matcher" => "^Bash$", "hooks" => [ { "type" => "command", "command" => command, "timeout" => 60 } ] }
|
|
132
130
|
end
|
|
133
131
|
path = "AGENTS.md"
|
|
134
132
|
current = exist?(path) ? File.read(File.join(@root, path)) : nil
|
|
@@ -139,8 +137,8 @@ module BrowserReviewGate
|
|
|
139
137
|
|
|
140
138
|
# A machine without the gem (a cloud agent, a fresh container) must not see a failing hook after
|
|
141
139
|
# every command.
|
|
142
|
-
def hook_command(agent
|
|
143
|
-
"command -v browser-review-gate >/dev/null 2>&1 && #{HOOK} #{agent}
|
|
140
|
+
def hook_command(agent)
|
|
141
|
+
"command -v browser-review-gate >/dev/null 2>&1 && #{HOOK} #{agent} || true"
|
|
144
142
|
end
|
|
145
143
|
|
|
146
144
|
def render(name)
|
|
@@ -161,14 +159,22 @@ module BrowserReviewGate
|
|
|
161
159
|
write(path, content, current)
|
|
162
160
|
end
|
|
163
161
|
|
|
164
|
-
# Adds the after-command
|
|
165
|
-
|
|
162
|
+
# Adds the after-command hook entry when the file lacks it, and removes the before-command entries
|
|
163
|
+
# version 0.1.1 put under `stale_events`: nothing holds a command back any more.
|
|
164
|
+
def add_hook(path, agent, stale_events, defaults = {})
|
|
166
165
|
current = exist?(path) ? File.read(File.join(@root, path)) : nil
|
|
167
|
-
|
|
168
|
-
return @log.puts(" unchanged #{path}") if
|
|
166
|
+
present = current.to_s.match?(/hook #{agent}(?! --before)/)
|
|
167
|
+
return @log.puts(" unchanged #{path}") if present && !current.include?("hook #{agent} --before")
|
|
169
168
|
|
|
170
169
|
settings = defaults.merge(current ? JSON.parse(current) : {})
|
|
171
|
-
|
|
170
|
+
hooks = (settings["hooks"] ||= {})
|
|
171
|
+
stale_events.each do |event|
|
|
172
|
+
next unless hooks[event]
|
|
173
|
+
|
|
174
|
+
hooks[event].reject! { |entry| JSON.generate(entry).include?("hook #{agent} --before") }
|
|
175
|
+
hooks.delete(event) if hooks[event].empty?
|
|
176
|
+
end
|
|
177
|
+
yield(hooks, hook_command(agent)) unless present
|
|
172
178
|
write(path, "#{JSON.pretty_generate(settings)}\n", current)
|
|
173
179
|
end
|
|
174
180
|
|
|
@@ -186,7 +192,7 @@ module BrowserReviewGate
|
|
|
186
192
|
def next_steps
|
|
187
193
|
<<~TEXT
|
|
188
194
|
Next:
|
|
189
|
-
1. Add
|
|
195
|
+
1. Add one secret to the repository: CLAUDE_CODE_OAUTH_TOKEN, ANTHROPIC_API_KEY, OPENAI_API_KEY or GEMINI_API_KEY.
|
|
190
196
|
2. Fill start_command, url and sign_in in #{Config::PATH}.
|
|
191
197
|
3. Describe what this application always checks in #{Prompts::RULES}.
|
|
192
198
|
4. Commit and merge to the default branch: the workflow runs from there.
|
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
require "json"
|
|
2
|
+
require "net/http"
|
|
3
|
+
require "uri"
|
|
4
|
+
|
|
5
|
+
module BrowserReviewGate
|
|
6
|
+
# One request to a model provider's HTTP API, with no tools and the answer bound to a JSON schema.
|
|
7
|
+
# For projects that assess with an API key instead of the Claude Code CLI.
|
|
8
|
+
class ModelApi
|
|
9
|
+
PROVIDERS = {
|
|
10
|
+
"anthropic" => { key: "ANTHROPIC_API_KEY", model: "claude-opus-5-5" },
|
|
11
|
+
"openai" => { key: "OPENAI_API_KEY" },
|
|
12
|
+
"gemini" => { key: "GEMINI_API_KEY" }
|
|
13
|
+
}.freeze
|
|
14
|
+
MAX_TOKENS = 16_000
|
|
15
|
+
|
|
16
|
+
# `transport` takes the URL, the headers and the request body, and returns [HTTP status, body].
|
|
17
|
+
def initialize(provider:, model: nil, transport: nil)
|
|
18
|
+
settings = PROVIDERS.fetch(provider) { raise Error, "Unknown model provider #{provider.inspect} (known: #{PROVIDERS.keys.join(", ")})" }
|
|
19
|
+
@provider = provider
|
|
20
|
+
@key_name = settings.fetch(:key)
|
|
21
|
+
@default_model = settings[:model]
|
|
22
|
+
@model = model.to_s.empty? ? @default_model : model
|
|
23
|
+
@transport = transport || method(:post)
|
|
24
|
+
raise Error, "Set `model` in #{Config::PATH} to assess with #{provider}" unless @model
|
|
25
|
+
end
|
|
26
|
+
|
|
27
|
+
# Returns the model's answer as JSON text. Raises Error when the call fails or gives no answer.
|
|
28
|
+
def complete(system:, user:, schema:)
|
|
29
|
+
key = ENV[@key_name].to_s
|
|
30
|
+
raise Error, "The model gave no answer: #{@key_name} is not set" if key.empty?
|
|
31
|
+
|
|
32
|
+
url, headers, body = send("#{@provider}_request", key, system, user, schema)
|
|
33
|
+
status, text = @transport.call(url, headers, body)
|
|
34
|
+
response = parse(text)
|
|
35
|
+
raise Error, "The model gave no answer: HTTP #{status} #{failure(response, text)}" unless status.to_i == 200
|
|
36
|
+
|
|
37
|
+
answer = parse(send("#{@provider}_answer", response))
|
|
38
|
+
return JSON.generate(answer) unless answer.empty?
|
|
39
|
+
|
|
40
|
+
raise Error, "The model gave no answer: #{failure(response, text)}"
|
|
41
|
+
end
|
|
42
|
+
|
|
43
|
+
private
|
|
44
|
+
|
|
45
|
+
# A request the provider's safeguards decline is retried on the provider's recommended fallback.
|
|
46
|
+
# That is asked for only with the built-in model: a model the project chose may not accept it.
|
|
47
|
+
def anthropic_request(key, system, user, schema)
|
|
48
|
+
headers = { "x-api-key" => key, "anthropic-version" => "2023-06-01" }
|
|
49
|
+
body = { model: @model, max_tokens: MAX_TOKENS, system: system, messages: [ { role: "user", content: user } ],
|
|
50
|
+
output_config: { format: { type: "json_schema", schema: schema } } }
|
|
51
|
+
if @model == @default_model
|
|
52
|
+
headers["anthropic-beta"] = "server-side-fallback-2026-07-01"
|
|
53
|
+
body[:fallbacks] = "default"
|
|
54
|
+
end
|
|
55
|
+
[ "https://api.anthropic.com/v1/messages", headers, body ]
|
|
56
|
+
end
|
|
57
|
+
|
|
58
|
+
def anthropic_answer(response)
|
|
59
|
+
return if response["stop_reason"] == "refusal"
|
|
60
|
+
|
|
61
|
+
Array(response["content"]).find { |block| block.is_a?(Hash) && block["type"] == "text" }&.fetch("text", nil)
|
|
62
|
+
end
|
|
63
|
+
|
|
64
|
+
def openai_request(key, system, user, schema)
|
|
65
|
+
body = { model: @model, messages: [ { role: "system", content: system }, { role: "user", content: user } ],
|
|
66
|
+
response_format: { type: "json_schema", json_schema: { name: "browser_assessment", schema: schema, strict: true } } }
|
|
67
|
+
[ "https://api.openai.com/v1/chat/completions", { "Authorization" => "Bearer #{key}" }, body ]
|
|
68
|
+
end
|
|
69
|
+
|
|
70
|
+
def openai_answer(response)
|
|
71
|
+
response.dig("choices", 0, "message", "content")
|
|
72
|
+
end
|
|
73
|
+
|
|
74
|
+
def gemini_request(key, system, user, schema)
|
|
75
|
+
body = { systemInstruction: { parts: [ { text: system } ] }, contents: [ { role: "user", parts: [ { text: user } ] } ],
|
|
76
|
+
generationConfig: { responseMimeType: "application/json", responseJsonSchema: schema } }
|
|
77
|
+
[ "https://generativelanguage.googleapis.com/v1beta/models/#{@model}:generateContent", { "x-goog-api-key" => key }, body ]
|
|
78
|
+
end
|
|
79
|
+
|
|
80
|
+
def gemini_answer(response)
|
|
81
|
+
Array(response.dig("candidates", 0, "content", "parts")).filter_map { |part| part["text"] if part.is_a?(Hash) }.join
|
|
82
|
+
end
|
|
83
|
+
|
|
84
|
+
def parse(text)
|
|
85
|
+
data = JSON.parse(text.to_s)
|
|
86
|
+
data.is_a?(Hash) ? data : {}
|
|
87
|
+
rescue JSON::ParserError
|
|
88
|
+
{}
|
|
89
|
+
end
|
|
90
|
+
|
|
91
|
+
# Providers explain an error in `error.message`; a declined request has no text to return.
|
|
92
|
+
def failure(response, text)
|
|
93
|
+
message = response.dig("error", "message") if response["error"].is_a?(Hash)
|
|
94
|
+
message ||= "the request was declined" if response["stop_reason"] == "refusal" || response.dig("choices", 0, "message", "refusal")
|
|
95
|
+
(message || text.to_s.strip)[0, 300]
|
|
96
|
+
end
|
|
97
|
+
|
|
98
|
+
def post(url, headers, body)
|
|
99
|
+
uri = URI(url)
|
|
100
|
+
request = Net::HTTP::Post.new(uri, headers.merge("Content-Type" => "application/json"))
|
|
101
|
+
request.body = JSON.generate(body)
|
|
102
|
+
response = Net::HTTP.start(uri.host, uri.port, use_ssl: true, open_timeout: 10, read_timeout: 300) { |http| http.request(request) }
|
|
103
|
+
[ response.code, response.body ]
|
|
104
|
+
rescue SystemCallError, SocketError, Timeout::Error, OpenSSL::SSL::SSLError => error
|
|
105
|
+
raise Error, "The model gave no answer: #{error.message}"
|
|
106
|
+
end
|
|
107
|
+
end
|
|
108
|
+
end
|
|
@@ -9,6 +9,22 @@ module BrowserReviewGate
|
|
|
9
9
|
NPX = %w[npx --yes @anthropic-ai/claude-code@2.1.288].freeze
|
|
10
10
|
CREDENTIALS = %w[CLAUDE_CODE_OAUTH_TOKEN ANTHROPIC_API_KEY].freeze
|
|
11
11
|
|
|
12
|
+
# The client the project's settings and the credentials at hand call for. The Claude Code CLI stays
|
|
13
|
+
# the default; an OpenAI or Gemini key alone selects that provider's API.
|
|
14
|
+
def self.for(config, environment: ENV)
|
|
15
|
+
provider = config.provider || detect(config, environment)
|
|
16
|
+
return new(command: config.claude_command, model: config.model) if provider == "claude-cli"
|
|
17
|
+
|
|
18
|
+
ModelApi.new(provider: provider, model: config.model)
|
|
19
|
+
end
|
|
20
|
+
|
|
21
|
+
def self.detect(config, environment)
|
|
22
|
+
set = ->(name) { !environment[name].to_s.empty? }
|
|
23
|
+
return "claude-cli" if config.claude_command || CREDENTIALS.any?(&set)
|
|
24
|
+
|
|
25
|
+
ModelApi::PROVIDERS.keys.find { |provider| set.call(ModelApi::PROVIDERS.fetch(provider).fetch(:key)) } || "claude-cli"
|
|
26
|
+
end
|
|
27
|
+
|
|
12
28
|
# `runner` takes the environment, argv and stdin, and returns [stdout, stderr, success?].
|
|
13
29
|
def initialize(command: nil, model: nil, runner: nil)
|
|
14
30
|
@command = command.to_s.empty? ? default_command : command.split
|
|
@@ -49,21 +49,29 @@ module BrowserReviewGate
|
|
|
49
49
|
|
|
50
50
|
PROJECT_RULES_NOTE = <<~TEXT.strip.freeze
|
|
51
51
|
The maintainers of this repository wrote these rules for their application. They refine the
|
|
52
|
-
decision above and win where they disagree with it.
|
|
53
|
-
|
|
52
|
+
decision above and win where they disagree with it. Include the scenarios they name for the areas
|
|
53
|
+
the change touches.
|
|
54
54
|
TEXT
|
|
55
55
|
|
|
56
56
|
FRAME_TAIL = <<~TEXT.strip.freeze
|
|
57
|
-
##
|
|
57
|
+
## Scenarios
|
|
58
58
|
|
|
59
|
-
|
|
60
|
-
every
|
|
61
|
-
|
|
59
|
+
When the decision is `required`, `scenarios` lists what a person must do in the browser to exercise
|
|
60
|
+
every behavior this pull request changes, including what the project rules require for the areas
|
|
61
|
+
it touches. One entry per scenario, at most 12, the fewest that cover the change. `name` says what
|
|
62
|
+
to do and what must be seen, in one sentence. `id` is a short label such as `SIGN-IN-1`: letters,
|
|
63
|
+
digits and hyphens.
|
|
64
|
+
|
|
65
|
+
When `report_cases` is present, those are the cases of an earlier browser run. If one of them
|
|
66
|
+
already exercises a scenario, give that scenario the `id` of that case, exactly. Give every other
|
|
67
|
+
scenario an `id` no case uses. Never reuse the `id` of a case that does not exercise the scenario.
|
|
68
|
+
|
|
69
|
+
When the decision is `not-required`, `scenarios` is an empty list.
|
|
62
70
|
|
|
63
71
|
## Reason
|
|
64
72
|
|
|
65
73
|
`reason` is one plain sentence of at most 30 words. No links, mentions, code, or HTML. No company
|
|
66
|
-
or customer names.
|
|
74
|
+
or customer names. The same holds for scenario names.
|
|
67
75
|
TEXT
|
|
68
76
|
|
|
69
77
|
private
|
|
@@ -30,7 +30,7 @@ module BrowserReviewGate
|
|
|
30
30
|
actor = @github.login
|
|
31
31
|
data["verified_by"] = "AI browser agent (run by #{actor})"
|
|
32
32
|
|
|
33
|
-
comments = @github.
|
|
33
|
+
comments = @github.trusted_comments(number)
|
|
34
34
|
ensure_browser_run_wanted!(number, comments, local_sha)
|
|
35
35
|
report = merged_report(data, comments)
|
|
36
36
|
|
|
@@ -54,7 +54,7 @@ module BrowserReviewGate
|
|
|
54
54
|
data = read_report_file
|
|
55
55
|
ensure_report_sha!(data, local_sha)
|
|
56
56
|
report = Report.new(data.merge("verified_by" => "pending"))
|
|
57
|
-
|
|
57
|
+
ensure_acceptable!(report)
|
|
58
58
|
|
|
59
59
|
saved_report = SavedReport.new(sha: local_sha, shell: @shell)
|
|
60
60
|
saved_report.write(data)
|
|
@@ -107,7 +107,7 @@ module BrowserReviewGate
|
|
|
107
107
|
|
|
108
108
|
def merged_report(data, comments)
|
|
109
109
|
submitted = Report.new(data)
|
|
110
|
-
|
|
110
|
+
ensure_acceptable!(submitted)
|
|
111
111
|
|
|
112
112
|
prior = Report.latest_comment(comments)
|
|
113
113
|
report = Report.merge(prior ? Report.data_from(prior["body"]) : {}, submitted.to_h)
|
|
@@ -116,6 +116,12 @@ module BrowserReviewGate
|
|
|
116
116
|
report
|
|
117
117
|
end
|
|
118
118
|
|
|
119
|
+
# A run is taken only with the evidence of each case: the page and what was observed there.
|
|
120
|
+
def ensure_acceptable!(report)
|
|
121
|
+
problems = report.valid? ? report.evidence_errors : report.errors
|
|
122
|
+
raise Error, problems.join("; ") if problems.any?
|
|
123
|
+
end
|
|
124
|
+
|
|
119
125
|
def body(report, complete)
|
|
120
126
|
return report.markdown if complete
|
|
121
127
|
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
require "digest"
|
|
1
2
|
require "json"
|
|
2
3
|
|
|
3
4
|
module BrowserReviewGate
|
|
@@ -7,6 +8,7 @@ module BrowserReviewGate
|
|
|
7
8
|
SHA_PATTERN = /\A\h{7,40}\z/
|
|
8
9
|
CASE_ID_PATTERN = /\A[A-Za-z0-9][A-Za-z0-9_.-]{0,63}\z/
|
|
9
10
|
RESULTS = %w[pass fail].freeze
|
|
11
|
+
URL_PATTERN = %r{\A(?:https?://|/)\S*\z}
|
|
10
12
|
|
|
11
13
|
attr_reader :errors
|
|
12
14
|
|
|
@@ -21,24 +23,32 @@ module BrowserReviewGate
|
|
|
21
23
|
raise Error, "Existing browser verification report has a malformed data marker"
|
|
22
24
|
end
|
|
23
25
|
|
|
24
|
-
# The newest
|
|
25
|
-
def self.
|
|
26
|
+
# The newest report on the PR when it is well formed, or nil.
|
|
27
|
+
def self.latest(comments)
|
|
26
28
|
comment = latest_comment(comments)
|
|
27
29
|
report = comment && new(data_from(comment["body"]))
|
|
28
|
-
report if report&.
|
|
30
|
+
report if report&.valid?
|
|
29
31
|
rescue Error
|
|
30
32
|
nil
|
|
31
33
|
end
|
|
32
34
|
|
|
33
|
-
#
|
|
35
|
+
# The newest report on the PR when it passes, or nil.
|
|
36
|
+
def self.latest_passing(comments)
|
|
37
|
+
report = latest(comments)
|
|
38
|
+
report if report&.passed?
|
|
39
|
+
end
|
|
40
|
+
|
|
41
|
+
# Later results replace earlier ones case by case; cases that were not run again are kept. The
|
|
42
|
+
# merged report counts the runs published on the PR and how many of them had a failing case.
|
|
34
43
|
def self.merge(previous_data, latest_data)
|
|
35
44
|
previous = new(previous_data)
|
|
36
45
|
latest = new(latest_data)
|
|
37
|
-
return latest unless
|
|
46
|
+
return latest unless latest.valid?
|
|
38
47
|
|
|
39
|
-
combined = previous.cases.to_h { |test_case| [ test_case["id"], test_case ] }
|
|
48
|
+
combined = previous.valid? ? previous.cases.to_h { |test_case| [ test_case["id"], test_case ] } : {}
|
|
40
49
|
latest.cases.each { |test_case| combined[test_case["id"]] = test_case }
|
|
41
|
-
|
|
50
|
+
failed = latest.cases.all? { |test_case| test_case["result"] == "pass" } ? 0 : 1
|
|
51
|
+
new(latest.to_h.merge("cases" => combined.values, "runs" => previous.runs + 1, "failed_runs" => previous.failed_runs + failed))
|
|
42
52
|
end
|
|
43
53
|
|
|
44
54
|
def initialize(data)
|
|
@@ -53,7 +63,28 @@ module BrowserReviewGate
|
|
|
53
63
|
def verified_by = @data["verified_by"]
|
|
54
64
|
def cases = @data["cases"].is_a?(Array) ? @data["cases"] : []
|
|
55
65
|
def excluded = @data["excluded"].is_a?(Array) ? @data["excluded"] : []
|
|
66
|
+
def runs = @data["runs"].is_a?(Integer) ? @data["runs"] : 0
|
|
67
|
+
def failed_runs = @data["failed_runs"].is_a?(Integer) ? @data["failed_runs"] : 0
|
|
56
68
|
def to_h = JSON.parse(JSON.generate(@data))
|
|
69
|
+
def passed_ids = cases.select { |test_case| test_case["result"] == "pass" }.map { |test_case| test_case["id"] }
|
|
70
|
+
|
|
71
|
+
# Changes whenever a case is added, re-run or changes its outcome.
|
|
72
|
+
def fingerprint
|
|
73
|
+
Digest::SHA256.hexdigest(JSON.generate(cases.map { |test_case| test_case.values_at("id", "result", "tested_sha") }))[0, 16]
|
|
74
|
+
end
|
|
75
|
+
|
|
76
|
+
# What a newly submitted run must say about each case: where it was exercised and what was seen.
|
|
77
|
+
# Cases carried over from reports published before this was required are not checked.
|
|
78
|
+
def evidence_errors
|
|
79
|
+
cases.each_with_index.flat_map do |test_case, index|
|
|
80
|
+
next [] unless test_case.is_a?(Hash)
|
|
81
|
+
|
|
82
|
+
found = []
|
|
83
|
+
found << "case #{index + 1} url must be the page that was exercised: an http(s) URL or a path starting with /" unless url?(test_case["url"])
|
|
84
|
+
found << "case #{index + 1} details must say what was observed" unless text?(test_case["details"], 500)
|
|
85
|
+
found
|
|
86
|
+
end
|
|
87
|
+
end
|
|
57
88
|
|
|
58
89
|
def passed?
|
|
59
90
|
valid? && @data["coverage_complete"] && cases.all? { |test_case| test_case["result"] == "pass" }
|
|
@@ -122,6 +153,7 @@ module BrowserReviewGate
|
|
|
122
153
|
errors << "case #{position} result must be pass or fail" unless RESULTS.include?(test_case["result"])
|
|
123
154
|
errors << "case #{position} tested_sha must be a 7–40 character hexadecimal commit SHA" unless sha?(test_case["tested_sha"])
|
|
124
155
|
errors << "case #{position} verified_by must be a non-empty string of at most 100 characters" unless text?(test_case["verified_by"], 100)
|
|
156
|
+
errors << "case #{position} url must be a string of at most 300 characters" unless test_case["url"].nil? || url?(test_case["url"])
|
|
125
157
|
details = test_case["details"]
|
|
126
158
|
errors << "case #{position} details must be a string of at most 500 characters" unless details.nil? || (details.is_a?(String) && details.length <= 500)
|
|
127
159
|
end
|
|
@@ -140,10 +172,12 @@ module BrowserReviewGate
|
|
|
140
172
|
|
|
141
173
|
def sha?(value) = value.is_a?(String) && SHA_PATTERN.match?(value)
|
|
142
174
|
def text?(value, limit) = value.is_a?(String) && !value.strip.empty? && value.length <= limit
|
|
175
|
+
def url?(value) = value.is_a?(String) && value.length <= 300 && URL_PATTERN.match?(value)
|
|
143
176
|
|
|
144
177
|
def markdown_case(test_case)
|
|
145
178
|
detail = test_case["details"].to_s.empty? ? "" : ": #{Markdown.escape(test_case["details"])}"
|
|
146
|
-
|
|
179
|
+
place = test_case["url"].to_s.empty? ? "" : " at #{Markdown.escape(test_case["url"])}"
|
|
180
|
+
"- #{test_case["result"].upcase} — #{Markdown.escape(test_case["name"])}#{place} (`#{test_case["id"]}`, tested " \
|
|
147
181
|
"`#{test_case["tested_sha"]}` by #{Markdown.escape(test_case["verified_by"])})#{detail}"
|
|
148
182
|
end
|
|
149
183
|
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
module BrowserReviewGate
|
|
2
|
+
# What the gate did on recent pull requests, read back from its own comments and labels.
|
|
3
|
+
class Stats
|
|
4
|
+
def initialize(github:, config: Config.new)
|
|
5
|
+
@github = github
|
|
6
|
+
@config = config
|
|
7
|
+
end
|
|
8
|
+
|
|
9
|
+
def to_h(since:)
|
|
10
|
+
counts = Hash.new(0)
|
|
11
|
+
@github.pull_requests_updated_since(since).each { |pull_request| count(pull_request, counts) }
|
|
12
|
+
%w[pull_requests assessed required verified waived owed found_a_failure].to_h { |key| [ key, counts[key] ] }
|
|
13
|
+
end
|
|
14
|
+
|
|
15
|
+
private
|
|
16
|
+
|
|
17
|
+
def count(pull_request, counts)
|
|
18
|
+
counts["pull_requests"] += 1
|
|
19
|
+
comments = @github.trusted_comments(pull_request.fetch("number"))
|
|
20
|
+
assessment = Assessment.latest(comments)
|
|
21
|
+
return unless assessment
|
|
22
|
+
|
|
23
|
+
counts["assessed"] += 1
|
|
24
|
+
return unless assessment.required?
|
|
25
|
+
|
|
26
|
+
labels = Array(pull_request["labels"]).map { |label| label["name"] }
|
|
27
|
+
counts["required"] += 1
|
|
28
|
+
counts["verified"] += 1 if labels.include?(@config.label)
|
|
29
|
+
counts["waived"] += 1 if labels.include?(@config.waiver_label)
|
|
30
|
+
counts["owed"] += 1 if pull_request["state"] == "open" && !labels.intersect?([ @config.label, @config.waiver_label ])
|
|
31
|
+
counts["found_a_failure"] += 1 if Report.latest(comments)&.failed_runs.to_i.positive?
|
|
32
|
+
end
|
|
33
|
+
end
|
|
34
|
+
end
|
|
@@ -8,7 +8,8 @@ module BrowserReviewGate
|
|
|
8
8
|
RUN_STEPS = %w[run_and_publish run_and_save].freeze
|
|
9
9
|
|
|
10
10
|
# `probe` takes the app URL and says whether something answers there.
|
|
11
|
-
def initialize(github:, config: Config.new, shell: Shell.new, probe: nil)
|
|
11
|
+
def initialize(github:, config: Config.new, shell: Shell.new, probe: nil, prompts: Prompts.new)
|
|
12
|
+
@prompts = prompts
|
|
12
13
|
@github = github
|
|
13
14
|
@config = config
|
|
14
15
|
@shell = shell
|
|
@@ -17,7 +18,7 @@ module BrowserReviewGate
|
|
|
17
18
|
|
|
18
19
|
def to_h(number = nil)
|
|
19
20
|
result = facts(number)
|
|
20
|
-
result["project_rules"] = Prompts::RULES if
|
|
21
|
+
result["project_rules"] = Prompts::RULES if @prompts.rules
|
|
21
22
|
result["app"] = app
|
|
22
23
|
# The agent never starts the app itself: a person decides what runs on their machine.
|
|
23
24
|
result["next"] = "ask_to_start_app" if RUN_STEPS.include?(result["next"]) && result["app"]["running"] == false
|
|
@@ -33,16 +34,19 @@ module BrowserReviewGate
|
|
|
33
34
|
return { "pull_request" => nil, "local_head" => local_sha, "saved_run" => saved, "next" => saved ? "open_pull_request" : "run_and_save" } unless number
|
|
34
35
|
|
|
35
36
|
pull_request = @github.pull_request(number)
|
|
36
|
-
comments = @github.
|
|
37
|
+
comments = @github.trusted_comments(number)
|
|
37
38
|
head_sha = pull_request.fetch("head").fetch("sha")
|
|
38
39
|
assessment = Assessment.latest(comments)
|
|
39
40
|
assessment = nil unless assessment&.for?(head_sha)
|
|
40
41
|
labels = pull_request.fetch("labels").map { |label| label["name"] }
|
|
41
|
-
|
|
42
|
+
report = Report.latest(comments)
|
|
43
|
+
verified = labels.include?(@config.label) && (assessment ? assessment.covered_by?(report) : report&.passed? == true)
|
|
42
44
|
waived = labels.include?(@config.waiver_label)
|
|
43
45
|
|
|
44
46
|
{ "pull_request" => number, "pr_head" => head_sha, "local_head" => local_sha,
|
|
45
|
-
"decision" => assessment&.decision, "reason" => assessment&.reason,
|
|
47
|
+
"decision" => assessment&.decision, "reason" => assessment&.reason,
|
|
48
|
+
# What the run must exercise; each id is the `id` of the case that reports it.
|
|
49
|
+
"scenarios" => assessment ? assessment.outstanding(report) : [], "verified" => verified, "waived" => waived,
|
|
46
50
|
"saved_run" => saved, "next" => next_step(assessment, verified || waived, saved, head_sha == local_sha) }
|
|
47
51
|
end
|
|
48
52
|
|
|
@@ -6,6 +6,12 @@ url:<%= url ? " #{url.inspect}" : "" %>
|
|
|
6
6
|
# How to sign in locally, e.g. "use the seed account from db/seeds.rb". No real credentials here.
|
|
7
7
|
sign_in:
|
|
8
8
|
|
|
9
|
+
# Who assesses a PR in CI. Unset: the Claude Code CLI with a CLAUDE_CODE_OAUTH_TOKEN or
|
|
10
|
+
# ANTHROPIC_API_KEY secret; with only an OPENAI_API_KEY or GEMINI_API_KEY secret, that provider's API.
|
|
11
|
+
# Set `provider` to claude-cli, anthropic, openai or gemini to choose; openai and gemini need `model`.
|
|
12
|
+
provider:
|
|
13
|
+
model:
|
|
14
|
+
|
|
9
15
|
# Workflow file name under .github/workflows, used to request an assessment by hand.
|
|
10
16
|
workflow: <%= workflow %>
|
|
11
17
|
|
|
@@ -15,10 +15,13 @@ This file is the playbook for the AI coding agent that does the run. The person
|
|
|
15
15
|
- `open_pull_request`: a run is saved and waits for the PR. Tell the person to open it.
|
|
16
16
|
- `ask_to_start_app`: the app is not running. Do not start it yourself: ask the person to start it (`app.start_command` says how) and stop until they have.
|
|
17
17
|
- `run_and_publish` or `run_and_save`: continue with step 2.
|
|
18
|
-
2.
|
|
18
|
+
2. Decide what to run.
|
|
19
|
+
- `scenarios` in the status output is the list CI wrote for this PR. Run every one of them, and give each case the `id` of its scenario. You may add cases of your own; you may not drop or rename one from the list.
|
|
20
|
+
- With no PR yet, or an empty list, choose the scenarios yourself: read the diff against the base branch, then `<%= rules_path %>`, which says what this project always checks and which scenarios each area needs. Its rules are mandatory. CI compares your cases with its own list once the PR is open and asks for what is missing.
|
|
21
|
+
- Do not re-run behavior an earlier passing case still covers.
|
|
19
22
|
3. Open the running app in a browser tool. `app` in the status output gives the URL and a sign-in hint. Never start, restart or stop the app yourself: what runs on the person's machine is their decision. Use local test data and do not destroy existing data. If the app, the data, or a browser tool is missing, report the blocker. Reading code and running unit tests does not replace a browser run.
|
|
20
|
-
4. Run every scenario end to end. After key actions check the page state, console errors, and failed requests.
|
|
21
|
-
5. Write the report as JSON outside the repository (see the shape below). Set `coverage_complete` to true only when every scenario from step 2 ran.
|
|
23
|
+
4. Run every scenario end to end. After key actions check the page state, console errors, and failed requests. For each case record the page it was exercised on (`url`) and what you observed there, including console errors and failed requests (`details`). A case without both is refused. A skipped or blocked scenario is not a pass.
|
|
24
|
+
5. Write the report as JSON outside the repository (see the shape below). Set `coverage_complete` to true only when every scenario from step 2 ran. CI checks the report against its list by case `id`: a missing scenario takes the label away again.
|
|
22
25
|
6. Publish or save, from the checkout that was tested, with everything committed and pushed:
|
|
23
26
|
- PR exists: `browser-review-gate publish --report PATH` (add `--pr NUMBER` when the branch has several). It posts the cases with their outcomes. Only a complete all-pass report adds the `<%= label %>` label.
|
|
24
27
|
- No PR yet: `browser-review-gate save --report PATH`. The run is published when the PR for this commit is opened. A new commit makes it stale.
|
|
@@ -36,7 +39,7 @@ Some scenarios cannot be exercised locally, for example a sign-in through an ext
|
|
|
36
39
|
"tested_sha": "<full commit SHA that was tested>",
|
|
37
40
|
"coverage_complete": true,
|
|
38
41
|
"cases": [
|
|
39
|
-
{ "id": "BROWSER-1", "name": "The changed interaction completes", "result": "pass", "details": "What was observed." }
|
|
42
|
+
{ "id": "BROWSER-1", "name": "The changed interaction completes", "result": "pass", "url": "/the/page/exercised", "details": "What was observed, with console errors and failed requests if any." }
|
|
40
43
|
],
|
|
41
44
|
"excluded": [
|
|
42
45
|
{ "name": "Sign-in through an external provider", "reason": "Needs a real provider account; covered by integration tests." }
|
|
@@ -44,7 +47,7 @@ Some scenarios cannot be exercised locally, for example a sign-in through an ext
|
|
|
44
47
|
}
|
|
45
48
|
```
|
|
46
49
|
|
|
47
|
-
`result` is `pass` or `fail`. `excluded` is optional: list a scenario there only when the person decided it stays outside the local run. Never exclude a scenario on your own to make the run complete. Keep customer and real company names out of the report.
|
|
50
|
+
`result` is `pass` or `fail`. `url` is an http(s) URL or a path starting with `/`. `excluded` is optional: list a scenario there only when the person decided it stays outside the local run. Never exclude a scenario on your own to make the run complete. Keep customer and real company names out of the report.
|
|
48
51
|
|
|
49
52
|
## Rules
|
|
50
53
|
|