wrangle 0.1.0 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +24 -0
- data/README.md +94 -3
- data/exe/wrangle +556 -33
- data/lib/wrangle/decision_provider.rb +128 -0
- data/lib/wrangle/desktop_autonomy.rb +61 -0
- data/lib/wrangle/desktop_decider.rb +263 -0
- data/lib/wrangle/desktop_dispatch.rb +67 -0
- data/lib/wrangle/desktop_effect.rb +198 -0
- data/lib/wrangle/desktop_observation.rb +252 -0
- data/lib/wrangle/desktop_policy.rb +47 -0
- data/lib/wrangle/desktop_progressive_observation.rb +34 -0
- data/lib/wrangle/desktop_proposal.rb +75 -0
- data/lib/wrangle/desktop_session_autonomy.rb +344 -0
- data/lib/wrangle/desktop_session_server.rb +374 -0
- data/lib/wrangle/desktop_task.rb +67 -0
- data/lib/wrangle/errors.rb +36 -0
- data/lib/wrangle/event_log.rb +51 -0
- data/lib/wrangle/jev.rb +4 -2
- data/lib/wrangle/macos/helper.swift +593 -0
- data/lib/wrangle/macos_driver.rb +200 -0
- data/lib/wrangle/macos_helper.rb +194 -0
- data/lib/wrangle/observation.rb +1 -1
- data/lib/wrangle/provider_conformance.jsonl +8 -0
- data/lib/wrangle/provider_factory.rb +106 -0
- data/lib/wrangle/provider_qualification.rb +73 -0
- data/lib/wrangle/run_loop.rb +4 -2
- data/lib/wrangle/safari.rb +7 -4
- data/lib/wrangle/scope_registry.rb +152 -0
- data/lib/wrangle/session_server.rb +15 -9
- data/lib/wrangle/tart_guest_driver.rb +251 -0
- data/lib/wrangle/timing.rb +12 -0
- data/lib/wrangle/version.rb +1 -1
- data/lib/wrangle.rb +20 -1
- data/skills/wrangle/SKILL.md +37 -4
- metadata +29 -7
|
@@ -0,0 +1,344 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "json"
|
|
4
|
+
require "securerandom"
|
|
5
|
+
|
|
6
|
+
require_relative "desktop_autonomy"
|
|
7
|
+
require_relative "desktop_observation"
|
|
8
|
+
|
|
9
|
+
module Wrangle
|
|
10
|
+
# Goal-driven preview and bounded continuation mixed into the persistent desktop session.
|
|
11
|
+
# The proposal, receipt, and lifecycle paths stay together so authorization cannot drift.
|
|
12
|
+
# rubocop:disable-next Metrics/ModuleLength
|
|
13
|
+
module DesktopSessionAutonomy
|
|
14
|
+
DEFAULT_BUDGET = 8
|
|
15
|
+
MAX_BUDGET = 40
|
|
16
|
+
TASK_MAX_BUDGET = 8
|
|
17
|
+
TASK_MAX_DRILLS = 3
|
|
18
|
+
TASK_EVIDENCE_ITEMS = 12
|
|
19
|
+
TASK_EVIDENCE_BYTES = 4 * 1024
|
|
20
|
+
|
|
21
|
+
# One natural request owns the complete bounded loop. Low-level capabilities stay internal:
|
|
22
|
+
# every mutation still passes through preview, provider qualification, one-shot execution,
|
|
23
|
+
# fresh-target validation, and independent effect verification.
|
|
24
|
+
def autonomous_task(request)
|
|
25
|
+
raise ConfigurationError, "A desktop task requires an explicit decision provider" unless @autonomy
|
|
26
|
+
|
|
27
|
+
budget = Integer(request.fetch("steps", DEFAULT_BUDGET))
|
|
28
|
+
unless budget.between?(1, TASK_MAX_BUDGET)
|
|
29
|
+
raise ArgumentError, "Desktop task budget must be between 1 and #{TASK_MAX_BUDGET}"
|
|
30
|
+
end
|
|
31
|
+
|
|
32
|
+
@task_drills = 0
|
|
33
|
+
@task_goal = request.fetch("goal")
|
|
34
|
+
@task_text_values = Array(request["literals"]&.values) + quoted_task_spans(@task_goal)
|
|
35
|
+
observe unless @observation
|
|
36
|
+
state = { remaining: budget, history: [], actions: [] }
|
|
37
|
+
@task_history = state.fetch(:history)
|
|
38
|
+
loop do
|
|
39
|
+
result = autonomous_task_step(request, state)
|
|
40
|
+
return result if result
|
|
41
|
+
end
|
|
42
|
+
end
|
|
43
|
+
|
|
44
|
+
def autonomous_preview(request)
|
|
45
|
+
raise ConfigurationError, "Attach with an explicit decision provider first" unless @autonomy
|
|
46
|
+
|
|
47
|
+
budget = Integer(request.fetch("steps", DEFAULT_BUDGET))
|
|
48
|
+
unless budget.between?(1, MAX_BUDGET)
|
|
49
|
+
raise ArgumentError, "Desktop run budget must be between 1 and #{MAX_BUDGET}"
|
|
50
|
+
end
|
|
51
|
+
|
|
52
|
+
observe unless @observation
|
|
53
|
+
assessment = assess(request)
|
|
54
|
+
choice = assessment.choice
|
|
55
|
+
log_decision(choice)
|
|
56
|
+
return @autonomy.compact(assessment) if choice.terminal? || assessment.paused
|
|
57
|
+
|
|
58
|
+
proposal = preview("ref" => choice.number, "operation" => choice.operation, "text" => choice.text)
|
|
59
|
+
run = start_run(request, budget)
|
|
60
|
+
qualify_proposal(proposal, choice, run:)
|
|
61
|
+
@autonomy.compact(assessment, proposal:).merge("run_id" => run.fetch("id"))
|
|
62
|
+
end
|
|
63
|
+
|
|
64
|
+
def continue_run(request)
|
|
65
|
+
run = @runs[request["run_id"]]
|
|
66
|
+
raise ArgumentError, "Unknown desktop run" unless run
|
|
67
|
+
raise PolicyDenied, "Execute the current proposal before continuing" unless run["authorized"]
|
|
68
|
+
raise PolicyDenied, "Execute the pending consequential proposal before continuing" if run["proposal_id"]
|
|
69
|
+
|
|
70
|
+
receipts = []
|
|
71
|
+
while run["remaining"].positive?
|
|
72
|
+
result = continuation_step(run, receipts)
|
|
73
|
+
return result if result
|
|
74
|
+
end
|
|
75
|
+
run_result(run, "budget_exhausted", receipts:)
|
|
76
|
+
end
|
|
77
|
+
|
|
78
|
+
def record_run_receipt(proposal, action_receipt)
|
|
79
|
+
run = @runs[proposal["run_id"]]
|
|
80
|
+
return action_receipt unless run
|
|
81
|
+
|
|
82
|
+
can_continue = false
|
|
83
|
+
if action_receipt["dispatch"] == "delivered"
|
|
84
|
+
run["authorized"] = true
|
|
85
|
+
run["remaining"] -= 1
|
|
86
|
+
run["history"] << { "operation" => proposal["operation"], "effect" => action_receipt["effect"] }
|
|
87
|
+
run["proposal_id"] = nil
|
|
88
|
+
can_continue = run["remaining"].positive? && !action_receipt["terminal"]
|
|
89
|
+
@runs.delete(run["id"]) unless can_continue
|
|
90
|
+
elsif action_receipt["reason"] != "approval_required"
|
|
91
|
+
run["proposal_id"] = nil
|
|
92
|
+
@runs.delete(run["id"])
|
|
93
|
+
end
|
|
94
|
+
action_receipt.merge("run_id" => run["id"], "can_continue" => can_continue)
|
|
95
|
+
end
|
|
96
|
+
|
|
97
|
+
private
|
|
98
|
+
|
|
99
|
+
def autonomous_task_step(request, state)
|
|
100
|
+
assessment = assess(request.merge("history" => state[:history]), drill_limit: TASK_MAX_DRILLS)
|
|
101
|
+
choice = assessment.choice
|
|
102
|
+
log_decision(choice)
|
|
103
|
+
return task_result("low_confidence", **state.slice(:remaining, :actions), assessment:) if assessment.paused
|
|
104
|
+
return task_result(choice.operation.downcase, **state.slice(:remaining, :actions), assessment:) if
|
|
105
|
+
choice.terminal?
|
|
106
|
+
|
|
107
|
+
# The first choice may have seen only a skeleton. Re-decide from the completed observation
|
|
108
|
+
# rather than carrying a target selection across a perception boundary.
|
|
109
|
+
return nil if escalate_partial_task_observation?
|
|
110
|
+
|
|
111
|
+
candidate = @observation.fetch("candidates").fetch(choice.number - 1)
|
|
112
|
+
proposal = preview("ref" => choice.number, "operation" => choice.operation, "text" => choice.text)
|
|
113
|
+
qualify_task_proposal(proposal, choice)
|
|
114
|
+
pending = task_action(candidate, choice)
|
|
115
|
+
halted = task_preflight_result(proposal, pending, assessment, state)
|
|
116
|
+
return halted if halted
|
|
117
|
+
|
|
118
|
+
task_receipt_result(proposal, pending, assessment, state)
|
|
119
|
+
end
|
|
120
|
+
|
|
121
|
+
def escalate_partial_task_observation?
|
|
122
|
+
return false unless @observation.dig("coverage", "truncated")
|
|
123
|
+
unless escalate_observation?
|
|
124
|
+
raise PartialObservation, "Selected action remains inside an unresolved progressive view"
|
|
125
|
+
end
|
|
126
|
+
if @observation.dig("coverage", "truncated")
|
|
127
|
+
raise PartialObservation, "Full desktop observation still has unresolved branches"
|
|
128
|
+
end
|
|
129
|
+
|
|
130
|
+
true
|
|
131
|
+
end
|
|
132
|
+
|
|
133
|
+
def task_preflight_result(proposal, pending, assessment, state)
|
|
134
|
+
status = if proposal.dig("policy", "consequential")
|
|
135
|
+
"approval_required"
|
|
136
|
+
elsif !@provider.mutation_qualified?
|
|
137
|
+
"provider_not_qualified"
|
|
138
|
+
end
|
|
139
|
+
task_result(status, **state.slice(:remaining, :actions), assessment:, pending:) if status
|
|
140
|
+
end
|
|
141
|
+
|
|
142
|
+
def task_receipt_result(proposal, pending, assessment, state)
|
|
143
|
+
action_receipt = execute("proposal_id" => proposal.fetch("proposal_id"))
|
|
144
|
+
state[:actions] << pending.merge(action_receipt.slice("dispatch", "effect", "reason", "terminal"))
|
|
145
|
+
dispatch = action_receipt.fetch("dispatch")
|
|
146
|
+
return task_result(dispatch, **state.slice(:remaining, :actions), assessment:) unless dispatch == "delivered"
|
|
147
|
+
|
|
148
|
+
state[:remaining] -= 1
|
|
149
|
+
state[:history] << task_history_entry(pending, action_receipt["effect"])
|
|
150
|
+
return task_result("terminal", **state.slice(:remaining, :actions), assessment:) if action_receipt["terminal"]
|
|
151
|
+
|
|
152
|
+
task_result("budget_exhausted", **state.slice(:remaining, :actions), assessment:) if
|
|
153
|
+
state[:remaining].zero?
|
|
154
|
+
end
|
|
155
|
+
|
|
156
|
+
def assess(request, drill_limit: nil)
|
|
157
|
+
@autonomy.assess(
|
|
158
|
+
goal: request["goal"], literals: request["literals"] || {}, observation: @observation,
|
|
159
|
+
history: Array(request["history"]), min_confidence: request.fetch("min_confidence", 0.5)
|
|
160
|
+
) do |ref|
|
|
161
|
+
if drill_limit && @task_drills >= drill_limit
|
|
162
|
+
raise PartialObservation, "Desktop task exceeded the progressive DRILL budget"
|
|
163
|
+
end
|
|
164
|
+
|
|
165
|
+
@task_drills += 1 if drill_limit
|
|
166
|
+
drill("ref" => ref) && @observation
|
|
167
|
+
end
|
|
168
|
+
end
|
|
169
|
+
|
|
170
|
+
def qualify_task_proposal(proposal, choice)
|
|
171
|
+
qualified = @provider.mutation_qualified?
|
|
172
|
+
stored = @proposals.fetch(proposal.fetch("proposal_id"))
|
|
173
|
+
stored["provider"] = { "name" => choice.provider, "model" => choice.model, "qualified" => qualified }
|
|
174
|
+
proposal["policy"] = proposal.fetch("policy").merge("provider_qualified" => qualified)
|
|
175
|
+
end
|
|
176
|
+
|
|
177
|
+
def task_history_entry(action, effect)
|
|
178
|
+
entry = action.slice("operation", "role", "label")
|
|
179
|
+
label = entry["label"]
|
|
180
|
+
entry["label"] = label[0, DesktopObservation::VISIBLE_TEXT_CHARS] if label.is_a?(String)
|
|
181
|
+
entry.merge("effect" => effect)
|
|
182
|
+
end
|
|
183
|
+
|
|
184
|
+
def task_result(status, remaining:, actions:, assessment:, pending: nil)
|
|
185
|
+
result = {
|
|
186
|
+
"schema" => "wrangle.task.v1", "status" => status, "app" => @scope.app,
|
|
187
|
+
"actions_taken" => @actions, "remaining" => remaining, "root_preserved" => true,
|
|
188
|
+
"message" => task_message(status), "decision" => task_decision(assessment),
|
|
189
|
+
"pending_action" => pending, "actions" => actions, "evidence" => task_evidence(status)
|
|
190
|
+
}.compact
|
|
191
|
+
@log&.record(
|
|
192
|
+
"task", "scope_id" => @scope.id, "status" => status, "actions_taken" => @actions,
|
|
193
|
+
"remaining" => remaining
|
|
194
|
+
)
|
|
195
|
+
result
|
|
196
|
+
end
|
|
197
|
+
|
|
198
|
+
def task_decision(assessment)
|
|
199
|
+
choice = assessment.choice
|
|
200
|
+
{ "operation" => choice.operation, "confidence" => choice.confidence,
|
|
201
|
+
"provider" => choice.provider, "model" => choice.model }
|
|
202
|
+
end
|
|
203
|
+
|
|
204
|
+
def task_action(candidate, choice)
|
|
205
|
+
{ "operation" => choice.operation, "role" => candidate["role"], "label" => candidate["label"],
|
|
206
|
+
"text" => choice.text && { "source" => choice.text_source, "characters" => choice.text.length } }.compact
|
|
207
|
+
end
|
|
208
|
+
|
|
209
|
+
def task_evidence(status)
|
|
210
|
+
observation = @observation
|
|
211
|
+
return unless observation
|
|
212
|
+
|
|
213
|
+
all = DesktopObservation.evidence_items(observation).map { |item| sanitize_evidence(item) }
|
|
214
|
+
selected, selection = select_task_evidence(status, observation, all)
|
|
215
|
+
selected.pop while selected.length > 1 && JSON.generate(selected).bytesize > TASK_EVIDENCE_BYTES
|
|
216
|
+
{
|
|
217
|
+
"complete" => observation["complete"], "items" => selected,
|
|
218
|
+
"total" => all.length, "omitted" => all.length - selected.length,
|
|
219
|
+
"explicit_truncation" => all.length > selected.length, "selection" => selection
|
|
220
|
+
}
|
|
221
|
+
rescue ProviderError
|
|
222
|
+
fallback_task_evidence(observation, all, "provider_failed")
|
|
223
|
+
end
|
|
224
|
+
|
|
225
|
+
def select_task_evidence(status, observation, all)
|
|
226
|
+
if status == "done" && all.length > TASK_EVIDENCE_ITEMS
|
|
227
|
+
selected = @autonomy.evidence(
|
|
228
|
+
goal: @task_goal, observation:, history: @task_history, limit: 3
|
|
229
|
+
).map { |candidate| sanitize_evidence(candidate) }
|
|
230
|
+
[selected, "provider"]
|
|
231
|
+
else
|
|
232
|
+
[sample_evidence(all), "bounded"]
|
|
233
|
+
end
|
|
234
|
+
end
|
|
235
|
+
|
|
236
|
+
def fallback_task_evidence(observation, all, selection)
|
|
237
|
+
selected = sample_evidence(all)
|
|
238
|
+
selected.pop while selected.length > 1 && JSON.generate(selected).bytesize > TASK_EVIDENCE_BYTES
|
|
239
|
+
{
|
|
240
|
+
"complete" => observation["complete"], "items" => selected,
|
|
241
|
+
"total" => all.length, "omitted" => all.length - selected.length,
|
|
242
|
+
"explicit_truncation" => all.length > selected.length, "selection" => selection
|
|
243
|
+
}
|
|
244
|
+
end
|
|
245
|
+
|
|
246
|
+
def sample_evidence(all)
|
|
247
|
+
return all if all.length <= TASK_EVIDENCE_ITEMS
|
|
248
|
+
|
|
249
|
+
half = TASK_EVIDENCE_ITEMS / 2
|
|
250
|
+
all.first(half) + all.last(half)
|
|
251
|
+
end
|
|
252
|
+
|
|
253
|
+
def sanitize_evidence(candidate)
|
|
254
|
+
candidate.slice("role", "label", "value", "states").transform_values do |value|
|
|
255
|
+
next value unless value.is_a?(String)
|
|
256
|
+
|
|
257
|
+
redacted = @task_text_values.reduce(value) do |text, literal|
|
|
258
|
+
literal.empty? ? text : text.gsub(literal, "[typed text omitted]")
|
|
259
|
+
end
|
|
260
|
+
redacted.length > 500 ? "#{redacted[0, 500]}…" : redacted
|
|
261
|
+
end
|
|
262
|
+
end
|
|
263
|
+
|
|
264
|
+
def quoted_task_spans(goal)
|
|
265
|
+
goal.scan(/"([^"]+)"|'([^']+)'/).map { |double, single| double || single }.uniq
|
|
266
|
+
end
|
|
267
|
+
|
|
268
|
+
def task_message(status)
|
|
269
|
+
{
|
|
270
|
+
"done" => "The requested result is visible in the application.",
|
|
271
|
+
"blocked" => "No observed safe action can progress the request.",
|
|
272
|
+
"handoff" => "The request needs a person or exact missing input.",
|
|
273
|
+
"low_confidence" => "Wrangle paused because the next action was uncertain.",
|
|
274
|
+
"approval_required" => "A consequential action requires separate approval; nothing was sent.",
|
|
275
|
+
"provider_not_qualified" => "The decision provider may inspect but is not qualified to change the app.",
|
|
276
|
+
"not_delivered" => "The app did not receive the proposed action.",
|
|
277
|
+
"refused" => "Wrangle refused the proposed action.",
|
|
278
|
+
"delivery_unknown" => "The action may or may not have happened and was not retried.",
|
|
279
|
+
"terminal" => "Wrangle stopped after a terminal verification failure.",
|
|
280
|
+
"budget_exhausted" => "Wrangle stopped at the task action limit."
|
|
281
|
+
}.fetch(status, "Wrangle stopped without claiming success.")
|
|
282
|
+
end
|
|
283
|
+
|
|
284
|
+
def continuation_step(run, receipts)
|
|
285
|
+
assessment = assess(run)
|
|
286
|
+
choice = assessment.choice
|
|
287
|
+
log_decision(choice)
|
|
288
|
+
if assessment.paused || choice.terminal?
|
|
289
|
+
status = assessment.paused ? "low_confidence" : choice.operation.downcase
|
|
290
|
+
return run_result(run, status, receipts:, assessment:)
|
|
291
|
+
end
|
|
292
|
+
|
|
293
|
+
proposal = preview("ref" => choice.number, "operation" => choice.operation, "text" => choice.text)
|
|
294
|
+
qualify_proposal(proposal, choice, run:)
|
|
295
|
+
if proposal.dig("policy", "consequential")
|
|
296
|
+
return run_result(run, "approval_required", receipts:, assessment:, proposal:)
|
|
297
|
+
end
|
|
298
|
+
unless @provider.mutation_qualified?
|
|
299
|
+
return run_result(run, "provider_not_qualified", receipts:, assessment:, proposal:)
|
|
300
|
+
end
|
|
301
|
+
|
|
302
|
+
action_receipt = execute("proposal_id" => proposal["proposal_id"])
|
|
303
|
+
receipts << action_receipt
|
|
304
|
+
return run_result(run, "paused", receipts:) unless action_receipt["dispatch"] == "delivered"
|
|
305
|
+
return run_result(run, "terminal", receipts:) if action_receipt["terminal"]
|
|
306
|
+
|
|
307
|
+
nil
|
|
308
|
+
end
|
|
309
|
+
|
|
310
|
+
def start_run(request, budget)
|
|
311
|
+
run = { "id" => SecureRandom.hex(12), "goal" => request["goal"], "literals" => request["literals"] || {},
|
|
312
|
+
"history" => [], "min_confidence" => request.fetch("min_confidence", 0.5),
|
|
313
|
+
"remaining" => budget, "authorized" => false, "proposal_id" => nil }
|
|
314
|
+
@runs[run["id"]] = run
|
|
315
|
+
run
|
|
316
|
+
end
|
|
317
|
+
|
|
318
|
+
def qualify_proposal(proposal, choice, run:)
|
|
319
|
+
qualified = @provider.mutation_qualified?
|
|
320
|
+
stored = @proposals.fetch(proposal["proposal_id"])
|
|
321
|
+
stored["provider"] = { "name" => choice.provider, "model" => choice.model, "qualified" => qualified }
|
|
322
|
+
stored["run_id"] = run["id"]
|
|
323
|
+
run["proposal_id"] = proposal["proposal_id"]
|
|
324
|
+
proposal["policy"] = proposal["policy"].merge("provider_qualified" => qualified)
|
|
325
|
+
end
|
|
326
|
+
|
|
327
|
+
def log_decision(choice)
|
|
328
|
+
@log&.record(
|
|
329
|
+
"decision", "scope_id" => @scope.id, "revision" => @observation["revision"],
|
|
330
|
+
"operation" => choice.operation, "confidence" => choice.confidence,
|
|
331
|
+
"latency_ms" => choice.latency_ms, "provider" => choice.provider, "model" => choice.model
|
|
332
|
+
)
|
|
333
|
+
end
|
|
334
|
+
|
|
335
|
+
def run_result(run, status, receipts:, assessment: nil, proposal: nil)
|
|
336
|
+
@runs.delete(run["id"]) unless status == "approval_required"
|
|
337
|
+
{
|
|
338
|
+
"schema" => "wrangle.run.v1", "run_id" => run["id"], "status" => status,
|
|
339
|
+
"remaining" => run["remaining"], "decision" => assessment && @autonomy.compact(assessment),
|
|
340
|
+
"proposal" => proposal, "receipts" => receipts
|
|
341
|
+
}.compact
|
|
342
|
+
end
|
|
343
|
+
end
|
|
344
|
+
end
|