gl-browser-use-binary 0.0.0b5__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- gl_browser_use/__init__.py +60 -0
- gl_browser_use/__init__.pyi +12 -0
- gl_browser_use/client.py +639 -0
- gl_browser_use/client.pyi +57 -0
- gl_browser_use/config.py +169 -0
- gl_browser_use/config.pyi +42 -0
- gl_browser_use/errors.py +119 -0
- gl_browser_use/errors.pyi +40 -0
- gl_browser_use/events.py +147 -0
- gl_browser_use/events.pyi +29 -0
- gl_browser_use/infrastructure/__init__.py +39 -0
- gl_browser_use/infrastructure/__init__.pyi +6 -0
- gl_browser_use/infrastructure/base.py +103 -0
- gl_browser_use/infrastructure/base.pyi +56 -0
- gl_browser_use/infrastructure/steel.py +296 -0
- gl_browser_use/infrastructure/steel.pyi +56 -0
- gl_browser_use/models.py +45 -0
- gl_browser_use/models.pyi +32 -0
- gl_browser_use/parsing.py +264 -0
- gl_browser_use/parsing.pyi +41 -0
- gl_browser_use/parsing_recovery.py +204 -0
- gl_browser_use/parsing_recovery.pyi +35 -0
- gl_browser_use/session_errors.py +72 -0
- gl_browser_use/session_errors.pyi +16 -0
- gl_browser_use/storage/__init__.py +39 -0
- gl_browser_use/storage/__init__.pyi +6 -0
- gl_browser_use/storage/base.py +125 -0
- gl_browser_use/storage/base.pyi +46 -0
- gl_browser_use/storage/minio_compatible.py +84 -0
- gl_browser_use/storage/minio_compatible.pyi +15 -0
- gl_browser_use_binary-0.0.0b5.dist-info/METADATA +236 -0
- gl_browser_use_binary-0.0.0b5.dist-info/RECORD +34 -0
- gl_browser_use_binary-0.0.0b5.dist-info/WHEEL +5 -0
- gl_browser_use_binary-0.0.0b5.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,264 @@
|
|
|
1
|
+
"""Parsing and tool-call helpers for GL Browser Use SDK.
|
|
2
|
+
|
|
3
|
+
Authors:
|
|
4
|
+
Reinhart Linanda (reinhart.linanda@gdplabs.id)
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import json
|
|
10
|
+
import logging
|
|
11
|
+
from dataclasses import dataclass
|
|
12
|
+
from typing import Any
|
|
13
|
+
|
|
14
|
+
from browser_use import Agent
|
|
15
|
+
|
|
16
|
+
from gl_browser_use.parsing_recovery import (
|
|
17
|
+
extract_balanced_json_value,
|
|
18
|
+
recover_concatenated_json_objects,
|
|
19
|
+
repair_json_blob,
|
|
20
|
+
)
|
|
21
|
+
|
|
22
|
+
logger = logging.getLogger(__name__)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def coerce_string(value: Any) -> str:
|
|
26
|
+
"""Return a stable ``str`` representation of ``value`` (``None`` becomes ``""``)."""
|
|
27
|
+
if value is None:
|
|
28
|
+
return ""
|
|
29
|
+
if isinstance(value, str):
|
|
30
|
+
return value
|
|
31
|
+
return str(value)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def summarize_terminal_error(error_message: str, *, max_length: int = 200) -> str:
|
|
35
|
+
"""Collapse whitespace in ``error_message`` and truncate to ``max_length`` characters."""
|
|
36
|
+
normalized = " ".join(coerce_string(error_message).split())
|
|
37
|
+
if len(normalized) <= max_length:
|
|
38
|
+
return normalized
|
|
39
|
+
return normalized[: max_length - 1] + "…"
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@dataclass
|
|
43
|
+
class ToolCall:
|
|
44
|
+
"""Normalized tool-call snapshot extracted from the underlying agent state."""
|
|
45
|
+
|
|
46
|
+
name: str
|
|
47
|
+
args: dict[str, Any]
|
|
48
|
+
output: str = ""
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def read_tool_call(call: Any) -> tuple[str, dict[str, Any], str]:
|
|
52
|
+
"""Return a ``(name, args, output)`` triple for a tool call of any supported shape.
|
|
53
|
+
|
|
54
|
+
``name`` may be an empty string when the underlying payload is missing
|
|
55
|
+
a name; callers requiring a fallback label should apply one explicitly.
|
|
56
|
+
"""
|
|
57
|
+
if isinstance(call, dict):
|
|
58
|
+
name = coerce_string(call.get("name"))
|
|
59
|
+
args_value = call.get("args", {})
|
|
60
|
+
output = coerce_string(call.get("output", ""))
|
|
61
|
+
else:
|
|
62
|
+
name = coerce_string(getattr(call, "name", ""))
|
|
63
|
+
args_value = getattr(call, "args", {})
|
|
64
|
+
output = coerce_string(getattr(call, "output", ""))
|
|
65
|
+
|
|
66
|
+
args = args_value if isinstance(args_value, dict) else {"value": args_value}
|
|
67
|
+
return name, args, output
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def extract_outputs(last_result: list[Any]) -> list[str]:
|
|
71
|
+
"""Return one display string per entry in ``last_result``."""
|
|
72
|
+
outputs: list[str] = []
|
|
73
|
+
for result in last_result or []:
|
|
74
|
+
extracted = coerce_string(getattr(result, "extracted_content", None))
|
|
75
|
+
if extracted:
|
|
76
|
+
outputs.append(extracted)
|
|
77
|
+
elif getattr(result, "error", None):
|
|
78
|
+
outputs.append(f"Error: {result.error}")
|
|
79
|
+
else:
|
|
80
|
+
outputs.append("")
|
|
81
|
+
return outputs
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def extract_tool_calls(agent: Agent) -> list[ToolCall]:
|
|
85
|
+
"""Extract serialized tool-call snapshots from the agent's last model output."""
|
|
86
|
+
state = getattr(agent, "state", None)
|
|
87
|
+
if not state:
|
|
88
|
+
return []
|
|
89
|
+
|
|
90
|
+
model_output = getattr(state, "last_model_output", None)
|
|
91
|
+
action = getattr(model_output, "action", None)
|
|
92
|
+
if not action:
|
|
93
|
+
return []
|
|
94
|
+
|
|
95
|
+
last_result = getattr(state, "last_result", None) or []
|
|
96
|
+
outputs = extract_outputs(last_result)
|
|
97
|
+
|
|
98
|
+
calls: list[ToolCall] = []
|
|
99
|
+
for index, action_model in enumerate(action):
|
|
100
|
+
try:
|
|
101
|
+
root_action = getattr(action_model, "root", action_model)
|
|
102
|
+
payload = None
|
|
103
|
+
if hasattr(root_action, "model_dump"):
|
|
104
|
+
payload = root_action.model_dump(exclude_unset=True)
|
|
105
|
+
elif hasattr(root_action, "dict"):
|
|
106
|
+
payload = root_action.dict(exclude_unset=True)
|
|
107
|
+
elif isinstance(root_action, dict):
|
|
108
|
+
payload = root_action
|
|
109
|
+
else:
|
|
110
|
+
payload = getattr(root_action, "__dict__", {})
|
|
111
|
+
|
|
112
|
+
if isinstance(payload, dict) and payload:
|
|
113
|
+
name = str(next(iter(payload)))
|
|
114
|
+
args = payload[name]
|
|
115
|
+
if not isinstance(args, dict):
|
|
116
|
+
args = {"value": args}
|
|
117
|
+
else:
|
|
118
|
+
name = "unknown"
|
|
119
|
+
args = {}
|
|
120
|
+
output = outputs[index] if index < len(outputs) else ""
|
|
121
|
+
calls.append(ToolCall(name=name, args=args, output=output))
|
|
122
|
+
except Exception as exc:
|
|
123
|
+
output = outputs[index] if index < len(outputs) else ""
|
|
124
|
+
calls.append(
|
|
125
|
+
ToolCall(name="parsing_error", args={"error": summarize_terminal_error(str(exc))}, output=output)
|
|
126
|
+
)
|
|
127
|
+
return calls
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def extract_final_output(agent: Agent) -> str:
|
|
131
|
+
"""Return the final textual output from the agent, or an empty string if absent."""
|
|
132
|
+
state = getattr(agent, "state", None)
|
|
133
|
+
if not state:
|
|
134
|
+
return ""
|
|
135
|
+
|
|
136
|
+
last_result = getattr(state, "last_result", None) or []
|
|
137
|
+
if not last_result:
|
|
138
|
+
return ""
|
|
139
|
+
|
|
140
|
+
done_content: str | None = None
|
|
141
|
+
any_content: str | None = None
|
|
142
|
+
any_error: str | None = None
|
|
143
|
+
for result in reversed(list(last_result)):
|
|
144
|
+
extracted = coerce_string(getattr(result, "extracted_content", None))
|
|
145
|
+
if extracted and done_content is None and getattr(result, "is_done", False):
|
|
146
|
+
done_content = extracted
|
|
147
|
+
if extracted and any_content is None:
|
|
148
|
+
any_content = extracted
|
|
149
|
+
if any_error is None:
|
|
150
|
+
error = getattr(result, "error", None)
|
|
151
|
+
if error:
|
|
152
|
+
any_error = coerce_string(error)
|
|
153
|
+
|
|
154
|
+
if done_content:
|
|
155
|
+
return done_content
|
|
156
|
+
if any_content:
|
|
157
|
+
return any_content
|
|
158
|
+
if any_error:
|
|
159
|
+
return f"Error: {any_error}"
|
|
160
|
+
return ""
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def _strip_trailers(remainder: str) -> str:
|
|
164
|
+
for trailer in ("</extracted_content>", "<file_system>", "</file_system>"):
|
|
165
|
+
idx = remainder.find(trailer)
|
|
166
|
+
if idx != -1:
|
|
167
|
+
remainder = remainder[:idx]
|
|
168
|
+
return remainder.strip()
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def _try_repair_and_parse(json_blob: str) -> tuple[dict[str, Any] | list[Any] | None, str | None]:
|
|
172
|
+
repaired = repair_json_blob(json_blob)
|
|
173
|
+
if repaired is None:
|
|
174
|
+
return None, "JSON parsing failed."
|
|
175
|
+
try:
|
|
176
|
+
return json.loads(repaired), None
|
|
177
|
+
except json.JSONDecodeError:
|
|
178
|
+
return None, "JSON parsing failed after repair."
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def parse_structured_data(output: str) -> tuple[dict[str, Any] | list[Any] | None, str | None]:
|
|
182
|
+
"""Parse the ``Extracted Content:`` block in ``output`` with fallback recovery.
|
|
183
|
+
|
|
184
|
+
The parser (1) trims trailers and slices from the first ``{``/``[``;
|
|
185
|
+
(2) attempts a direct ``json.loads``; (3) attempts to recover
|
|
186
|
+
concatenated top-level JSON objects; (4) extracts a balanced-prefix
|
|
187
|
+
JSON value with a depth-aware scanner that respects string literals
|
|
188
|
+
and parses it; and finally (5) falls back to ``json_repair.repair_json``.
|
|
189
|
+
"""
|
|
190
|
+
marker = "Extracted Content:"
|
|
191
|
+
if marker not in output:
|
|
192
|
+
return None, None
|
|
193
|
+
_, remainder = output.split(marker, 1)
|
|
194
|
+
remainder = _strip_trailers(remainder)
|
|
195
|
+
if not remainder:
|
|
196
|
+
return None, None
|
|
197
|
+
|
|
198
|
+
candidate = _slice_from_first_opener(remainder)
|
|
199
|
+
if candidate is None:
|
|
200
|
+
return None, None
|
|
201
|
+
|
|
202
|
+
try:
|
|
203
|
+
return json.loads(candidate), None
|
|
204
|
+
except json.JSONDecodeError as error:
|
|
205
|
+
logger.debug("Primary JSON parse failed for structured extraction output: %s", error)
|
|
206
|
+
|
|
207
|
+
recovered = recover_concatenated_json_objects(candidate)
|
|
208
|
+
if recovered is not None:
|
|
209
|
+
return recovered, None
|
|
210
|
+
|
|
211
|
+
balanced_prefix = extract_balanced_json_value(candidate)
|
|
212
|
+
if balanced_prefix and balanced_prefix != candidate:
|
|
213
|
+
try:
|
|
214
|
+
return json.loads(balanced_prefix), None
|
|
215
|
+
except json.JSONDecodeError as error:
|
|
216
|
+
logger.debug("Balanced-prefix JSON parse failed: %s", error)
|
|
217
|
+
|
|
218
|
+
return _try_repair_and_parse(candidate)
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
def _slice_from_first_opener(text: str) -> str | None:
|
|
222
|
+
openers = [text.find(opener) for opener in ("{", "[")]
|
|
223
|
+
positions = [position for position in openers if position != -1]
|
|
224
|
+
if not positions:
|
|
225
|
+
return None
|
|
226
|
+
return text[min(positions) :].strip() or None
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
def validate_structured_data_output(tool_calls: list[ToolCall]) -> str | None:
|
|
230
|
+
"""Return a user-facing error when structured data extraction failed or yielded no usable entries."""
|
|
231
|
+
for call in tool_calls:
|
|
232
|
+
if call.name != "extract_structured_data":
|
|
233
|
+
continue
|
|
234
|
+
payload, parse_error = parse_structured_data(call.output)
|
|
235
|
+
if parse_error:
|
|
236
|
+
return f"Structured data extraction emitted invalid JSON. {parse_error}"
|
|
237
|
+
if payload is None:
|
|
238
|
+
continue
|
|
239
|
+
if isinstance(payload, dict):
|
|
240
|
+
status = payload.get("status")
|
|
241
|
+
if isinstance(status, str) and status.lower() == "error":
|
|
242
|
+
message = payload.get("message") or payload.get("error") or call.output
|
|
243
|
+
return f"Structured data extraction failed: {summarize_terminal_error(coerce_string(message))}"
|
|
244
|
+
if (
|
|
245
|
+
(isinstance(payload.get("products"), list) and len(payload.get("products", [])) == 0)
|
|
246
|
+
or payload.get("count") == 0
|
|
247
|
+
or payload.get("products_found") == 0
|
|
248
|
+
or payload.get("available") is False
|
|
249
|
+
):
|
|
250
|
+
return "Structured data extraction returned no usable entries."
|
|
251
|
+
return None
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
__all__ = [
|
|
255
|
+
"ToolCall",
|
|
256
|
+
"coerce_string",
|
|
257
|
+
"extract_final_output",
|
|
258
|
+
"extract_outputs",
|
|
259
|
+
"extract_tool_calls",
|
|
260
|
+
"parse_structured_data",
|
|
261
|
+
"read_tool_call",
|
|
262
|
+
"summarize_terminal_error",
|
|
263
|
+
"validate_structured_data_output",
|
|
264
|
+
]
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
from browser_use import Agent
|
|
2
|
+
from dataclasses import dataclass
|
|
3
|
+
from typing import Any
|
|
4
|
+
|
|
5
|
+
__all__ = ['ToolCall', 'coerce_string', 'extract_final_output', 'extract_outputs', 'extract_tool_calls', 'parse_structured_data', 'read_tool_call', 'summarize_terminal_error', 'validate_structured_data_output']
|
|
6
|
+
|
|
7
|
+
def coerce_string(value: Any) -> str:
|
|
8
|
+
'''Return a stable ``str`` representation of ``value`` (``None`` becomes ``""``).'''
|
|
9
|
+
def summarize_terminal_error(error_message: str, *, max_length: int = 200) -> str:
|
|
10
|
+
"""Collapse whitespace in ``error_message`` and truncate to ``max_length`` characters."""
|
|
11
|
+
|
|
12
|
+
@dataclass
|
|
13
|
+
class ToolCall:
|
|
14
|
+
"""Normalized tool-call snapshot extracted from the underlying agent state."""
|
|
15
|
+
name: str
|
|
16
|
+
args: dict[str, Any]
|
|
17
|
+
output: str = ...
|
|
18
|
+
|
|
19
|
+
def read_tool_call(call: Any) -> tuple[str, dict[str, Any], str]:
|
|
20
|
+
"""Return a ``(name, args, output)`` triple for a tool call of any supported shape.
|
|
21
|
+
|
|
22
|
+
``name`` may be an empty string when the underlying payload is missing
|
|
23
|
+
a name; callers requiring a fallback label should apply one explicitly.
|
|
24
|
+
"""
|
|
25
|
+
def extract_outputs(last_result: list[Any]) -> list[str]:
|
|
26
|
+
"""Return one display string per entry in ``last_result``."""
|
|
27
|
+
def extract_tool_calls(agent: Agent) -> list[ToolCall]:
|
|
28
|
+
"""Extract serialized tool-call snapshots from the agent's last model output."""
|
|
29
|
+
def extract_final_output(agent: Agent) -> str:
|
|
30
|
+
"""Return the final textual output from the agent, or an empty string if absent."""
|
|
31
|
+
def parse_structured_data(output: str) -> tuple[dict[str, Any] | list[Any] | None, str | None]:
|
|
32
|
+
"""Parse the ``Extracted Content:`` block in ``output`` with fallback recovery.
|
|
33
|
+
|
|
34
|
+
The parser (1) trims trailers and slices from the first ``{``/``[``;
|
|
35
|
+
(2) attempts a direct ``json.loads``; (3) attempts to recover
|
|
36
|
+
concatenated top-level JSON objects; (4) extracts a balanced-prefix
|
|
37
|
+
JSON value with a depth-aware scanner that respects string literals
|
|
38
|
+
and parses it; and finally (5) falls back to ``json_repair.repair_json``.
|
|
39
|
+
"""
|
|
40
|
+
def validate_structured_data_output(tool_calls: list[ToolCall]) -> str | None:
|
|
41
|
+
"""Return a user-facing error when structured data extraction failed or yielded no usable entries."""
|
|
@@ -0,0 +1,204 @@
|
|
|
1
|
+
"""Utilities for recovering malformed structured-data payloads emitted by browser-use.
|
|
2
|
+
|
|
3
|
+
The helpers in this module normalize JSON-like strings that the structured
|
|
4
|
+
data extractor may return when model output is partially corrupted (e.g.
|
|
5
|
+
concatenated top-level JSON objects or trailing commas). They are used
|
|
6
|
+
internally by :mod:`gl_browser_use.parsing`.
|
|
7
|
+
|
|
8
|
+
Authors:
|
|
9
|
+
Reinhart Linanda (reinhart.linanda@gdplabs.id)
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import json
|
|
15
|
+
import logging
|
|
16
|
+
from typing import Any
|
|
17
|
+
|
|
18
|
+
from json_repair import repair_json
|
|
19
|
+
|
|
20
|
+
logger = logging.getLogger(__name__)
|
|
21
|
+
|
|
22
|
+
_OPENERS: tuple[str, ...] = ("{", "[")
|
|
23
|
+
_MATCHING_CLOSER: dict[str, str] = {"{": "}", "[": "]"}
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def extract_balanced_json_value(text: str) -> str | None:
|
|
27
|
+
"""Return the first balanced JSON value found in ``text`` (respecting string literals).
|
|
28
|
+
|
|
29
|
+
Returns ``None`` when no ``{``/``[`` is found or when the scan never
|
|
30
|
+
closes cleanly (unbalanced, truncated, or ends inside a string).
|
|
31
|
+
"""
|
|
32
|
+
start = _find_first_opener(text)
|
|
33
|
+
if start is None:
|
|
34
|
+
return None
|
|
35
|
+
end = _scan_balanced_close(text, start)
|
|
36
|
+
if end is None:
|
|
37
|
+
return None
|
|
38
|
+
return text[start : end + 1]
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def recover_concatenated_json_objects(json_blob: str) -> dict[str, Any] | None:
|
|
42
|
+
"""Normalize concatenated JSON object strings into a single structured payload.
|
|
43
|
+
|
|
44
|
+
Returns ``None`` if ``json_blob`` does not contain multiple top-level JSON
|
|
45
|
+
objects or they cannot be parsed individually.
|
|
46
|
+
"""
|
|
47
|
+
segments = _split_top_level_json_objects(json_blob)
|
|
48
|
+
if len(segments) <= 1:
|
|
49
|
+
return None
|
|
50
|
+
|
|
51
|
+
try:
|
|
52
|
+
records = [json.loads(segment) for segment in segments]
|
|
53
|
+
except json.JSONDecodeError:
|
|
54
|
+
return None
|
|
55
|
+
|
|
56
|
+
count = len(records)
|
|
57
|
+
logger.info("Structured data extractor returned concatenated JSON objects. recovered=%s", count)
|
|
58
|
+
return {
|
|
59
|
+
"status": "ok",
|
|
60
|
+
"items": records,
|
|
61
|
+
"products": records,
|
|
62
|
+
"count": count,
|
|
63
|
+
"products_found": count,
|
|
64
|
+
"available": bool(records),
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def repair_json_blob(json_blob: str) -> str | None:
|
|
69
|
+
"""Apply ``json_repair.repair_json`` to ``json_blob`` and return the mutated payload.
|
|
70
|
+
|
|
71
|
+
Returns ``None`` when ``json_repair`` could not repair the input or when
|
|
72
|
+
the repair produced no changes.
|
|
73
|
+
"""
|
|
74
|
+
try:
|
|
75
|
+
repaired = repair_json(json_blob)
|
|
76
|
+
except Exception:
|
|
77
|
+
logger.debug("json_repair failed to repair structured data output.", exc_info=True)
|
|
78
|
+
return None
|
|
79
|
+
|
|
80
|
+
if not repaired or repaired == json_blob:
|
|
81
|
+
return None
|
|
82
|
+
return repaired
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def _find_first_opener(text: str) -> int | None:
|
|
86
|
+
positions = [text.find(opener) for opener in _OPENERS]
|
|
87
|
+
positions = [position for position in positions if position != -1]
|
|
88
|
+
return min(positions) if positions else None
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def _scan_balanced_close(text: str, start: int) -> int | None:
|
|
92
|
+
"""Return the index of the matching closer for the opener at ``start`` or ``None``."""
|
|
93
|
+
opener = text[start]
|
|
94
|
+
closer = _MATCHING_CLOSER.get(opener)
|
|
95
|
+
if closer is None:
|
|
96
|
+
return None
|
|
97
|
+
|
|
98
|
+
depth = 0
|
|
99
|
+
in_string = False
|
|
100
|
+
escaping = False
|
|
101
|
+
for index in range(start, len(text)):
|
|
102
|
+
char = text[index]
|
|
103
|
+
if escaping:
|
|
104
|
+
escaping = False
|
|
105
|
+
continue
|
|
106
|
+
if char == "\\":
|
|
107
|
+
escaping = True
|
|
108
|
+
continue
|
|
109
|
+
if char == '"':
|
|
110
|
+
in_string = not in_string
|
|
111
|
+
continue
|
|
112
|
+
if in_string:
|
|
113
|
+
continue
|
|
114
|
+
if char == opener:
|
|
115
|
+
depth += 1
|
|
116
|
+
elif char == closer:
|
|
117
|
+
depth -= 1
|
|
118
|
+
if depth == 0:
|
|
119
|
+
return index
|
|
120
|
+
return None
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def _split_top_level_json_objects(json_blob: str) -> list[str]:
|
|
124
|
+
splitter = _JsonObjectSplitter(json_blob)
|
|
125
|
+
return splitter.split_objects()
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def _has_only_separators(value: str) -> bool:
|
|
129
|
+
return value.strip(" \t\r\n,") == ""
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
class _JsonObjectSplitter:
|
|
133
|
+
"""Split concatenated top-level JSON objects while respecting string literals."""
|
|
134
|
+
|
|
135
|
+
def __init__(self, json_blob: str) -> None:
|
|
136
|
+
self.json_blob = json_blob
|
|
137
|
+
self.segments: list[str] = []
|
|
138
|
+
self.depth = 0
|
|
139
|
+
self.start: int | None = None
|
|
140
|
+
self.last_end = 0
|
|
141
|
+
self.in_string = False
|
|
142
|
+
self.escaping = False
|
|
143
|
+
|
|
144
|
+
def split_objects(self) -> list[str]:
|
|
145
|
+
if not self._parse_characters():
|
|
146
|
+
return []
|
|
147
|
+
if not self._validate_final_state():
|
|
148
|
+
return []
|
|
149
|
+
return self.segments
|
|
150
|
+
|
|
151
|
+
def _parse_characters(self) -> bool:
|
|
152
|
+
for index, char in enumerate(self.json_blob):
|
|
153
|
+
if not self._process_character(char, index):
|
|
154
|
+
return False
|
|
155
|
+
return True
|
|
156
|
+
|
|
157
|
+
def _process_character(self, char: str, index: int) -> bool:
|
|
158
|
+
if self.escaping:
|
|
159
|
+
self.escaping = False
|
|
160
|
+
return True
|
|
161
|
+
if char == "\\":
|
|
162
|
+
self.escaping = True
|
|
163
|
+
return True
|
|
164
|
+
if char == '"':
|
|
165
|
+
self.in_string = not self.in_string
|
|
166
|
+
return True
|
|
167
|
+
if self.in_string:
|
|
168
|
+
return True
|
|
169
|
+
return self._process_non_string_char(char, index)
|
|
170
|
+
|
|
171
|
+
def _process_non_string_char(self, char: str, index: int) -> bool:
|
|
172
|
+
if char == "{":
|
|
173
|
+
return self._handle_open_brace(index)
|
|
174
|
+
if char == "}":
|
|
175
|
+
return self._handle_close_brace(index)
|
|
176
|
+
return True
|
|
177
|
+
|
|
178
|
+
def _handle_open_brace(self, index: int) -> bool:
|
|
179
|
+
if self.depth == 0:
|
|
180
|
+
if not _has_only_separators(self.json_blob[self.last_end : index]):
|
|
181
|
+
return False
|
|
182
|
+
self.start = index
|
|
183
|
+
self.depth += 1
|
|
184
|
+
return True
|
|
185
|
+
|
|
186
|
+
def _handle_close_brace(self, index: int) -> bool:
|
|
187
|
+
self.depth -= 1
|
|
188
|
+
if self.depth == 0 and self.start is not None:
|
|
189
|
+
end = index + 1
|
|
190
|
+
self.segments.append(self.json_blob[self.start : end])
|
|
191
|
+
self.last_end = end
|
|
192
|
+
return True
|
|
193
|
+
|
|
194
|
+
def _validate_final_state(self) -> bool:
|
|
195
|
+
if self.depth != 0 or self.in_string:
|
|
196
|
+
return False
|
|
197
|
+
return _has_only_separators(self.json_blob[self.last_end :])
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
__all__ = [
|
|
201
|
+
"extract_balanced_json_value",
|
|
202
|
+
"recover_concatenated_json_objects",
|
|
203
|
+
"repair_json_blob",
|
|
204
|
+
]
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
from _typeshed import Incomplete
|
|
2
|
+
from typing import Any
|
|
3
|
+
|
|
4
|
+
__all__ = ['extract_balanced_json_value', 'recover_concatenated_json_objects', 'repair_json_blob']
|
|
5
|
+
|
|
6
|
+
def extract_balanced_json_value(text: str) -> str | None:
|
|
7
|
+
"""Return the first balanced JSON value found in ``text`` (respecting string literals).
|
|
8
|
+
|
|
9
|
+
Returns ``None`` when no ``{``/``[`` is found or when the scan never
|
|
10
|
+
closes cleanly (unbalanced, truncated, or ends inside a string).
|
|
11
|
+
"""
|
|
12
|
+
def recover_concatenated_json_objects(json_blob: str) -> dict[str, Any] | None:
|
|
13
|
+
"""Normalize concatenated JSON object strings into a single structured payload.
|
|
14
|
+
|
|
15
|
+
Returns ``None`` if ``json_blob`` does not contain multiple top-level JSON
|
|
16
|
+
objects or they cannot be parsed individually.
|
|
17
|
+
"""
|
|
18
|
+
def repair_json_blob(json_blob: str) -> str | None:
|
|
19
|
+
"""Apply ``json_repair.repair_json`` to ``json_blob`` and return the mutated payload.
|
|
20
|
+
|
|
21
|
+
Returns ``None`` when ``json_repair`` could not repair the input or when
|
|
22
|
+
the repair produced no changes.
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
class _JsonObjectSplitter:
|
|
26
|
+
"""Split concatenated top-level JSON objects while respecting string literals."""
|
|
27
|
+
json_blob: Incomplete
|
|
28
|
+
segments: list[str]
|
|
29
|
+
depth: int
|
|
30
|
+
start: int | None
|
|
31
|
+
last_end: int
|
|
32
|
+
in_string: bool
|
|
33
|
+
escaping: bool
|
|
34
|
+
def __init__(self, json_blob: str) -> None: ...
|
|
35
|
+
def split_objects(self) -> list[str]: ...
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
"""Session-aware helpers for retry decisions for GL Browser Use SDK.
|
|
2
|
+
|
|
3
|
+
Authors:
|
|
4
|
+
Reinhart Linanda (reinhart.linanda@gdplabs.id)
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from dataclasses import dataclass
|
|
10
|
+
|
|
11
|
+
from gl_browser_use.parsing import coerce_string
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@dataclass(frozen=True)
|
|
15
|
+
class SessionErrorCategory:
|
|
16
|
+
"""Session error classification for retry decisions."""
|
|
17
|
+
|
|
18
|
+
name: str
|
|
19
|
+
markers: tuple[str, ...]
|
|
20
|
+
fatal: bool
|
|
21
|
+
retryable: bool
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
_FATAL_SESSION_ERROR_CATEGORIES: tuple[SessionErrorCategory, ...] = (
|
|
25
|
+
SessionErrorCategory(
|
|
26
|
+
name="browser_closed",
|
|
27
|
+
markers=(
|
|
28
|
+
"browser has been closed",
|
|
29
|
+
"target page, context or browser has been closed",
|
|
30
|
+
),
|
|
31
|
+
fatal=True,
|
|
32
|
+
retryable=True,
|
|
33
|
+
),
|
|
34
|
+
SessionErrorCategory(
|
|
35
|
+
name="websocket_disconnect",
|
|
36
|
+
markers=(
|
|
37
|
+
"code=1006",
|
|
38
|
+
"websocket was closed before the connection was established",
|
|
39
|
+
"websocket error",
|
|
40
|
+
),
|
|
41
|
+
fatal=True,
|
|
42
|
+
retryable=True,
|
|
43
|
+
),
|
|
44
|
+
)
|
|
45
|
+
|
|
46
|
+
_FATAL_SESSION_ERROR_LOOKUP: dict[str, SessionErrorCategory] = {
|
|
47
|
+
marker.lower(): category for category in _FATAL_SESSION_ERROR_CATEGORIES for marker in category.markers
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def categorize_fatal_session_message(message: str) -> SessionErrorCategory | None:
|
|
52
|
+
"""Return the matching :class:`SessionErrorCategory` for ``message`` or ``None``."""
|
|
53
|
+
if not message:
|
|
54
|
+
return None
|
|
55
|
+
lowered = coerce_string(message).lower()
|
|
56
|
+
for marker, category in _FATAL_SESSION_ERROR_LOOKUP.items():
|
|
57
|
+
if marker in lowered:
|
|
58
|
+
return category
|
|
59
|
+
return None
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def is_recoverable_session_error(message: str) -> bool:
|
|
63
|
+
"""Return True when ``message`` matches a recoverable session-failure category."""
|
|
64
|
+
category = categorize_fatal_session_message(message)
|
|
65
|
+
return bool(category and category.retryable)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
__all__ = [
|
|
69
|
+
"SessionErrorCategory",
|
|
70
|
+
"categorize_fatal_session_message",
|
|
71
|
+
"is_recoverable_session_error",
|
|
72
|
+
]
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
from dataclasses import dataclass
|
|
2
|
+
|
|
3
|
+
__all__ = ['SessionErrorCategory', 'categorize_fatal_session_message', 'is_recoverable_session_error']
|
|
4
|
+
|
|
5
|
+
@dataclass(frozen=True)
|
|
6
|
+
class SessionErrorCategory:
|
|
7
|
+
"""Session error classification for retry decisions."""
|
|
8
|
+
name: str
|
|
9
|
+
markers: tuple[str, ...]
|
|
10
|
+
fatal: bool
|
|
11
|
+
retryable: bool
|
|
12
|
+
|
|
13
|
+
def categorize_fatal_session_message(message: str) -> SessionErrorCategory | None:
|
|
14
|
+
"""Return the matching :class:`SessionErrorCategory` for ``message`` or ``None``."""
|
|
15
|
+
def is_recoverable_session_error(message: str) -> bool:
|
|
16
|
+
"""Return True when ``message`` matches a recoverable session-failure category."""
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
"""Object storage abstractions and implementations.
|
|
2
|
+
|
|
3
|
+
Authors:
|
|
4
|
+
Reinhart Linanda (reinhart.linanda@gdplabs.id)
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from gl_browser_use.errors import MINIO_EXTRA, BrowserUseMissingDependencyError, install_hint
|
|
10
|
+
from gl_browser_use.storage.base import OBJECT_NAME_PREFIX, ObjectStorage
|
|
11
|
+
|
|
12
|
+
__all__ = [
|
|
13
|
+
"OBJECT_NAME_PREFIX",
|
|
14
|
+
"ObjectStorage",
|
|
15
|
+
"MinIOS3CompatibleStorage",
|
|
16
|
+
]
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def __getattr__(name: str):
|
|
20
|
+
"""Lazy-load storage providers behind optional dependencies."""
|
|
21
|
+
if name == "MinIOS3CompatibleStorage":
|
|
22
|
+
try:
|
|
23
|
+
from gl_browser_use.storage.minio_compatible import MinIOS3CompatibleStorage
|
|
24
|
+
except ImportError as exc:
|
|
25
|
+
hint = install_hint(MINIO_EXTRA)
|
|
26
|
+
raise BrowserUseMissingDependencyError(
|
|
27
|
+
f"The MinIO storage provider requires optional dependencies that are not installed. "
|
|
28
|
+
f"Install them via `{hint}`.",
|
|
29
|
+
missing_dependency="minio",
|
|
30
|
+
package_hint=hint,
|
|
31
|
+
origin=exc,
|
|
32
|
+
) from exc
|
|
33
|
+
return MinIOS3CompatibleStorage
|
|
34
|
+
|
|
35
|
+
raise AttributeError(f"module 'gl_browser_use.storage' has no attribute {name!r}")
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def __dir__() -> list[str]:
|
|
39
|
+
return sorted(__all__)
|