agent-flywheel 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_flywheel/__init__.py +1067 -0
- agent_flywheel/_http.py +176 -0
- agent_flywheel/adapters/__init__.py +49 -0
- agent_flywheel/adapters/adk.py +268 -0
- agent_flywheel/adapters/langchain.py +288 -0
- agent_flywheel/adapters/prime_agent.py +1232 -0
- agent_flywheel/binding.py +360 -0
- agent_flywheel/child.py +63 -0
- agent_flywheel/cli.py +228 -0
- agent_flywheel/contract.py +411 -0
- agent_flywheel/endpoint.py +225 -0
- agent_flywheel/entrypoint.py +146 -0
- agent_flywheel/errors.py +87 -0
- agent_flywheel/event_stream.py +281 -0
- agent_flywheel/evidence.py +140 -0
- agent_flywheel/execution_identity.py +217 -0
- agent_flywheel/export.py +344 -0
- agent_flywheel/host_description.py +870 -0
- agent_flywheel/hosts/__init__.py +53 -0
- agent_flywheel/hosts/goose/plugin.json +7 -0
- agent_flywheel/hosts/scripts/percepteye_record.py +494 -0
- agent_flywheel/hosts/scripts/percepteye_report.py +192 -0
- agent_flywheel/install_hooks.py +301 -0
- agent_flywheel/introspection.py +619 -0
- agent_flywheel/isolation.py +227 -0
- agent_flywheel/outcomes.py +831 -0
- agent_flywheel/policy.py +490 -0
- agent_flywheel/production.py +1589 -0
- agent_flywheel/prompt.py +178 -0
- agent_flywheel/runner.py +1623 -0
- agent_flywheel/tracing.py +279 -0
- agent_flywheel/transport.py +328 -0
- agent_flywheel-0.1.0.dist-info/METADATA +429 -0
- agent_flywheel-0.1.0.dist-info/RECORD +37 -0
- agent_flywheel-0.1.0.dist-info/WHEEL +4 -0
- agent_flywheel-0.1.0.dist-info/entry_points.txt +2 -0
- agent_flywheel-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,1067 @@
|
|
|
1
|
+
"""agent-flywheel -- run your own agent against PerceptEye rollouts.
|
|
2
|
+
|
|
3
|
+
import agent_flywheel as fw
|
|
4
|
+
|
|
5
|
+
fw.serve(
|
|
6
|
+
"my_package.agent:build", # importable, so a child can rebuild it
|
|
7
|
+
agent_id="my-coding-agent",
|
|
8
|
+
input_shape="text",
|
|
9
|
+
control_plane_url="https://launch.percepteye.ai/api/flywheel/v1",
|
|
10
|
+
)
|
|
11
|
+
|
|
12
|
+
Outbound only. The flywheel dials out, claims a rollout, runs it, reports back.
|
|
13
|
+
No listening socket, no inbound firewall rule, no public DNS.
|
|
14
|
+
|
|
15
|
+
Two things are worth knowing before you read further.
|
|
16
|
+
|
|
17
|
+
**There are two modes, and they are a deployment switch.** ``training`` --
|
|
18
|
+
Percepteye supplies the model endpoint, your flywheel supplies the agent loop
|
|
19
|
+
-- and ``production``, in which ``serve()`` returns 0 without connecting to
|
|
20
|
+
anything. Set ``PERCEPTEYE_AGENT_MODE`` and the same image does either, because
|
|
21
|
+
"stop contributing capacity" is an operational decision and it should not require
|
|
22
|
+
editing the source to express it.
|
|
23
|
+
|
|
24
|
+
That is the ONLY axis this word names here. An earlier SDK shipped five modes
|
|
25
|
+
of which four were byte-identical strings in one payload field the server
|
|
26
|
+
never read, and the rule that came out of it stands: a mode has to change what
|
|
27
|
+
the SDK does. These two do -- one claims rollouts and the other opens no
|
|
28
|
+
socket.
|
|
29
|
+
|
|
30
|
+
On the wire, ``serve()`` registers as ``"rollout"`` -- that field names the
|
|
31
|
+
protocol this process speaks, and it is the value intake has always received
|
|
32
|
+
from a training donor. ``attach()`` registers as ``"production"``, because a
|
|
33
|
+
training donor and a production reader are the one distinction a ``mode`` on a
|
|
34
|
+
registration exists to draw, and while both sent ``"rollout"`` the field
|
|
35
|
+
answered nothing. So the wire carries two values and this parameter carries
|
|
36
|
+
two modes, but they are not the same axis and do not have to agree.
|
|
37
|
+
|
|
38
|
+
**Reporting "I don't know" is a supported answer.** When your tool wrapper
|
|
39
|
+
cannot tell whether a call succeeded, say ``outcome="unknown"``. It is
|
|
40
|
+
recorded as unknown and the verifiers that depend on it abstain. It is never
|
|
41
|
+
rounded up. This matters more than it sounds: the alternative is a reward
|
|
42
|
+
signal that scores a failed call as a landed one, and a policy trained to
|
|
43
|
+
believe it.
|
|
44
|
+
"""
|
|
45
|
+
from __future__ import annotations
|
|
46
|
+
|
|
47
|
+
import logging
|
|
48
|
+
import os
|
|
49
|
+
import warnings
|
|
50
|
+
from collections.abc import Callable, Mapping
|
|
51
|
+
from typing import Any, Literal
|
|
52
|
+
|
|
53
|
+
from .contract import (
|
|
54
|
+
CONTRACT_VERSION,
|
|
55
|
+
ExclusionReason,
|
|
56
|
+
Outcome,
|
|
57
|
+
RolloutOutput,
|
|
58
|
+
RolloutRequest,
|
|
59
|
+
ToolCallOutcome,
|
|
60
|
+
)
|
|
61
|
+
from .entrypoint import TASK, InputShape, resolve_entrypoint
|
|
62
|
+
from .errors import (
|
|
63
|
+
ConfigurationError,
|
|
64
|
+
ContractViolation,
|
|
65
|
+
EntrypointError,
|
|
66
|
+
FlywheelError,
|
|
67
|
+
LeaseLost,
|
|
68
|
+
ModeNotAvailable,
|
|
69
|
+
RoutingNotInstallable,
|
|
70
|
+
TransportError,
|
|
71
|
+
)
|
|
72
|
+
from .event_stream import CommandOutputAdapter
|
|
73
|
+
from .execution_identity import ExecutionChild, ExecutionSnapshot
|
|
74
|
+
from .host_description import WARMUP_PROMPT, warm_up_command
|
|
75
|
+
|
|
76
|
+
# Re-exported, not used in this module. The redundant alias is the
|
|
77
|
+
# explicit-re-export form: it keeps `from agent_flywheel import
|
|
78
|
+
# read_discovered_agent` working -- which it does today -- without adding the
|
|
79
|
+
# name to __all__ and thereby claiming it as a supported public surface.
|
|
80
|
+
from .host_description import read_discovered_agent as read_discovered_agent
|
|
81
|
+
from .introspection import DiscoveredAgent, ToolDefinition
|
|
82
|
+
from .introspection import introspect as introspect_agent
|
|
83
|
+
from .isolation import unroutable_providers_present
|
|
84
|
+
from .outcomes import (
|
|
85
|
+
CaptureScope,
|
|
86
|
+
Trajectory,
|
|
87
|
+
current_capture,
|
|
88
|
+
current_rollout_id,
|
|
89
|
+
record_tool_call,
|
|
90
|
+
)
|
|
91
|
+
from .policy import (
|
|
92
|
+
ARCHIVE_REF_PREFIXES,
|
|
93
|
+
PolicySnapshot,
|
|
94
|
+
PolicySource,
|
|
95
|
+
is_archive_ref,
|
|
96
|
+
missing_credentials_cause,
|
|
97
|
+
policy_source,
|
|
98
|
+
)
|
|
99
|
+
from .production import Attachment, CaptureVerdict, attach, capture_verdict
|
|
100
|
+
from .prompt import (
|
|
101
|
+
SERVABLE_PROMPT_STATUSES,
|
|
102
|
+
PromptDecision,
|
|
103
|
+
prompt_decision,
|
|
104
|
+
)
|
|
105
|
+
from .runner import RolloutRunner
|
|
106
|
+
from .transport import AttachTransport, FlywheelTransport
|
|
107
|
+
|
|
108
|
+
__version__ = "0.1.0"
|
|
109
|
+
|
|
110
|
+
#: This package's logger. It has no handler and never installs one -- it
|
|
111
|
+
#: emits into whatever logging the customer's process already configured,
|
|
112
|
+
#: exactly as the OpenTelemetry extra emits into their TracerProvider.
|
|
113
|
+
#: Used for the answers a caller CANNOT see in the return value: `{}` from
|
|
114
|
+
#: `endpoint_kwargs` and `current_prompt` is the same value for "nothing is
|
|
115
|
+
#: deployed" and "you never told me where the control plane is", and only
|
|
116
|
+
#: one of those is a thing an operator can fix.
|
|
117
|
+
_log = logging.getLogger(__name__)
|
|
118
|
+
|
|
119
|
+
__all__ = [
|
|
120
|
+
"ARCHIVE_REF_PREFIXES",
|
|
121
|
+
"CONTRACT_VERSION",
|
|
122
|
+
"SERVABLE_PROMPT_STATUSES",
|
|
123
|
+
"TASK",
|
|
124
|
+
"AttachTransport",
|
|
125
|
+
"Attachment",
|
|
126
|
+
"CaptureScope",
|
|
127
|
+
"CaptureVerdict",
|
|
128
|
+
"CommandOutputAdapter",
|
|
129
|
+
"ConfigurationError",
|
|
130
|
+
"ContractViolation",
|
|
131
|
+
"DiscoveredAgent",
|
|
132
|
+
"EntrypointError",
|
|
133
|
+
"ExclusionReason",
|
|
134
|
+
"ExecutionChild",
|
|
135
|
+
"ExecutionSnapshot",
|
|
136
|
+
"FlywheelError",
|
|
137
|
+
"FlywheelTransport",
|
|
138
|
+
"InputShape",
|
|
139
|
+
"LeaseLost",
|
|
140
|
+
"ModeNotAvailable",
|
|
141
|
+
"Outcome",
|
|
142
|
+
"PolicySnapshot",
|
|
143
|
+
"PolicySource",
|
|
144
|
+
"PromptDecision",
|
|
145
|
+
"RolloutOutput",
|
|
146
|
+
"RolloutRequest",
|
|
147
|
+
"RolloutRunner",
|
|
148
|
+
"RoutingNotInstallable",
|
|
149
|
+
"ToolCallOutcome",
|
|
150
|
+
"ToolDefinition",
|
|
151
|
+
"Trajectory",
|
|
152
|
+
"TransportError",
|
|
153
|
+
"__version__",
|
|
154
|
+
"apply_current_policy",
|
|
155
|
+
"attach",
|
|
156
|
+
"capture_verdict",
|
|
157
|
+
"current_capture",
|
|
158
|
+
"current_mode",
|
|
159
|
+
"current_policy",
|
|
160
|
+
"current_prompt",
|
|
161
|
+
"current_rollout_id",
|
|
162
|
+
"endpoint_kwargs",
|
|
163
|
+
"introspect_agent",
|
|
164
|
+
"is_archive_ref",
|
|
165
|
+
"policy_source",
|
|
166
|
+
"prompt_decision",
|
|
167
|
+
"record_tool_call",
|
|
168
|
+
"resolve_entrypoint",
|
|
169
|
+
"resolve_mode",
|
|
170
|
+
"serve",
|
|
171
|
+
]
|
|
172
|
+
|
|
173
|
+
#: Every environment variable this package reads. Two of them select
|
|
174
|
+
#: behaviour -- PERCEPTEYE_AGENT_MODE and PERCEPTEYE_INTROSPECT -- and the rest are
|
|
175
|
+
#: addresses and credentials.
|
|
176
|
+
#:
|
|
177
|
+
#: PERCEPTEYE_AGENT_MODE names what the AGENT is for -- training or serving
|
|
178
|
+
#: users -- rather than naming a knob on this SDK, which is why "agent" is in
|
|
179
|
+
#: it. It carries the prefix because an unprefixed AGENT_MODE is a name we
|
|
180
|
+
#: would not own: it is plausible enough that a customer already exports it for
|
|
181
|
+
#: something unrelated, and an unrecognised value RAISES here, so a collision
|
|
182
|
+
#: would stop a flywheel over a word that was never addressed to us. The prefix
|
|
183
|
+
#: buys the right to refuse a value we cannot parse.
|
|
184
|
+
#:
|
|
185
|
+
#: Being inside the namespace also means `build_child_env` strips it from a
|
|
186
|
+
#: rollout child along with everything else PERCEPTEYE_, which is correct: it
|
|
187
|
+
#: configures the process that DONATES capacity, and a child does not call
|
|
188
|
+
#: serve().
|
|
189
|
+
ENV_CONTROL_PLANE_URL = "PERCEPTEYE_CONTROL_PLANE_URL"
|
|
190
|
+
ENV_API_KEY = "PERCEPTEYE_API_KEY"
|
|
191
|
+
ENV_MODE = "PERCEPTEYE_AGENT_MODE"
|
|
192
|
+
ENV_ROLLOUT_ID = "PERCEPTEYE_ROLLOUT_ID" # set by us, for the child
|
|
193
|
+
ENV_TRAJECTORY_DIR = "PERCEPTEYE_TRAJECTORY_DIR" # set by us, for the child
|
|
194
|
+
|
|
195
|
+
#: What this PROCESS is for. ``"rollout"`` is the legacy spelling of
|
|
196
|
+
#: ``"training"`` and still means exactly that.
|
|
197
|
+
#:
|
|
198
|
+
#: This used to be ``Literal["rollout"]`` -- one value, which `serve()` checked
|
|
199
|
+
#: and rejected everything else, so the parameter carried zero bits and the
|
|
200
|
+
#: field of the same name on the registration payload carried zero bits with
|
|
201
|
+
#: it. It is now the deployment switch, because the alternative was a SECOND
|
|
202
|
+
#: thing called "mode" sitting next to a vestigial first one.
|
|
203
|
+
FlywheelMode = Literal["training", "production", "rollout"]
|
|
204
|
+
|
|
205
|
+
#: Accepted spellings, resolved to the two real modes. Deliberately a small
|
|
206
|
+
#: closed set: an unrecognised value RAISES rather than guessing, because the
|
|
207
|
+
#: two ways of guessing wrong are "silently contribute capacity the operator meant
|
|
208
|
+
#: to switch off" and "silently contribute nothing while they believe training is
|
|
209
|
+
#: running". Neither is recoverable by the operator, because both look exactly
|
|
210
|
+
#: like success.
|
|
211
|
+
_MODE_ALIASES: dict[str, str] = {
|
|
212
|
+
"training": "training",
|
|
213
|
+
"train": "training",
|
|
214
|
+
"rollout": "training", # the legacy value, unchanged in meaning
|
|
215
|
+
"rollouts": "training",
|
|
216
|
+
"fine-tuning": "training",
|
|
217
|
+
"finetuning": "training",
|
|
218
|
+
"improving": "training",
|
|
219
|
+
"production": "production",
|
|
220
|
+
"prod": "production",
|
|
221
|
+
"serving": "production",
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
def _mode_from(raw: Any, *, source: str) -> str:
|
|
226
|
+
"""One spelling to one mode, or a refusal that names where it came from."""
|
|
227
|
+
key = str(raw).strip().lower()
|
|
228
|
+
if key not in _MODE_ALIASES:
|
|
229
|
+
raise ConfigurationError(
|
|
230
|
+
f"{source}{raw!r} is not a mode. Use 'training' (this process "
|
|
231
|
+
f"does training work: it claims rollouts and runs them) or "
|
|
232
|
+
f"'production' (it does not). Recognised spellings: "
|
|
233
|
+
f"{', '.join(sorted(_MODE_ALIASES))}."
|
|
234
|
+
)
|
|
235
|
+
return _MODE_ALIASES[key]
|
|
236
|
+
|
|
237
|
+
|
|
238
|
+
def _resolve_mode_source(explicit: str | None = None) -> tuple[str, str]:
|
|
239
|
+
"""The mode, and which of ``argument``/``environment``/``default`` set it.
|
|
240
|
+
|
|
241
|
+
An EMPTY explicit argument falls through to the environment rather than
|
|
242
|
+
being honoured as a value. ``mode=cfg.get("percepteye_mode", "")`` is the
|
|
243
|
+
ordinary way a config layer says "not set", and treating it as a value
|
|
244
|
+
silently pinned the process to "training" without ever reading
|
|
245
|
+
``PERCEPTEYE_AGENT_MODE`` -- so an operator who had flipped the switch to stop
|
|
246
|
+
contributing kept contributing, on their own credentials, with nothing printed.
|
|
247
|
+
That is the exact failure ``_MODE_ALIASES`` calls unrecoverable, arrived at
|
|
248
|
+
through the one input that skipped the closed-set check entirely.
|
|
249
|
+
"""
|
|
250
|
+
if explicit is not None and str(explicit).strip() != "":
|
|
251
|
+
return _mode_from(explicit, source="mode="), "argument"
|
|
252
|
+
env = os.environ.get(ENV_MODE)
|
|
253
|
+
if env is not None and str(env).strip() != "":
|
|
254
|
+
return _mode_from(env, source=f"{ENV_MODE}="), "environment"
|
|
255
|
+
return "training", "default"
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
def resolve_mode(explicit: str | None = None) -> str:
|
|
259
|
+
"""Which mode this process is in: ``"training"`` or ``"production"``.
|
|
260
|
+
|
|
261
|
+
Precedence is explicit argument, then ``PERCEPTEYE_AGENT_MODE``, then
|
|
262
|
+
``"training"``. The argument wins so that a caller who has hard-coded a
|
|
263
|
+
mode gets what they wrote; the environment provides the default in the
|
|
264
|
+
ordinary case where nobody passed one, which is what lets an operator flip
|
|
265
|
+
a deployment WITHOUT editing code. An empty or whitespace argument counts
|
|
266
|
+
as "nobody passed one".
|
|
267
|
+
|
|
268
|
+
Raises:
|
|
269
|
+
ConfigurationError: for any value that is not a recognised mode. There
|
|
270
|
+
is no safe guess -- see ``_MODE_ALIASES``.
|
|
271
|
+
"""
|
|
272
|
+
return _resolve_mode_source(explicit)[0]
|
|
273
|
+
|
|
274
|
+
|
|
275
|
+
def current_mode() -> str:
|
|
276
|
+
"""The mode this process would serve in, from the environment alone.
|
|
277
|
+
|
|
278
|
+
``"training"`` or ``"production"``. Read it to log which way a deployment
|
|
279
|
+
is configured, or to assert it in a smoke test. Raises for an unrecognised
|
|
280
|
+
``PERCEPTEYE_AGENT_MODE``, exactly as :func:`serve` does, so a typo surfaces
|
|
281
|
+
wherever it is first read rather than only at startup.
|
|
282
|
+
"""
|
|
283
|
+
return resolve_mode()
|
|
284
|
+
|
|
285
|
+
|
|
286
|
+
def _env_true(name: str, *, default: bool) -> bool:
|
|
287
|
+
"""An env kill switch that only a DELIBERATE value can flip.
|
|
288
|
+
|
|
289
|
+
Unset means the default. ``0``/``false``/``no``/``off`` disable; anything
|
|
290
|
+
else is treated as enabled, so a typo cannot silently switch capture off
|
|
291
|
+
and leave the operator believing it is on.
|
|
292
|
+
"""
|
|
293
|
+
raw = os.environ.get(name)
|
|
294
|
+
if raw is None or raw.strip() == "":
|
|
295
|
+
return default
|
|
296
|
+
return raw.strip().lower() not in ("0", "false", "no", "off")
|
|
297
|
+
|
|
298
|
+
|
|
299
|
+
def current_policy(
|
|
300
|
+
*,
|
|
301
|
+
agent_id: str,
|
|
302
|
+
control_plane_url: str | None = None,
|
|
303
|
+
api_key: str | None = None,
|
|
304
|
+
) -> dict[str, Any]:
|
|
305
|
+
"""Ask the control plane what model this agent should be running.
|
|
306
|
+
|
|
307
|
+
For the customer's OWN agent process — the one serving their users —
|
|
308
|
+
which is a different process from :func:`serve`. ``serve`` contributes
|
|
309
|
+
capacity for training; this is how the production agent picks up what
|
|
310
|
+
that training produced.
|
|
311
|
+
|
|
312
|
+
Returns ``{}`` when the control plane cannot answer, and callers should
|
|
313
|
+
read that as "keep your current configuration". An unreachable control
|
|
314
|
+
plane must never stop a customer's agent from serving its users.
|
|
315
|
+
|
|
316
|
+
What comes back is an ENDPOINT, because that is the only form an agent
|
|
317
|
+
can act on — an artifact URI is not callable. It is shaped for an
|
|
318
|
+
OpenAI-compatible client::
|
|
319
|
+
|
|
320
|
+
pol = fw.current_policy(agent_id="coding-agent")
|
|
321
|
+
ep = pol.get("endpoint")
|
|
322
|
+
if ep:
|
|
323
|
+
client = OpenAI(base_url=ep["base_url"],
|
|
324
|
+
api_key=os.environ[ep["api_key_env"]])
|
|
325
|
+
model = ep["model"]
|
|
326
|
+
|
|
327
|
+
``endpoint`` is absent when the control plane has nothing deployed to
|
|
328
|
+
hand you — keep your current configuration, do not fall back to
|
|
329
|
+
something you guessed.
|
|
330
|
+
"""
|
|
331
|
+
url = control_plane_url or os.environ.get(ENV_CONTROL_PLANE_URL) or ""
|
|
332
|
+
key = api_key or os.environ.get(ENV_API_KEY) or ""
|
|
333
|
+
cause = missing_credentials_cause(url, key)
|
|
334
|
+
if cause:
|
|
335
|
+
_log.warning(
|
|
336
|
+
"agent-flywheel: not reading the served policy for agent_id=%r: %s, so "
|
|
337
|
+
"the control plane was never asked. This agent keeps the model it "
|
|
338
|
+
"is on.", agent_id, cause,
|
|
339
|
+
)
|
|
340
|
+
return {}
|
|
341
|
+
from ._http import ControlPlaneClient
|
|
342
|
+
|
|
343
|
+
client = ControlPlaneClient(base_url=url, api_key=key)
|
|
344
|
+
try:
|
|
345
|
+
from .transport import AttachTransport
|
|
346
|
+
|
|
347
|
+
return AttachTransport(client, agent_id=agent_id).policy_current()
|
|
348
|
+
finally:
|
|
349
|
+
client.close()
|
|
350
|
+
|
|
351
|
+
|
|
352
|
+
def current_prompt(
|
|
353
|
+
*,
|
|
354
|
+
agent_id: str,
|
|
355
|
+
control_plane_url: str | None = None,
|
|
356
|
+
api_key: str | None = None,
|
|
357
|
+
) -> dict[str, Any]:
|
|
358
|
+
"""The approved system prompt this agent should be running, or ``{}``.
|
|
359
|
+
|
|
360
|
+
The other half of :func:`current_policy`. That one answers "which model";
|
|
361
|
+
training also produces a system PROMPT, co-optimized with that model and
|
|
362
|
+
gated by the same human approval — and until this existed an agent could
|
|
363
|
+
learn only that the prompt had moved, never what it moved to. The snapshot
|
|
364
|
+
in :class:`PolicySource` says as much in its own comment: its generation
|
|
365
|
+
changes for a prompt change, "which this snapshot does not carry and a
|
|
366
|
+
client rebuild would not apply".
|
|
367
|
+
|
|
368
|
+
Returns the deliverable::
|
|
369
|
+
|
|
370
|
+
{"name": ..., "version": 3, "sha256": ..., "text": "...",
|
|
371
|
+
"candidate_kind": "replacement", "status": "approved",
|
|
372
|
+
"model_prompt_checksum": ..., "stale_baseline": false}
|
|
373
|
+
|
|
374
|
+
``{}`` means KEEP THE PROMPT YOU HAVE — nothing has passed the approval
|
|
375
|
+
gate for this agent, or the control plane could not answer. It never means
|
|
376
|
+
"use no prompt", and an unreachable control plane must never blank a
|
|
377
|
+
running agent's instructions.
|
|
378
|
+
|
|
379
|
+
**Deliberately not applied for you, and that is the difference from**
|
|
380
|
+
:func:`apply_current_policy`. Swapping a model endpoint is invisible to an
|
|
381
|
+
agent's behaviour; swapping its system prompt changes what it DOES. That is
|
|
382
|
+
a decision for the person who owns the agent, so this hands back the text
|
|
383
|
+
and stops.
|
|
384
|
+
|
|
385
|
+
**Do not re-derive whether it may be applied — ask** :func:`prompt_decision`.
|
|
386
|
+
This docstring used to list the fields to check and the order to check them
|
|
387
|
+
in, which made every caller reimplement a rule with no mechanism behind it.
|
|
388
|
+
The rule is a function now::
|
|
389
|
+
|
|
390
|
+
served = fw.current_prompt(agent_id="my-agent")
|
|
391
|
+
decision = fw.prompt_decision(served)
|
|
392
|
+
if decision.apply:
|
|
393
|
+
system_prompt = decision.text # never None here
|
|
394
|
+
else:
|
|
395
|
+
log.info("keeping the current prompt: %s", decision.reason)
|
|
396
|
+
|
|
397
|
+
It answers the same four questions the Node SDK's ``promptDecision`` asks,
|
|
398
|
+
in the same order and with the same sentences, so the two packages refuse
|
|
399
|
+
the same bundles for the same stated reason. See
|
|
400
|
+
:mod:`agent_flywheel.prompt`.
|
|
401
|
+
"""
|
|
402
|
+
url = control_plane_url or os.environ.get(ENV_CONTROL_PLANE_URL) or ""
|
|
403
|
+
key = api_key or os.environ.get(ENV_API_KEY) or ""
|
|
404
|
+
cause = missing_credentials_cause(url, key)
|
|
405
|
+
if cause:
|
|
406
|
+
# SAID, because `prompt_decision({})` blames the control plane for
|
|
407
|
+
# serving no approved prompt -- true of the empty answer and wrong
|
|
408
|
+
# about why, which sends an operator to the dashboard to look for an
|
|
409
|
+
# approval that was never the problem.
|
|
410
|
+
_log.warning(
|
|
411
|
+
"agent-flywheel: not reading the served prompt for agent_id=%r: %s, so "
|
|
412
|
+
"the control plane was never asked. This agent keeps the prompt it "
|
|
413
|
+
"assembles itself.", agent_id, cause,
|
|
414
|
+
)
|
|
415
|
+
return {}
|
|
416
|
+
from ._http import ControlPlaneClient
|
|
417
|
+
|
|
418
|
+
client = ControlPlaneClient(base_url=url, api_key=key)
|
|
419
|
+
try:
|
|
420
|
+
from .transport import AttachTransport
|
|
421
|
+
|
|
422
|
+
tr = AttachTransport(client, agent_id=agent_id)
|
|
423
|
+
ask = getattr(tr, "prompt_current", None)
|
|
424
|
+
return (ask() if callable(ask) else {}) or {}
|
|
425
|
+
finally:
|
|
426
|
+
client.close()
|
|
427
|
+
|
|
428
|
+
|
|
429
|
+
def endpoint_kwargs(
|
|
430
|
+
policy: Mapping[str, Any] | None, *, api_key: str | None = None,
|
|
431
|
+
) -> dict[str, Any]:
|
|
432
|
+
"""The ONE place a ``/policy/current`` body becomes client kwargs.
|
|
433
|
+
|
|
434
|
+
Extracted because there are now two callers — :func:`apply_current_policy`
|
|
435
|
+
and :class:`PolicySource` — and "a caller who unpacks it slightly
|
|
436
|
+
differently is a caller sampling from something other than what was
|
|
437
|
+
trained" applies at least as much between two of OUR surfaces as it does
|
|
438
|
+
between two of the customer's.
|
|
439
|
+
|
|
440
|
+
``{}`` means nothing to apply: no endpoint, no ``base_url``, or no
|
|
441
|
+
credential we can authenticate with. That last one is deliberate — an
|
|
442
|
+
endpoint we cannot authenticate to is not usable, and returning it would
|
|
443
|
+
produce 401s at the caller's first request rather than a legible "nothing
|
|
444
|
+
to apply" here.
|
|
445
|
+
"""
|
|
446
|
+
if not isinstance(policy, Mapping):
|
|
447
|
+
return {}
|
|
448
|
+
ep = policy.get("endpoint")
|
|
449
|
+
if not isinstance(ep, Mapping) or not ep.get("base_url"):
|
|
450
|
+
return {}
|
|
451
|
+
# PRECEDENCE, and the order is the contract:
|
|
452
|
+
#
|
|
453
|
+
# 1. an explicit ``api_key=`` argument — the caller's own override always
|
|
454
|
+
# wins, so a customer who wants full control keeps it;
|
|
455
|
+
# 2. ``endpoint.api_key`` from the control plane, when it sent one;
|
|
456
|
+
# 3. the environment variable the control plane NAMED;
|
|
457
|
+
# 4. ``PERCEPTEYE_API_KEY``.
|
|
458
|
+
#
|
|
459
|
+
# (2) exists for one case, and without it that case fails silently. A
|
|
460
|
+
# checkpoint promotion normally moves only ``model`` — base_url and the
|
|
461
|
+
# credential stay put, so an agent upgrades with nothing new. They do NOT
|
|
462
|
+
# stay put when the control plane moves an agent to a DIFFERENT provider:
|
|
463
|
+
# then base_url and the credential both change, and an agent cannot read an
|
|
464
|
+
# env var it was never given. Without (2) this function returns ``{}``, the
|
|
465
|
+
# caller keeps its previous configuration, and the agent FREEZES on its
|
|
466
|
+
# last policy — receiving no further upgrades and reporting nothing wrong.
|
|
467
|
+
#
|
|
468
|
+
# The value is never persisted or logged here; it goes straight into the
|
|
469
|
+
# client kwargs the caller builds, exactly like a key read from the
|
|
470
|
+
# environment.
|
|
471
|
+
key_env = str(ep.get("api_key_env") or ENV_API_KEY)
|
|
472
|
+
key = (
|
|
473
|
+
api_key
|
|
474
|
+
or str(ep.get("api_key") or "").strip()
|
|
475
|
+
or os.environ.get(key_env)
|
|
476
|
+
or os.environ.get(ENV_API_KEY)
|
|
477
|
+
)
|
|
478
|
+
if not key:
|
|
479
|
+
return {}
|
|
480
|
+
model = str(ep.get("model") or "")
|
|
481
|
+
# AN ARCHIVE URI IS NOT A MODEL NAME, and this is the model half of the
|
|
482
|
+
# pair rule `prompt_decision` enforces for the prompt half. The control
|
|
483
|
+
# plane already refuses to serve one; re-asserted here on the same argument
|
|
484
|
+
# as `SERVABLE_PROMPT_STATUSES`, because a candidate whose checkpoint has
|
|
485
|
+
# been uploaded but not yet deployed behind a serving endpoint would
|
|
486
|
+
# otherwise be handed to `OpenAI(model="gs://...")` and 404 on every
|
|
487
|
+
# completion. Refusing keeps the champion, which is what every other gate
|
|
488
|
+
# on this path fails closed to.
|
|
489
|
+
#
|
|
490
|
+
# SAID OUT LOUD, because silence here is indistinguishable from "nothing is
|
|
491
|
+
# deployed": `{}` is this function's ordinary answer and the caller is
|
|
492
|
+
# documented to read it as "keep your current configuration". An operator
|
|
493
|
+
# whose promotion never arrives needs the sentence, not the empty dict.
|
|
494
|
+
if is_archive_ref(model):
|
|
495
|
+
_log.warning(
|
|
496
|
+
"agent-flywheel: NOT applying the served policy for this agent: the "
|
|
497
|
+
"served model %r is a stored-artifact reference, not a name an "
|
|
498
|
+
"endpoint can serve; that address would 404 on every completion. "
|
|
499
|
+
"Keeping the model you are on.", model,
|
|
500
|
+
)
|
|
501
|
+
return {}
|
|
502
|
+
return {
|
|
503
|
+
"base_url": str(ep["base_url"]),
|
|
504
|
+
"api_key": str(key),
|
|
505
|
+
"model": model,
|
|
506
|
+
}
|
|
507
|
+
|
|
508
|
+
|
|
509
|
+
def apply_current_policy(
|
|
510
|
+
*,
|
|
511
|
+
agent_id: str,
|
|
512
|
+
control_plane_url: str | None = None,
|
|
513
|
+
api_key: str | None = None,
|
|
514
|
+
set_env: bool = False,
|
|
515
|
+
) -> dict[str, Any]:
|
|
516
|
+
"""Turn :func:`current_policy` into arguments you can hand a client.
|
|
517
|
+
|
|
518
|
+
The last mile. ``current_policy`` already returns the right answer, but
|
|
519
|
+
every caller had to unpack it the same way, and a caller who unpacks it
|
|
520
|
+
slightly differently is a caller sampling from something other than what
|
|
521
|
+
was trained. Returns kwargs shaped for an OpenAI-compatible client::
|
|
522
|
+
|
|
523
|
+
kw = fw.apply_current_policy(agent_id="coding-agent")
|
|
524
|
+
if kw:
|
|
525
|
+
client, model = OpenAI(**{k: v for k, v in kw.items()
|
|
526
|
+
if k != "model"}), kw["model"]
|
|
527
|
+
|
|
528
|
+
Returns ``{}`` when nothing is deployed or the control plane cannot
|
|
529
|
+
answer — keep your current configuration. Same fail-open as
|
|
530
|
+
``current_policy``, for the same reason: an unreachable control plane must
|
|
531
|
+
never stop a customer's agent from serving its users.
|
|
532
|
+
|
|
533
|
+
NOT FOR :func:`serve`, and that is a correctness boundary rather than a
|
|
534
|
+
style preference. A ``serve`` rollout samples through the per-rollout
|
|
535
|
+
GATEWAY url, which is what attributes the completion to that rollout and
|
|
536
|
+
captures it — and the gateway already rewrites the model field to the
|
|
537
|
+
policy under test. Pointing a donor agent at the policy endpoint directly
|
|
538
|
+
would bypass capture entirely and silently produce a run with nothing to
|
|
539
|
+
train on. Use this in your PRODUCTION agent, the one serving your users.
|
|
540
|
+
|
|
541
|
+
``set_env=True`` additionally exports ``OPENAI_BASE_URL`` /
|
|
542
|
+
``OPENAI_API_KEY`` for libraries that read the environment rather than
|
|
543
|
+
accept arguments. Off by default: mutating the process environment is a
|
|
544
|
+
side effect a caller should ask for.
|
|
545
|
+
"""
|
|
546
|
+
pol = current_policy(
|
|
547
|
+
agent_id=agent_id, control_plane_url=control_plane_url, api_key=api_key,
|
|
548
|
+
)
|
|
549
|
+
out = endpoint_kwargs(pol, api_key=api_key)
|
|
550
|
+
if not out:
|
|
551
|
+
return {}
|
|
552
|
+
if set_env:
|
|
553
|
+
os.environ["OPENAI_BASE_URL"] = out["base_url"]
|
|
554
|
+
os.environ["OPENAI_API_KEY"] = out["api_key"]
|
|
555
|
+
return out
|
|
556
|
+
|
|
557
|
+
|
|
558
|
+
def _describe_via_host(
|
|
559
|
+
command: list[Any], *, artifacts_dir: str, agent_id: str,
|
|
560
|
+
execution_snapshot: ExecutionSnapshot | None = None,
|
|
561
|
+
endpoint_env: tuple[str, str] | None = None,
|
|
562
|
+
) -> DiscoveredAgent | None:
|
|
563
|
+
"""Have the host describe a command agent, by running it once.
|
|
564
|
+
|
|
565
|
+
Returns None when the host produced nothing, having said why. The failure
|
|
566
|
+
is silent everywhere else -- the plugin's hook simply never fires -- and an
|
|
567
|
+
operator who is not told here finds out as an empty workflow list an hour
|
|
568
|
+
later, with nothing in between naming the cause.
|
|
569
|
+
"""
|
|
570
|
+
# The TASK sentinel marks where a rollout's task goes in the argv. There is
|
|
571
|
+
# no rollout here, so it is filled with the warm-up prompt: an argv still
|
|
572
|
+
# holding the sentinel would be passed to the binary as the literal string.
|
|
573
|
+
argv = [WARMUP_PROMPT if a is TASK else str(a) for a in command]
|
|
574
|
+
warm_dir = os.path.join(artifacts_dir, "_describe")
|
|
575
|
+
|
|
576
|
+
snapshot = execution_snapshot
|
|
577
|
+
snapshot_child: ExecutionChild | None = None
|
|
578
|
+
try:
|
|
579
|
+
env = None
|
|
580
|
+
cwd = None
|
|
581
|
+
if snapshot is not None:
|
|
582
|
+
# Registration must describe the same scaffold the rollouts use.
|
|
583
|
+
# Letting this one preflight read ambient HOME would register a
|
|
584
|
+
# prompt/tool identity that no trainable child actually ran.
|
|
585
|
+
try:
|
|
586
|
+
snapshot_child = snapshot.prepare_child("describe")
|
|
587
|
+
except Exception as exc:
|
|
588
|
+
raise ConfigurationError(
|
|
589
|
+
"execution snapshot could not prepare the discovery child: "
|
|
590
|
+
f"{type(exc).__name__}"
|
|
591
|
+
) from exc
|
|
592
|
+
env = dict(os.environ)
|
|
593
|
+
cwd = warm_dir
|
|
594
|
+
try:
|
|
595
|
+
rollout_environment = dict(snapshot_child.environment)
|
|
596
|
+
discovery_environment = dict(
|
|
597
|
+
snapshot_child.discovery_environment
|
|
598
|
+
)
|
|
599
|
+
except Exception as exc:
|
|
600
|
+
raise ConfigurationError(
|
|
601
|
+
f"{snapshot.adapter_name} execution child environments "
|
|
602
|
+
"must be string mappings"
|
|
603
|
+
) from exc
|
|
604
|
+
if any(
|
|
605
|
+
not isinstance(name, str) or not isinstance(value, str)
|
|
606
|
+
for mapping in (rollout_environment, discovery_environment)
|
|
607
|
+
for name, value in mapping.items()
|
|
608
|
+
):
|
|
609
|
+
raise ConfigurationError(
|
|
610
|
+
f"{snapshot.adapter_name} execution child environments "
|
|
611
|
+
"must contain only string names and values"
|
|
612
|
+
)
|
|
613
|
+
if any(
|
|
614
|
+
rollout_environment.get(name) != value
|
|
615
|
+
for name, value in discovery_environment.items()
|
|
616
|
+
):
|
|
617
|
+
raise ConfigurationError(
|
|
618
|
+
f"{snapshot.adapter_name} discovery environment must be a "
|
|
619
|
+
"non-secret subset of its rollout environment"
|
|
620
|
+
)
|
|
621
|
+
else:
|
|
622
|
+
discovery_environment = None
|
|
623
|
+
described = warm_up_command(
|
|
624
|
+
argv, trajectory_dir=warm_dir, agent_name=agent_id, env=env, cwd=cwd,
|
|
625
|
+
endpoint_env=endpoint_env,
|
|
626
|
+
isolation_environment=(
|
|
627
|
+
discovery_environment
|
|
628
|
+
),
|
|
629
|
+
)
|
|
630
|
+
finally:
|
|
631
|
+
if snapshot_child is not None:
|
|
632
|
+
lifecycle_failure: Exception | None = None
|
|
633
|
+
try:
|
|
634
|
+
assert snapshot is not None
|
|
635
|
+
snapshot.attest_child(snapshot_child)
|
|
636
|
+
except Exception as exc:
|
|
637
|
+
lifecycle_failure = exc
|
|
638
|
+
try:
|
|
639
|
+
snapshot.release_child(snapshot_child)
|
|
640
|
+
except Exception as exc:
|
|
641
|
+
lifecycle_failure = lifecycle_failure or exc
|
|
642
|
+
if lifecycle_failure is not None:
|
|
643
|
+
raise ConfigurationError(
|
|
644
|
+
"execution snapshot lifecycle failed during command-agent "
|
|
645
|
+
f"discovery ({type(lifecycle_failure).__name__}); "
|
|
646
|
+
"registration was refused"
|
|
647
|
+
) from lifecycle_failure
|
|
648
|
+
if described is not None:
|
|
649
|
+
n = described.tool_definitions
|
|
650
|
+
print(
|
|
651
|
+
f"[percepteye] described {agent_id} from the host: "
|
|
652
|
+
f"{len(n) if n is not None else 'no'} tool(s), "
|
|
653
|
+
f"{len(described.system_prompt or '')} chars of system prompt"
|
|
654
|
+
# WHICH SOURCE answered, named rather than inferred. The two carry
|
|
655
|
+
# different amounts -- the wire drops read-only annotations and
|
|
656
|
+
# sub-agents -- so an operator reading a thin catalogue needs to
|
|
657
|
+
# know whether that is the agent or the vantage point.
|
|
658
|
+
f" (via {'the wire' if described.source_type == 'host_wire' else 'the host plugin'})."
|
|
659
|
+
)
|
|
660
|
+
return described
|
|
661
|
+
|
|
662
|
+
warnings.warn(
|
|
663
|
+
f"Could not describe {agent_id!r}: the agent ran, its host wrote no "
|
|
664
|
+
f"description, and no completion request it made declared any tools. "
|
|
665
|
+
f"This agent will register as introspection_state=unreadable, no "
|
|
666
|
+
f"workflows will be generated for it, and every phase downstream of "
|
|
667
|
+
f"that (reward suites, graded rollouts, training) has nothing to work "
|
|
668
|
+
f"from.\n"
|
|
669
|
+
f"Two things to check, in this order:\n"
|
|
670
|
+
f" 1. Does the agent reach its provider from an env var? The warm-up "
|
|
671
|
+
f"points OPENAI_BASE_URL / OPENAI_API_BASE / ANTHROPIC_BASE_URL at a "
|
|
672
|
+
f"local endpoint and reads the tool catalogue off the request. An "
|
|
673
|
+
f"agent that resolves its endpoint some other way never arrives, and "
|
|
674
|
+
f"is also not routable for rollouts -- pass endpoint_env= to serve().\n"
|
|
675
|
+
f" 2. Does it declare any tools at all? A CLI agent whose extensions "
|
|
676
|
+
f"are configured per-profile can start with an empty catalogue; goose, "
|
|
677
|
+
f"for one, sends zero tools unless an extension is enabled.\n"
|
|
678
|
+
f"For OpenClaw specifically, the usual cause is the conversation-hook "
|
|
679
|
+
f"opt-in: the host silently refuses to register `llm_input` for a "
|
|
680
|
+
f"non-bundled plugin unless the config sets\n"
|
|
681
|
+
f" plugins.entries.agent-flywheel.hooks."
|
|
682
|
+
f"allowConversationAccess = true",
|
|
683
|
+
RuntimeWarning,
|
|
684
|
+
stacklevel=3,
|
|
685
|
+
)
|
|
686
|
+
return None
|
|
687
|
+
|
|
688
|
+
|
|
689
|
+
def serve(
|
|
690
|
+
entrypoint: Any,
|
|
691
|
+
*,
|
|
692
|
+
agent_id: str,
|
|
693
|
+
input_shape: InputShape,
|
|
694
|
+
mode: FlywheelMode | None = None,
|
|
695
|
+
transport: Literal["attach"] = "attach",
|
|
696
|
+
control_plane_url: str | None = None,
|
|
697
|
+
api_key: str | None = None,
|
|
698
|
+
artifacts_dir: str | None = None,
|
|
699
|
+
isolation: Literal["subprocess", "inprocess"] = "subprocess",
|
|
700
|
+
concurrency: int = 1,
|
|
701
|
+
rollout_timeout_s: float = 900.0,
|
|
702
|
+
#: How this agent's CLI spells its completion-gate option, e.g.
|
|
703
|
+
#: ``"--autonomous-gate"``. Required before any gate is passed: the flag is
|
|
704
|
+
#: a fact about the binary and cannot be guessed, and handing an unknown
|
|
705
|
+
#: option to an arbitrary agent would break it. The gate COMMAND itself is
|
|
706
|
+
#: a property of the task and arrives per rollout. Leave unset for an agent
|
|
707
|
+
#: with no gate concept — it then behaves exactly as before.
|
|
708
|
+
gate_flag: str | None = None,
|
|
709
|
+
#: The environment variables THIS agent reads its endpoint from, when they
|
|
710
|
+
#: are not the OpenAI/Anthropic names set by default. Declared, never
|
|
711
|
+
#: sniffed: an agent that reads neither default looks identical from here
|
|
712
|
+
#: to one that is correctly routed, right up until the completions arrive
|
|
713
|
+
#: at the live provider instead of the gateway.
|
|
714
|
+
endpoint_env: tuple[str, str] | None = None,
|
|
715
|
+
command_output_adapter: CommandOutputAdapter | None = None,
|
|
716
|
+
execution_snapshot: ExecutionSnapshot | None = None,
|
|
717
|
+
harness_snapshot: ExecutionSnapshot | None = None,
|
|
718
|
+
attested_execution_components: Mapping[str, str] | None = None,
|
|
719
|
+
reports_tool_calls: bool = False,
|
|
720
|
+
method: str | None = None,
|
|
721
|
+
adapters: list[str] | str | None = "auto",
|
|
722
|
+
introspect: bool = True,
|
|
723
|
+
max_rollouts: int | None = None,
|
|
724
|
+
on_event: Callable[[str, dict[str, Any]], None] | None = None,
|
|
725
|
+
) -> int:
|
|
726
|
+
"""Claim rollouts and run them until stopped. Returns rollouts completed.
|
|
727
|
+
|
|
728
|
+
Everything that can be wrong with the configuration is checked here,
|
|
729
|
+
before the first rollout: the entrypoint resolves, the credential exists,
|
|
730
|
+
the control plane answers, and the isolation mode is compatible with the
|
|
731
|
+
entrypoint you passed.
|
|
732
|
+
|
|
733
|
+
Args:
|
|
734
|
+
entrypoint: ``"module:attr"`` (required for subprocess isolation, so a
|
|
735
|
+
child can rebuild the agent) or a constructed object
|
|
736
|
+
(``isolation="inprocess"`` only).
|
|
737
|
+
mode: ``"training"`` to contribute capacity, ``"production"`` to do
|
|
738
|
+
nothing. Omit it -- which is the point -- and ``PERCEPTEYE_AGENT_MODE``
|
|
739
|
+
decides, so the same deployed image serves or does not without an
|
|
740
|
+
edit. Unset means ``"training"``, preserving the behaviour of every
|
|
741
|
+
existing call. ``"rollout"`` is the legacy spelling of
|
|
742
|
+
``"training"``. In production mode this returns 0 before opening
|
|
743
|
+
any connection, and prints one line saying so; it does NOT raise,
|
|
744
|
+
because a switch that crashes the service it was meant to quiet is
|
|
745
|
+
not a switch. An unrecognised value raises
|
|
746
|
+
:class:`ConfigurationError` rather than guessing.
|
|
747
|
+
input_shape: how ``turn_input`` reaches your agent -- ``"text"``,
|
|
748
|
+
``"messages"`` or ``"dict"``. Declared, never guessed.
|
|
749
|
+
adapters: framework integrations, by NAME, bound inside whichever
|
|
750
|
+
process runs your agent. ``"auto"`` (the default) binds every
|
|
751
|
+
built-in whose framework is installed, so a LangChain or LangGraph
|
|
752
|
+
agent reports per-call outcomes without you writing anything.
|
|
753
|
+
Pass ``[]`` to bind none, or ``["langchain"]`` to require one --
|
|
754
|
+
a NAMED adapter that cannot bind raises, because naming it asserts
|
|
755
|
+
that it is how your calls are captured.
|
|
756
|
+
reports_tool_calls: assert that this flywheel reports a per-call
|
|
757
|
+
outcome for every tool call it makes. Without it, the reward
|
|
758
|
+
surface available to your rollouts is materially smaller, because
|
|
759
|
+
the verifiers that grade "did this call land" have nothing to
|
|
760
|
+
read and correctly abstain.
|
|
761
|
+
execution_snapshot: framework-neutral execution-adapter lifecycle.
|
|
762
|
+
The adapter supplies opaque exact component digests, prepares an
|
|
763
|
+
isolated child view, and attests/releases it after every rollout.
|
|
764
|
+
Command and import-spec children receive its environment mapping;
|
|
765
|
+
an in-process object requires an empty mapping because process-wide
|
|
766
|
+
environment mutation is not isolated. An incomplete adapter never
|
|
767
|
+
yields an OPSD-eligible execution identity.
|
|
768
|
+
harness_snapshot: deprecated compatibility alias for
|
|
769
|
+
``execution_snapshot``. Prime-specific construction lives in the
|
|
770
|
+
optional ``adapters.prime_agent`` module; core treats it like any
|
|
771
|
+
other execution adapter.
|
|
772
|
+
command_output_adapter: optional command-specific stdout projection.
|
|
773
|
+
Core never guesses a host vocabulary. An adapter may project final
|
|
774
|
+
text, independently observed model-call count, tool outcomes, and
|
|
775
|
+
explicit exclusion evidence.
|
|
776
|
+
attested_execution_components: framework-neutral opaque evidence from
|
|
777
|
+
an execution adapter, as stable names mapped to lowercase 64-hex
|
|
778
|
+
SHA-256 values. The SDK interprets none of the names: it validates,
|
|
779
|
+
sorts, and folds them into ``agent_fingerprint.execution_sha256``
|
|
780
|
+
with the discovered prompt/tools/sub-agents/model and serve config.
|
|
781
|
+
An adapter supplying a digest is asserting it covers mutable state
|
|
782
|
+
discovery cannot observe; do not put secrets or free-form values in
|
|
783
|
+
this mapping. Without at least one adapter-owned component (or a
|
|
784
|
+
component-providing ``execution_snapshot``/``harness_snapshot``),
|
|
785
|
+
the SDK withholds the
|
|
786
|
+
OPSD digest rather than claiming discovery identifies unobserved
|
|
787
|
+
implementation code.
|
|
788
|
+
max_rollouts: stop after this many rollouts COMPLETE, and return. The
|
|
789
|
+
only clean exit this function has: nothing in this package handles
|
|
790
|
+
SIGINT or SIGTERM, so otherwise it runs until the process is
|
|
791
|
+
killed and any in-flight rollout is orphaned rather than abandoned,
|
|
792
|
+
to be reclaimed by lease expiry. Note the count is completions, not
|
|
793
|
+
successes -- a crash and a graded pass both increment it.
|
|
794
|
+
on_event: ``Callable[[str, dict], None]``, called SYNCHRONOUSLY on
|
|
795
|
+
whichever thread produced the event -- the serving thread for
|
|
796
|
+
``claim_*``, a pool worker for the per-rollout kinds, and the
|
|
797
|
+
heartbeat thread for ``lease_lost``/``heartbeat_failed`` -- so a
|
|
798
|
+
slow callback slows that thread. Exceptions raised by the callback
|
|
799
|
+
are SWALLOWED: it is an observability hook and must not be able to
|
|
800
|
+
change what it observes. Keep it cheap and non-blocking; the
|
|
801
|
+
heartbeat thread renews every in-flight lease. Kinds:
|
|
802
|
+
``claim_empty``, ``claim_failed``,
|
|
803
|
+
``rollout_start``, ``rollout_reported``, ``rollout_crash``,
|
|
804
|
+
``report_failed``, ``artifact_failed``, ``lease_lost``,
|
|
805
|
+
``heartbeat_failed``. Without it a flywheel is silent, and an empty
|
|
806
|
+
queue is indistinguishable from a broken claim route --
|
|
807
|
+
``claim_empty`` versus ``claim_failed`` is what separates them.
|
|
808
|
+
"""
|
|
809
|
+
# ── the deployment switch ────────────────────────────────────────────────
|
|
810
|
+
# Resolved FIRST, before the credential checks and before any socket, so
|
|
811
|
+
# that a production deployment needs neither a control-plane URL nor a key
|
|
812
|
+
# to start clean. Turning training off must not require being configured
|
|
813
|
+
# for it.
|
|
814
|
+
resolved_mode, _mode_source = _resolve_mode_source(mode)
|
|
815
|
+
|
|
816
|
+
# An override has to be announced in BOTH directions. The production
|
|
817
|
+
# branch below prints when code says stop; this prints when code says GO
|
|
818
|
+
# while the environment said stop -- which is the direction an operator
|
|
819
|
+
# actually cares about, and the one that was silent. Flipping the variable
|
|
820
|
+
# and seeing the flywheel keep claiming, with nothing on stdout to explain
|
|
821
|
+
# it, is the same silence this function refuses everywhere else.
|
|
822
|
+
if _mode_source == "argument":
|
|
823
|
+
_env_raw = os.environ.get(ENV_MODE)
|
|
824
|
+
if _env_raw is not None and str(_env_raw).strip() != "":
|
|
825
|
+
try:
|
|
826
|
+
_env_mode = _mode_from(_env_raw, source=f"{ENV_MODE}=")
|
|
827
|
+
except ConfigurationError as exc:
|
|
828
|
+
# Not fatal: the argument is authoritative and this process has
|
|
829
|
+
# been told what to do. But an unreadable switch must not pass
|
|
830
|
+
# unremarked, or the next operator to rely on it is misled.
|
|
831
|
+
warnings.warn(str(exc), RuntimeWarning, stacklevel=2)
|
|
832
|
+
_env_mode = None
|
|
833
|
+
if _env_mode is not None and _env_mode != resolved_mode:
|
|
834
|
+
print(
|
|
835
|
+
f"[percepteye] mode={resolved_mode} — set by mode= in "
|
|
836
|
+
f"code, overriding {ENV_MODE}={_env_mode}. Remove mode= "
|
|
837
|
+
f"from the call to let this deployment decide.",
|
|
838
|
+
flush=True,
|
|
839
|
+
)
|
|
840
|
+
|
|
841
|
+
if resolved_mode == "production":
|
|
842
|
+
# Return, never raise: this is a running service, and a switch that
|
|
843
|
+
# crashes the process it was meant to quiet is not a switch. Return 0
|
|
844
|
+
# because zero rollouts is the literal truth.
|
|
845
|
+
#
|
|
846
|
+
# And never SILENTLY: a no-op serve() that printed nothing is the exact
|
|
847
|
+
# failure this package refuses everywhere else -- the operator believes
|
|
848
|
+
# capacity is being contributed and nothing says otherwise. One line, on
|
|
849
|
+
# the way out, naming what decided it.
|
|
850
|
+
print(
|
|
851
|
+
f"[percepteye] mode=production — not claiming rollouts. "
|
|
852
|
+
f"(set {ENV_MODE}=training to contribute capacity"
|
|
853
|
+
+ (", or remove mode= from this call"
|
|
854
|
+
if _mode_source == "argument" else "")
|
|
855
|
+
+ ")",
|
|
856
|
+
flush=True,
|
|
857
|
+
)
|
|
858
|
+
return 0
|
|
859
|
+
if transport != "attach":
|
|
860
|
+
raise ModeNotAvailable(
|
|
861
|
+
f"transport={transport!r} is not available in contract {CONTRACT_VERSION}."
|
|
862
|
+
)
|
|
863
|
+
|
|
864
|
+
url = control_plane_url or os.environ.get(ENV_CONTROL_PLANE_URL)
|
|
865
|
+
if not url:
|
|
866
|
+
raise ConfigurationError(
|
|
867
|
+
f"control_plane_url is required (or set {ENV_CONTROL_PLANE_URL})"
|
|
868
|
+
)
|
|
869
|
+
key = api_key or os.environ.get(ENV_API_KEY)
|
|
870
|
+
if not key:
|
|
871
|
+
raise ConfigurationError(f"an API key is required (or set {ENV_API_KEY})")
|
|
872
|
+
|
|
873
|
+
adir = artifacts_dir or os.path.join(os.getcwd(), ".percepteye-artifacts")
|
|
874
|
+
os.makedirs(adir, exist_ok=True)
|
|
875
|
+
|
|
876
|
+
if execution_snapshot is not None and harness_snapshot is not None:
|
|
877
|
+
raise ConfigurationError(
|
|
878
|
+
"pass execution_snapshot or the deprecated harness_snapshot alias, not both"
|
|
879
|
+
)
|
|
880
|
+
if harness_snapshot is not None:
|
|
881
|
+
warnings.warn(
|
|
882
|
+
"serve(harness_snapshot=...) is deprecated; use execution_snapshot=...",
|
|
883
|
+
DeprecationWarning,
|
|
884
|
+
stacklevel=2,
|
|
885
|
+
)
|
|
886
|
+
resolved_execution_snapshot = (
|
|
887
|
+
execution_snapshot if execution_snapshot is not None else harness_snapshot
|
|
888
|
+
)
|
|
889
|
+
|
|
890
|
+
from ._http import ControlPlaneClient
|
|
891
|
+
|
|
892
|
+
client = ControlPlaneClient(url, key)
|
|
893
|
+
tr = AttachTransport(client, agent_id)
|
|
894
|
+
|
|
895
|
+
# Fail at startup, loudly. A health check nobody calls is not a health
|
|
896
|
+
# check; an earlier SDK shipped one with zero call sites.
|
|
897
|
+
health = tr.health()
|
|
898
|
+
served = health.get("contract_versions_supported") or [health.get("contract_version")]
|
|
899
|
+
if CONTRACT_VERSION not in [v for v in served if v]:
|
|
900
|
+
raise ConfigurationError(
|
|
901
|
+
f"control plane speaks {served}, this SDK speaks {CONTRACT_VERSION}. "
|
|
902
|
+
f"Upgrade agent-flywheel."
|
|
903
|
+
)
|
|
904
|
+
|
|
905
|
+
# Provider configuration we cannot point at the gateway. Reported here, in
|
|
906
|
+
# front of the person who set it, rather than a run later as a
|
|
907
|
+
# training-grade exclusion nobody can explain. A warning and not a refusal:
|
|
908
|
+
# a name on this list is evidence, not proof -- the agent may not use that
|
|
909
|
+
# provider at all -- and the control plane's reported-vs-served call count
|
|
910
|
+
# is the check that actually decides.
|
|
911
|
+
stripped = unroutable_providers_present()
|
|
912
|
+
if stripped:
|
|
913
|
+
warnings.warn(
|
|
914
|
+
f"{', '.join(stripped)} set in this environment. The PerceptEye "
|
|
915
|
+
f"gateway speaks the OpenAI completions API, so these cannot be "
|
|
916
|
+
f"routed through it, and they are REMOVED from each rollout's "
|
|
917
|
+
f"child process. An agent built on one of those providers will "
|
|
918
|
+
f"raise rather than sample from the live provider unobserved. "
|
|
919
|
+
f"Use an OpenAI-compatible client to have your completions "
|
|
920
|
+
f"captured.",
|
|
921
|
+
RuntimeWarning,
|
|
922
|
+
stacklevel=2,
|
|
923
|
+
)
|
|
924
|
+
|
|
925
|
+
runner = RolloutRunner(
|
|
926
|
+
entrypoint,
|
|
927
|
+
transport=tr,
|
|
928
|
+
input_shape=input_shape,
|
|
929
|
+
artifacts_dir=adir,
|
|
930
|
+
isolation=isolation,
|
|
931
|
+
concurrency=concurrency,
|
|
932
|
+
rollout_timeout_s=rollout_timeout_s,
|
|
933
|
+
gate_flag=gate_flag,
|
|
934
|
+
endpoint_env=endpoint_env,
|
|
935
|
+
command_output_adapter=command_output_adapter,
|
|
936
|
+
execution_snapshot=resolved_execution_snapshot,
|
|
937
|
+
method=method,
|
|
938
|
+
adapters=adapters,
|
|
939
|
+
on_event=on_event,
|
|
940
|
+
reports_tool_calls=reports_tool_calls,
|
|
941
|
+
attested_execution_components=attested_execution_components,
|
|
942
|
+
)
|
|
943
|
+
|
|
944
|
+
# A command entrypoint has no resolved object -- ``resolved`` is None by
|
|
945
|
+
# construction (it is already its own process). Reading ``.name`` off it
|
|
946
|
+
# raised AttributeError here, so every non-Python agent failed at
|
|
947
|
+
# registration, before its first rollout. No test caught it because the
|
|
948
|
+
# tests build a RolloutRunner directly and never go through serve().
|
|
949
|
+
entrypoint_label = (
|
|
950
|
+
runner.resolved.name if runner.resolved is not None
|
|
951
|
+
# str() each part: a TASK sentinel is not a string, and joining it
|
|
952
|
+
# raised TypeError here — the same class of bug the line above records
|
|
953
|
+
# having shipped once, where reading .name off a command entrypoint
|
|
954
|
+
# failed every non-Python agent at registration.
|
|
955
|
+
else " ".join(str(a) for a in (getattr(runner, "_command", None) or []))
|
|
956
|
+
or "command"
|
|
957
|
+
)
|
|
958
|
+
|
|
959
|
+
payload: dict[str, Any] = {
|
|
960
|
+
"agent_id": agent_id,
|
|
961
|
+
# The WIRE value, which is not the deployment mode and never was.
|
|
962
|
+
# `attach()` sends "production" on this same field; a donor sends
|
|
963
|
+
# "rollout", which names the protocol it speaks and is what intake has
|
|
964
|
+
# always received here. Reaching this line at all means
|
|
965
|
+
# resolved_mode == "training".
|
|
966
|
+
"mode": "rollout",
|
|
967
|
+
"sdk": f"agent-flywheel-python/{__version__}",
|
|
968
|
+
"contract_version": CONTRACT_VERSION,
|
|
969
|
+
"reports_tool_calls": bool(reports_tool_calls),
|
|
970
|
+
"isolation": isolation,
|
|
971
|
+
"concurrency": concurrency,
|
|
972
|
+
"input_shape": input_shape,
|
|
973
|
+
"entrypoint": entrypoint_label,
|
|
974
|
+
}
|
|
975
|
+
|
|
976
|
+
# ── agent discovery (ADP), from inside the process ──────────────────────
|
|
977
|
+
# ON by default, with two ways to refuse: ``introspect=False`` here, or
|
|
978
|
+
# ``PERCEPTEYE_INTROSPECT=0`` in the environment, for an operator who
|
|
979
|
+
# cannot edit the call site. Both are documented prominently in the
|
|
980
|
+
# README, because this ships your system prompt and tool catalogue to the
|
|
981
|
+
# control plane and a default that surprises someone is worse than no
|
|
982
|
+
# default at all.
|
|
983
|
+
#
|
|
984
|
+
# ``discovered_agent: None`` is sent DELIBERATELY when we could not look.
|
|
985
|
+
# Omitting the key would make "not introspected" indistinguishable from
|
|
986
|
+
# "introspected, found nothing" -- the same absence-as-zero collapse the
|
|
987
|
+
# tool-outcome contract exists to prevent, one layer up.
|
|
988
|
+
discovery_enabled = introspect and _env_true("PERCEPTEYE_INTROSPECT", default=True)
|
|
989
|
+
described = None
|
|
990
|
+
if discovery_enabled:
|
|
991
|
+
described = introspect_agent(
|
|
992
|
+
runner.resolved.fn if runner.resolved is not None else None
|
|
993
|
+
)
|
|
994
|
+
# A COMMAND agent has no object to reflect on, so the line above always
|
|
995
|
+
# returns None for one and the agent registers `unreadable` forever --
|
|
996
|
+
# which blocks workflow generation, and with it every downstream phase.
|
|
997
|
+
# Ask the HOST instead: run the agent once, let its own plugin observe
|
|
998
|
+
# the assembled prompt and the tool list it sends to the model, and
|
|
999
|
+
# read that. See host_description.py for why this cannot wait for a
|
|
1000
|
+
# rollout (the launch gate and the first rollout are mutually
|
|
1001
|
+
# blocking).
|
|
1002
|
+
if described is None and getattr(runner, "_command", None):
|
|
1003
|
+
described = _describe_via_host(
|
|
1004
|
+
runner._command, artifacts_dir=adir, agent_id=agent_id,
|
|
1005
|
+
execution_snapshot=runner.execution_snapshot,
|
|
1006
|
+
endpoint_env=endpoint_env,
|
|
1007
|
+
)
|
|
1008
|
+
payload["discovered_agent"] = described.to_wire() if described else None
|
|
1009
|
+
|
|
1010
|
+
# The rollout fingerprint is authored here, after discovery, from the same
|
|
1011
|
+
# framework-neutral facts registration sends plus exact pre-projection
|
|
1012
|
+
# discovery evidence and any adapter-attested opaque components. A disabled,
|
|
1013
|
+
# unreadable, or lossy description deliberately yields no digest, so an OPSD
|
|
1014
|
+
# consumer cannot mistake an unknown scaffold for an exact identity.
|
|
1015
|
+
runner.configure_execution_identity(
|
|
1016
|
+
described,
|
|
1017
|
+
discovery_enabled=discovery_enabled,
|
|
1018
|
+
sdk=payload["sdk"],
|
|
1019
|
+
contract_version=CONTRACT_VERSION,
|
|
1020
|
+
)
|
|
1021
|
+
|
|
1022
|
+
# Neutral worker evidence, not agent logic and not a model identity. The
|
|
1023
|
+
# control plane uses this authenticated registration fact to select the
|
|
1024
|
+
# exact rollout worker whose end-to-end latency was qualified. It still
|
|
1025
|
+
# verifies every returned trajectory carries the same digest; registration
|
|
1026
|
+
# is the pre-dispatch identity, not a substitute for observed evidence.
|
|
1027
|
+
if runner.execution_sha256 is not None:
|
|
1028
|
+
payload["agent_execution_sha256"] = runner.execution_sha256
|
|
1029
|
+
|
|
1030
|
+
tr.register(payload)
|
|
1031
|
+
|
|
1032
|
+
# ── the last mile ───────────────────────────────────────────────────────
|
|
1033
|
+
# Training produces a better model; this is where the agent finds out.
|
|
1034
|
+
# The registration response used to be discarded and nothing ever asked
|
|
1035
|
+
# what to run, so a trained adapter reached the customer's process only
|
|
1036
|
+
# by a human editing their config.
|
|
1037
|
+
#
|
|
1038
|
+
# Resolved ONCE at startup rather than polled: a policy that changes
|
|
1039
|
+
# under a running agent is the same confound the control plane refuses on
|
|
1040
|
+
# its own training runs. A change lands on the next restart, and the
|
|
1041
|
+
# dashboard says so.
|
|
1042
|
+
# getattr, not a direct call: ``policy_current`` is not on the
|
|
1043
|
+
# FlywheelTransport Protocol, so a third-party or older transport need
|
|
1044
|
+
# not implement it. The last mile must not break an agent whose
|
|
1045
|
+
# transport predates it — the same fail-open the method itself keeps.
|
|
1046
|
+
_ask = getattr(tr, "policy_current", None)
|
|
1047
|
+
policy = _ask() if callable(_ask) else {}
|
|
1048
|
+
if policy:
|
|
1049
|
+
# INFORMATIONAL ONLY, deliberately. A serve rollout samples through the
|
|
1050
|
+
# per-rollout GATEWAY url — that is what attributes and captures the
|
|
1051
|
+
# completion, and the gateway already rewrites the model field to the
|
|
1052
|
+
# policy under test. Applying the endpoint here would bypass capture
|
|
1053
|
+
# and produce a run with nothing to train on. Production agents call
|
|
1054
|
+
# apply_current_policy() instead.
|
|
1055
|
+
_ep = policy.get("endpoint") if isinstance(policy.get("endpoint"), dict) else {}
|
|
1056
|
+
print(
|
|
1057
|
+
f"[percepteye] policy: {policy.get('arm', 'base')} — "
|
|
1058
|
+
f"{policy.get('reason', '')}"
|
|
1059
|
+
+ (f" → {_ep.get('model')} at {_ep.get('base_url')}"
|
|
1060
|
+
if _ep.get("base_url") else ""),
|
|
1061
|
+
flush=True,
|
|
1062
|
+
)
|
|
1063
|
+
|
|
1064
|
+
try:
|
|
1065
|
+
return runner.serve_forever(max_rollouts=max_rollouts)
|
|
1066
|
+
finally:
|
|
1067
|
+
client.close()
|